islkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
islkit/features.py ADDED
@@ -0,0 +1,323 @@
1
+ """MediaPipe landmarks to feature vectors.
2
+
3
+ Backend-neutral: the RawFrame dataclass is the contract. Legacy
4
+ mp.solutions.holistic and the Tasks API landmarkers both adapt into it,
5
+ so nothing downstream changes if you swap backends.
6
+
7
+ Feature vector: 183 floats per frame, 352 with velocity.
8
+ hand-local shapes 2 x 21 x 3 = 126
9
+ wrist positions 2 x 3 = 6
10
+ upper-body pose 11 x 3 = 33
11
+ non-manual scalars = 4
12
+ validity mask = 14
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import hashlib
18
+ from dataclasses import dataclass
19
+
20
+ import numpy as np
21
+
22
+ # --------------------------------------------------------------------------
23
+ # Landmark index selections
24
+ # --------------------------------------------------------------------------
25
+
26
+ # BlazePose 33-point model. Legs discarded — ISL signing space is torso-up.
27
+ POSE_SEL = [
28
+ 0, # nose
29
+ 7,
30
+ 8, # ears — head orientation
31
+ 11,
32
+ 12, # shoulders — the body frame
33
+ 13,
34
+ 14, # elbows
35
+ 15,
36
+ 16, # wrists
37
+ 23,
38
+ 24, # hips — torso scale reference
39
+ ]
40
+ L_SHOULDER, R_SHOULDER = 11, 12
41
+
42
+ # Canonical 468-point face mesh. Used only to derive 4 scalars.
43
+ FACE_IDX = {
44
+ "brow_l": 105,
45
+ "brow_r": 334,
46
+ "brow_inner_l": 55,
47
+ "brow_inner_r": 285,
48
+ "eye_upper_l": 159,
49
+ "eye_lower_l": 145,
50
+ "eye_upper_r": 386,
51
+ "eye_lower_r": 374,
52
+ "lip_upper": 13,
53
+ "lip_lower": 14,
54
+ "mouth_l": 61,
55
+ "mouth_r": 291,
56
+ "forehead": 10,
57
+ "chin": 152,
58
+ "cheek_l": 234,
59
+ "cheek_r": 454,
60
+ }
61
+
62
+ N_HAND, N_POSE_SEL = 21, len(POSE_SEL)
63
+ DIM_GEOM = 2 * N_HAND * 3 + 2 * 3 + N_POSE_SEL * 3 + 4 # 169
64
+ DIM_MASK = 2 + N_POSE_SEL + 1 # 14
65
+ DIM_FRAME = DIM_GEOM + DIM_MASK # 183
66
+
67
+
68
+ # --------------------------------------------------------------------------
69
+ # Backend-neutral frame contract
70
+ # --------------------------------------------------------------------------
71
+
72
+
73
+ @dataclass
74
+ class RawFrame:
75
+ """One frame of raw landmarks. None means the tracker lost that part."""
76
+
77
+ pose: np.ndarray | None # (33, 4) x, y, z, visibility
78
+ face: np.ndarray | None # (468, 3)
79
+ hand_left: np.ndarray | None # (21, 3)
80
+ hand_right: np.ndarray | None # (21, 3)
81
+
82
+
83
+ def _to_array(landmark_list, with_visibility=False):
84
+ if landmark_list is None:
85
+ return None
86
+ lms = landmark_list.landmark
87
+ if with_visibility:
88
+ return np.array([[p.x, p.y, p.z, p.visibility] for p in lms], np.float32)
89
+ return np.array([[p.x, p.y, p.z] for p in lms], np.float32)
90
+
91
+
92
+ def from_holistic(results) -> RawFrame:
93
+ """Adapter for legacy mp.solutions.holistic results."""
94
+ return RawFrame(
95
+ pose=_to_array(results.pose_landmarks, with_visibility=True),
96
+ face=_to_array(results.face_landmarks),
97
+ hand_left=_to_array(results.left_hand_landmarks),
98
+ hand_right=_to_array(results.right_hand_landmarks),
99
+ )
100
+
101
+
102
+ def from_tasks(pose_result, hand_result, face_result) -> RawFrame:
103
+ """Adapter for the Tasks API: PoseLandmarker + HandLandmarker + FaceLandmarker.
104
+
105
+ HandLandmarker returns a list of hands plus a parallel handedness list.
106
+ We only use handedness to fill the two slots; canonicalisation below does
107
+ not trust it.
108
+ """
109
+
110
+ def pack(lms, vis=False):
111
+ if not lms:
112
+ return None
113
+ if vis:
114
+ return np.array(
115
+ [[p.x, p.y, p.z, getattr(p, "visibility", 1.0)] for p in lms], np.float32
116
+ )
117
+ return np.array([[p.x, p.y, p.z] for p in lms], np.float32)
118
+
119
+ left = right = None
120
+ if hand_result and hand_result.hand_landmarks:
121
+ for lms, hd in zip(hand_result.hand_landmarks, hand_result.handedness, strict=False):
122
+ arr = pack(lms)
123
+ if hd[0].category_name.lower().startswith("l"):
124
+ left = arr
125
+ else:
126
+ right = arr
127
+
128
+ pose = face = None
129
+ if pose_result and pose_result.pose_landmarks:
130
+ pose = pack(pose_result.pose_landmarks[0], vis=True)
131
+ if face_result and face_result.face_landmarks:
132
+ face = pack(face_result.face_landmarks[0])
133
+
134
+ return RawFrame(pose=pose, face=face, hand_left=left, hand_right=right)
135
+
136
+
137
+ # --------------------------------------------------------------------------
138
+ # Feature encoding
139
+ # --------------------------------------------------------------------------
140
+
141
+
142
+ def _hand_local(hand: np.ndarray) -> np.ndarray:
143
+ """Wrist-origin, hand-size-scaled. Handshape independent of position."""
144
+ wrist = hand[0]
145
+ centred = hand - wrist
146
+ scale = np.linalg.norm(hand[9] - wrist) # wrist to middle-finger MCP
147
+ if scale < 1e-6:
148
+ return np.zeros_like(centred)
149
+ return centred / scale
150
+
151
+
152
+ def _nonmanual(face: np.ndarray, height: float, width: float) -> np.ndarray:
153
+ """Four scalars replacing 1404 raw face coordinates."""
154
+ f = FACE_IDX
155
+ brow_mid = (face[f["brow_l"]] + face[f["brow_r"]]) / 2
156
+ eye_mid = (face[f["eye_upper_l"]] + face[f["eye_upper_r"]]) / 2
157
+ return np.array(
158
+ [
159
+ np.linalg.norm(brow_mid - eye_mid) / height, # raise
160
+ np.linalg.norm(face[f["brow_inner_l"]] - face[f["brow_inner_r"]]) / width,
161
+ np.linalg.norm(face[f["lip_upper"]] - face[f["lip_lower"]]) / height,
162
+ np.linalg.norm(face[f["mouth_l"]] - face[f["mouth_r"]]) / width,
163
+ ],
164
+ np.float32,
165
+ )
166
+
167
+
168
+ def encode_frame(raw: RawFrame, dominant: str = "right") -> np.ndarray:
169
+ """RawFrame -> (183,) float32. All-zero mask means an unusable frame."""
170
+ out = np.zeros(DIM_FRAME, np.float32)
171
+
172
+ # Body frame requires shoulders. Without pose there is no usable geometry.
173
+ if raw.pose is None:
174
+ return out
175
+ pose_xyz = raw.pose[:, :3]
176
+ origin = (pose_xyz[L_SHOULDER] + pose_xyz[R_SHOULDER]) / 2
177
+ scale = float(np.linalg.norm(pose_xyz[L_SHOULDER] - pose_xyz[R_SHOULDER]))
178
+ if scale < 1e-6:
179
+ return out
180
+
181
+ # --- Canonicalise hand slots by geometry, never by the handedness label.
182
+ # Slot 0 is the dominant hand: the one nearer the dominant-side shoulder.
183
+ dom_shoulder = pose_xyz[R_SHOULDER if dominant == "right" else L_SHOULDER]
184
+ present = [
185
+ (h, np.linalg.norm(h[0] - dom_shoulder))
186
+ for h in (raw.hand_left, raw.hand_right)
187
+ if h is not None
188
+ ]
189
+ present.sort(key=lambda t: t[1])
190
+ slots: list[np.ndarray | None] = [None, None]
191
+ for i, (hand, _) in enumerate(present[:2]):
192
+ slots[i] = hand
193
+
194
+ i = 0
195
+ for s, hand in enumerate(slots):
196
+ if hand is not None:
197
+ out[i : i + 63] = _hand_local(hand).ravel()
198
+ out[126 + s * 3 : 126 + s * 3 + 3] = (hand[0] - origin) / scale
199
+ out[DIM_GEOM + s] = 1.0 # hand-present flag
200
+ i += 63
201
+
202
+ # --- Upper-body pose in the body frame
203
+ out[132 : 132 + N_POSE_SEL * 3] = ((pose_xyz[POSE_SEL] - origin) / scale).ravel()
204
+ out[DIM_GEOM + 2 : DIM_GEOM + 2 + N_POSE_SEL] = raw.pose[POSE_SEL, 3] # visibility
205
+
206
+ # --- Non-manual scalars
207
+ if raw.face is not None:
208
+ f = raw.face
209
+ h = float(np.linalg.norm(f[FACE_IDX["forehead"]] - f[FACE_IDX["chin"]]))
210
+ w = float(np.linalg.norm(f[FACE_IDX["cheek_l"]] - f[FACE_IDX["cheek_r"]]))
211
+ if h > 1e-6 and w > 1e-6:
212
+ out[165:169] = _nonmanual(f, h, w)
213
+ out[DIM_FRAME - 1] = 1.0 # face-present flag
214
+
215
+ return out
216
+
217
+
218
+ def encode_clip(
219
+ frames: list[RawFrame], T: int = 48, dominant: str = "right", with_velocity: bool = True
220
+ ) -> np.ndarray:
221
+ """Variable-length clip -> (T, 352). Resampled, not padded."""
222
+ if not frames:
223
+ return np.zeros(
224
+ (T, DIM_FRAME + DIM_GEOM if with_velocity else DIM_FRAME),
225
+ np.float32,
226
+ )
227
+
228
+ seq = np.stack([encode_frame(f, dominant) for f in frames])
229
+
230
+ # Linear resample to fixed T. Duration becomes an augmentation axis,
231
+ # not a confound the model has to learn around.
232
+ src = np.linspace(0, len(seq) - 1, len(seq))
233
+ dst = np.linspace(0, len(seq) - 1, T)
234
+ seq = np.stack([np.interp(dst, src, seq[:, d]) for d in range(seq.shape[1])], axis=1).astype(
235
+ np.float32
236
+ )
237
+
238
+ if not with_velocity:
239
+ return seq
240
+ vel = np.zeros((T, DIM_GEOM), np.float32)
241
+ vel[1:] = seq[1:, :DIM_GEOM] - seq[:-1, :DIM_GEOM]
242
+ return np.concatenate([seq, vel], axis=1)
243
+
244
+
245
+ # --------------------------------------------------------------------------
246
+ # Train/serve identity
247
+ # --------------------------------------------------------------------------
248
+
249
+ # A fixed probe clip, encoded and hashed. The point is to detect a change in
250
+ # what the encoder DOES, which no config field can catch: change how the body
251
+ # frame is scaled and every number moves while T, dominant and with_velocity
252
+ # stay identical. That failure is silent and confident, and this project has
253
+ # been bitten by its shape three times (the mirror, the visibility skew, the
254
+ # face block length).
255
+ #
256
+ # Hashing the source instead would change on every docstring edit, and an alarm
257
+ # that fires on cosmetic changes is one people learn to ignore.
258
+ FINGERPRINT_T = 16
259
+ FINGERPRINT_DECIMALS = 6 # absorbs float differences between x86 and aarch64
260
+ _FINGERPRINT_SEED = 20260906
261
+
262
+
263
+ def _fingerprint_clip(n_frames: int = 24) -> list[RawFrame]:
264
+ """Deterministic nonsense with the right shape. Never touches a camera.
265
+
266
+ Every third frame drops the left hand so the validity mask and the
267
+ geometric slot assignment are both exercised, not just the dense path.
268
+ """
269
+ rng = np.random.default_rng(_FINGERPRINT_SEED)
270
+ frames = []
271
+ for i in range(n_frames):
272
+ pose = np.zeros((33, 4), np.float32)
273
+ pose[:, :3] = rng.random((33, 3), dtype=np.float32)
274
+ pose[:, 3] = 1.0
275
+ frames.append(
276
+ RawFrame(
277
+ pose=pose,
278
+ face=rng.random((468, 3), dtype=np.float32), # FACE_IDX's canonical mesh
279
+ hand_left=rng.random((21, 3), dtype=np.float32) if i % 3 else None,
280
+ hand_right=rng.random((21, 3), dtype=np.float32),
281
+ )
282
+ )
283
+ return frames
284
+
285
+
286
+ # The hash above is exact, and exactness is the wrong test across machines.
287
+ # Measured 2026-09-07, same features.py and same numpy 1.26.4 on both ends:
288
+ # the probe clip differs between macOS/arm64 and the board's linux/aarch64 by
289
+ # up to 2.384e-07 in 1165 of 5632 values — float32 epsilon from libm and SIMD
290
+ # ordering, not a change in what the encoder does. Rounding cannot fix that
291
+ # reliably: a value sitting within the noise of a rounding boundary still flips,
292
+ # and at 4 dp roughly 27 of 5632 values are expected to sit there.
293
+ #
294
+ # So the guard compares numerically, with a tolerance two orders above the
295
+ # observed noise and orders below any real encoder change (the visibility skew
296
+ # and the mirror both moved values by O(0.1)). The hash stays, for logs and for
297
+ # reading a checkpoint's identity at a glance.
298
+ FINGERPRINT_ATOL = 1e-5
299
+
300
+
301
+ def encoder_signature() -> np.ndarray:
302
+ """The probe clip itself, for a comparison that tolerates float noise."""
303
+ return encode_clip(
304
+ _fingerprint_clip(), T=FINGERPRINT_T, dominant="right", with_velocity=True
305
+ )
306
+
307
+
308
+ def signatures_match(a, b, atol: float = FINGERPRINT_ATOL) -> bool:
309
+ """Whether two encoder signatures describe the same encoder behaviour."""
310
+ a, b = np.asarray(a), np.asarray(b)
311
+ return a.shape == b.shape and bool(np.allclose(a, b, atol=atol, rtol=0.0))
312
+
313
+
314
+ def encoder_fingerprint() -> str:
315
+ """16 hex chars identifying this encoder's behaviour.
316
+
317
+ Written into every checkpoint at training time and checked when one is
318
+ loaded, so a model can never be served by an encoder it was not fitted on.
319
+ """
320
+ clip = encode_clip(_fingerprint_clip(), T=FINGERPRINT_T, dominant="right", with_velocity=True)
321
+ # +0.0 normalises -0.0, which hashes differently from 0.0 for no useful reason.
322
+ quantised = np.round(clip.astype(np.float64), FINGERPRINT_DECIMALS) + 0.0
323
+ return hashlib.sha256(quantised.tobytes()).hexdigest()[:16]