islkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- islkit/__init__.py +88 -0
- islkit/adapters.py +237 -0
- islkit/baseline.py +287 -0
- islkit/data.py +320 -0
- islkit/device.py +38 -0
- islkit/domain.py +187 -0
- islkit/features.py +323 -0
- islkit/infer.py +911 -0
- islkit/labels.py +213 -0
- islkit/metrics.py +81 -0
- islkit/model.py +623 -0
- islkit/pipeline.py +717 -0
- islkit/plotting.py +131 -0
- islkit/seeding.py +19 -0
- islkit/server.py +246 -0
- islkit/view.py +287 -0
- islkit/viz.py +435 -0
- islkit-0.1.0.dist-info/METADATA +200 -0
- islkit-0.1.0.dist-info/RECORD +21 -0
- islkit-0.1.0.dist-info/WHEEL +4 -0
- islkit-0.1.0.dist-info/licenses/LICENSE +21 -0
islkit/features.py
ADDED
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
"""MediaPipe landmarks to feature vectors.
|
|
2
|
+
|
|
3
|
+
Backend-neutral: the RawFrame dataclass is the contract. Legacy
|
|
4
|
+
mp.solutions.holistic and the Tasks API landmarkers both adapt into it,
|
|
5
|
+
so nothing downstream changes if you swap backends.
|
|
6
|
+
|
|
7
|
+
Feature vector: 183 floats per frame, 352 with velocity.
|
|
8
|
+
hand-local shapes 2 x 21 x 3 = 126
|
|
9
|
+
wrist positions 2 x 3 = 6
|
|
10
|
+
upper-body pose 11 x 3 = 33
|
|
11
|
+
non-manual scalars = 4
|
|
12
|
+
validity mask = 14
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
|
|
22
|
+
# --------------------------------------------------------------------------
|
|
23
|
+
# Landmark index selections
|
|
24
|
+
# --------------------------------------------------------------------------
|
|
25
|
+
|
|
26
|
+
# BlazePose 33-point model. Legs discarded — ISL signing space is torso-up.
|
|
27
|
+
POSE_SEL = [
|
|
28
|
+
0, # nose
|
|
29
|
+
7,
|
|
30
|
+
8, # ears — head orientation
|
|
31
|
+
11,
|
|
32
|
+
12, # shoulders — the body frame
|
|
33
|
+
13,
|
|
34
|
+
14, # elbows
|
|
35
|
+
15,
|
|
36
|
+
16, # wrists
|
|
37
|
+
23,
|
|
38
|
+
24, # hips — torso scale reference
|
|
39
|
+
]
|
|
40
|
+
L_SHOULDER, R_SHOULDER = 11, 12
|
|
41
|
+
|
|
42
|
+
# Canonical 468-point face mesh. Used only to derive 4 scalars.
|
|
43
|
+
FACE_IDX = {
|
|
44
|
+
"brow_l": 105,
|
|
45
|
+
"brow_r": 334,
|
|
46
|
+
"brow_inner_l": 55,
|
|
47
|
+
"brow_inner_r": 285,
|
|
48
|
+
"eye_upper_l": 159,
|
|
49
|
+
"eye_lower_l": 145,
|
|
50
|
+
"eye_upper_r": 386,
|
|
51
|
+
"eye_lower_r": 374,
|
|
52
|
+
"lip_upper": 13,
|
|
53
|
+
"lip_lower": 14,
|
|
54
|
+
"mouth_l": 61,
|
|
55
|
+
"mouth_r": 291,
|
|
56
|
+
"forehead": 10,
|
|
57
|
+
"chin": 152,
|
|
58
|
+
"cheek_l": 234,
|
|
59
|
+
"cheek_r": 454,
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
N_HAND, N_POSE_SEL = 21, len(POSE_SEL)
|
|
63
|
+
DIM_GEOM = 2 * N_HAND * 3 + 2 * 3 + N_POSE_SEL * 3 + 4 # 169
|
|
64
|
+
DIM_MASK = 2 + N_POSE_SEL + 1 # 14
|
|
65
|
+
DIM_FRAME = DIM_GEOM + DIM_MASK # 183
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --------------------------------------------------------------------------
|
|
69
|
+
# Backend-neutral frame contract
|
|
70
|
+
# --------------------------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class RawFrame:
|
|
75
|
+
"""One frame of raw landmarks. None means the tracker lost that part."""
|
|
76
|
+
|
|
77
|
+
pose: np.ndarray | None # (33, 4) x, y, z, visibility
|
|
78
|
+
face: np.ndarray | None # (468, 3)
|
|
79
|
+
hand_left: np.ndarray | None # (21, 3)
|
|
80
|
+
hand_right: np.ndarray | None # (21, 3)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _to_array(landmark_list, with_visibility=False):
|
|
84
|
+
if landmark_list is None:
|
|
85
|
+
return None
|
|
86
|
+
lms = landmark_list.landmark
|
|
87
|
+
if with_visibility:
|
|
88
|
+
return np.array([[p.x, p.y, p.z, p.visibility] for p in lms], np.float32)
|
|
89
|
+
return np.array([[p.x, p.y, p.z] for p in lms], np.float32)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def from_holistic(results) -> RawFrame:
|
|
93
|
+
"""Adapter for legacy mp.solutions.holistic results."""
|
|
94
|
+
return RawFrame(
|
|
95
|
+
pose=_to_array(results.pose_landmarks, with_visibility=True),
|
|
96
|
+
face=_to_array(results.face_landmarks),
|
|
97
|
+
hand_left=_to_array(results.left_hand_landmarks),
|
|
98
|
+
hand_right=_to_array(results.right_hand_landmarks),
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def from_tasks(pose_result, hand_result, face_result) -> RawFrame:
|
|
103
|
+
"""Adapter for the Tasks API: PoseLandmarker + HandLandmarker + FaceLandmarker.
|
|
104
|
+
|
|
105
|
+
HandLandmarker returns a list of hands plus a parallel handedness list.
|
|
106
|
+
We only use handedness to fill the two slots; canonicalisation below does
|
|
107
|
+
not trust it.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
def pack(lms, vis=False):
|
|
111
|
+
if not lms:
|
|
112
|
+
return None
|
|
113
|
+
if vis:
|
|
114
|
+
return np.array(
|
|
115
|
+
[[p.x, p.y, p.z, getattr(p, "visibility", 1.0)] for p in lms], np.float32
|
|
116
|
+
)
|
|
117
|
+
return np.array([[p.x, p.y, p.z] for p in lms], np.float32)
|
|
118
|
+
|
|
119
|
+
left = right = None
|
|
120
|
+
if hand_result and hand_result.hand_landmarks:
|
|
121
|
+
for lms, hd in zip(hand_result.hand_landmarks, hand_result.handedness, strict=False):
|
|
122
|
+
arr = pack(lms)
|
|
123
|
+
if hd[0].category_name.lower().startswith("l"):
|
|
124
|
+
left = arr
|
|
125
|
+
else:
|
|
126
|
+
right = arr
|
|
127
|
+
|
|
128
|
+
pose = face = None
|
|
129
|
+
if pose_result and pose_result.pose_landmarks:
|
|
130
|
+
pose = pack(pose_result.pose_landmarks[0], vis=True)
|
|
131
|
+
if face_result and face_result.face_landmarks:
|
|
132
|
+
face = pack(face_result.face_landmarks[0])
|
|
133
|
+
|
|
134
|
+
return RawFrame(pose=pose, face=face, hand_left=left, hand_right=right)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# --------------------------------------------------------------------------
|
|
138
|
+
# Feature encoding
|
|
139
|
+
# --------------------------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _hand_local(hand: np.ndarray) -> np.ndarray:
|
|
143
|
+
"""Wrist-origin, hand-size-scaled. Handshape independent of position."""
|
|
144
|
+
wrist = hand[0]
|
|
145
|
+
centred = hand - wrist
|
|
146
|
+
scale = np.linalg.norm(hand[9] - wrist) # wrist to middle-finger MCP
|
|
147
|
+
if scale < 1e-6:
|
|
148
|
+
return np.zeros_like(centred)
|
|
149
|
+
return centred / scale
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _nonmanual(face: np.ndarray, height: float, width: float) -> np.ndarray:
|
|
153
|
+
"""Four scalars replacing 1404 raw face coordinates."""
|
|
154
|
+
f = FACE_IDX
|
|
155
|
+
brow_mid = (face[f["brow_l"]] + face[f["brow_r"]]) / 2
|
|
156
|
+
eye_mid = (face[f["eye_upper_l"]] + face[f["eye_upper_r"]]) / 2
|
|
157
|
+
return np.array(
|
|
158
|
+
[
|
|
159
|
+
np.linalg.norm(brow_mid - eye_mid) / height, # raise
|
|
160
|
+
np.linalg.norm(face[f["brow_inner_l"]] - face[f["brow_inner_r"]]) / width,
|
|
161
|
+
np.linalg.norm(face[f["lip_upper"]] - face[f["lip_lower"]]) / height,
|
|
162
|
+
np.linalg.norm(face[f["mouth_l"]] - face[f["mouth_r"]]) / width,
|
|
163
|
+
],
|
|
164
|
+
np.float32,
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def encode_frame(raw: RawFrame, dominant: str = "right") -> np.ndarray:
|
|
169
|
+
"""RawFrame -> (183,) float32. All-zero mask means an unusable frame."""
|
|
170
|
+
out = np.zeros(DIM_FRAME, np.float32)
|
|
171
|
+
|
|
172
|
+
# Body frame requires shoulders. Without pose there is no usable geometry.
|
|
173
|
+
if raw.pose is None:
|
|
174
|
+
return out
|
|
175
|
+
pose_xyz = raw.pose[:, :3]
|
|
176
|
+
origin = (pose_xyz[L_SHOULDER] + pose_xyz[R_SHOULDER]) / 2
|
|
177
|
+
scale = float(np.linalg.norm(pose_xyz[L_SHOULDER] - pose_xyz[R_SHOULDER]))
|
|
178
|
+
if scale < 1e-6:
|
|
179
|
+
return out
|
|
180
|
+
|
|
181
|
+
# --- Canonicalise hand slots by geometry, never by the handedness label.
|
|
182
|
+
# Slot 0 is the dominant hand: the one nearer the dominant-side shoulder.
|
|
183
|
+
dom_shoulder = pose_xyz[R_SHOULDER if dominant == "right" else L_SHOULDER]
|
|
184
|
+
present = [
|
|
185
|
+
(h, np.linalg.norm(h[0] - dom_shoulder))
|
|
186
|
+
for h in (raw.hand_left, raw.hand_right)
|
|
187
|
+
if h is not None
|
|
188
|
+
]
|
|
189
|
+
present.sort(key=lambda t: t[1])
|
|
190
|
+
slots: list[np.ndarray | None] = [None, None]
|
|
191
|
+
for i, (hand, _) in enumerate(present[:2]):
|
|
192
|
+
slots[i] = hand
|
|
193
|
+
|
|
194
|
+
i = 0
|
|
195
|
+
for s, hand in enumerate(slots):
|
|
196
|
+
if hand is not None:
|
|
197
|
+
out[i : i + 63] = _hand_local(hand).ravel()
|
|
198
|
+
out[126 + s * 3 : 126 + s * 3 + 3] = (hand[0] - origin) / scale
|
|
199
|
+
out[DIM_GEOM + s] = 1.0 # hand-present flag
|
|
200
|
+
i += 63
|
|
201
|
+
|
|
202
|
+
# --- Upper-body pose in the body frame
|
|
203
|
+
out[132 : 132 + N_POSE_SEL * 3] = ((pose_xyz[POSE_SEL] - origin) / scale).ravel()
|
|
204
|
+
out[DIM_GEOM + 2 : DIM_GEOM + 2 + N_POSE_SEL] = raw.pose[POSE_SEL, 3] # visibility
|
|
205
|
+
|
|
206
|
+
# --- Non-manual scalars
|
|
207
|
+
if raw.face is not None:
|
|
208
|
+
f = raw.face
|
|
209
|
+
h = float(np.linalg.norm(f[FACE_IDX["forehead"]] - f[FACE_IDX["chin"]]))
|
|
210
|
+
w = float(np.linalg.norm(f[FACE_IDX["cheek_l"]] - f[FACE_IDX["cheek_r"]]))
|
|
211
|
+
if h > 1e-6 and w > 1e-6:
|
|
212
|
+
out[165:169] = _nonmanual(f, h, w)
|
|
213
|
+
out[DIM_FRAME - 1] = 1.0 # face-present flag
|
|
214
|
+
|
|
215
|
+
return out
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def encode_clip(
|
|
219
|
+
frames: list[RawFrame], T: int = 48, dominant: str = "right", with_velocity: bool = True
|
|
220
|
+
) -> np.ndarray:
|
|
221
|
+
"""Variable-length clip -> (T, 352). Resampled, not padded."""
|
|
222
|
+
if not frames:
|
|
223
|
+
return np.zeros(
|
|
224
|
+
(T, DIM_FRAME + DIM_GEOM if with_velocity else DIM_FRAME),
|
|
225
|
+
np.float32,
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
seq = np.stack([encode_frame(f, dominant) for f in frames])
|
|
229
|
+
|
|
230
|
+
# Linear resample to fixed T. Duration becomes an augmentation axis,
|
|
231
|
+
# not a confound the model has to learn around.
|
|
232
|
+
src = np.linspace(0, len(seq) - 1, len(seq))
|
|
233
|
+
dst = np.linspace(0, len(seq) - 1, T)
|
|
234
|
+
seq = np.stack([np.interp(dst, src, seq[:, d]) for d in range(seq.shape[1])], axis=1).astype(
|
|
235
|
+
np.float32
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
if not with_velocity:
|
|
239
|
+
return seq
|
|
240
|
+
vel = np.zeros((T, DIM_GEOM), np.float32)
|
|
241
|
+
vel[1:] = seq[1:, :DIM_GEOM] - seq[:-1, :DIM_GEOM]
|
|
242
|
+
return np.concatenate([seq, vel], axis=1)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
# --------------------------------------------------------------------------
|
|
246
|
+
# Train/serve identity
|
|
247
|
+
# --------------------------------------------------------------------------
|
|
248
|
+
|
|
249
|
+
# A fixed probe clip, encoded and hashed. The point is to detect a change in
|
|
250
|
+
# what the encoder DOES, which no config field can catch: change how the body
|
|
251
|
+
# frame is scaled and every number moves while T, dominant and with_velocity
|
|
252
|
+
# stay identical. That failure is silent and confident, and this project has
|
|
253
|
+
# been bitten by its shape three times (the mirror, the visibility skew, the
|
|
254
|
+
# face block length).
|
|
255
|
+
#
|
|
256
|
+
# Hashing the source instead would change on every docstring edit, and an alarm
|
|
257
|
+
# that fires on cosmetic changes is one people learn to ignore.
|
|
258
|
+
FINGERPRINT_T = 16
|
|
259
|
+
FINGERPRINT_DECIMALS = 6 # absorbs float differences between x86 and aarch64
|
|
260
|
+
_FINGERPRINT_SEED = 20260906
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _fingerprint_clip(n_frames: int = 24) -> list[RawFrame]:
|
|
264
|
+
"""Deterministic nonsense with the right shape. Never touches a camera.
|
|
265
|
+
|
|
266
|
+
Every third frame drops the left hand so the validity mask and the
|
|
267
|
+
geometric slot assignment are both exercised, not just the dense path.
|
|
268
|
+
"""
|
|
269
|
+
rng = np.random.default_rng(_FINGERPRINT_SEED)
|
|
270
|
+
frames = []
|
|
271
|
+
for i in range(n_frames):
|
|
272
|
+
pose = np.zeros((33, 4), np.float32)
|
|
273
|
+
pose[:, :3] = rng.random((33, 3), dtype=np.float32)
|
|
274
|
+
pose[:, 3] = 1.0
|
|
275
|
+
frames.append(
|
|
276
|
+
RawFrame(
|
|
277
|
+
pose=pose,
|
|
278
|
+
face=rng.random((468, 3), dtype=np.float32), # FACE_IDX's canonical mesh
|
|
279
|
+
hand_left=rng.random((21, 3), dtype=np.float32) if i % 3 else None,
|
|
280
|
+
hand_right=rng.random((21, 3), dtype=np.float32),
|
|
281
|
+
)
|
|
282
|
+
)
|
|
283
|
+
return frames
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
# The hash above is exact, and exactness is the wrong test across machines.
|
|
287
|
+
# Measured 2026-09-07, same features.py and same numpy 1.26.4 on both ends:
|
|
288
|
+
# the probe clip differs between macOS/arm64 and the board's linux/aarch64 by
|
|
289
|
+
# up to 2.384e-07 in 1165 of 5632 values — float32 epsilon from libm and SIMD
|
|
290
|
+
# ordering, not a change in what the encoder does. Rounding cannot fix that
|
|
291
|
+
# reliably: a value sitting within the noise of a rounding boundary still flips,
|
|
292
|
+
# and at 4 dp roughly 27 of 5632 values are expected to sit there.
|
|
293
|
+
#
|
|
294
|
+
# So the guard compares numerically, with a tolerance two orders above the
|
|
295
|
+
# observed noise and orders below any real encoder change (the visibility skew
|
|
296
|
+
# and the mirror both moved values by O(0.1)). The hash stays, for logs and for
|
|
297
|
+
# reading a checkpoint's identity at a glance.
|
|
298
|
+
FINGERPRINT_ATOL = 1e-5
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def encoder_signature() -> np.ndarray:
|
|
302
|
+
"""The probe clip itself, for a comparison that tolerates float noise."""
|
|
303
|
+
return encode_clip(
|
|
304
|
+
_fingerprint_clip(), T=FINGERPRINT_T, dominant="right", with_velocity=True
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def signatures_match(a, b, atol: float = FINGERPRINT_ATOL) -> bool:
|
|
309
|
+
"""Whether two encoder signatures describe the same encoder behaviour."""
|
|
310
|
+
a, b = np.asarray(a), np.asarray(b)
|
|
311
|
+
return a.shape == b.shape and bool(np.allclose(a, b, atol=atol, rtol=0.0))
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def encoder_fingerprint() -> str:
|
|
315
|
+
"""16 hex chars identifying this encoder's behaviour.
|
|
316
|
+
|
|
317
|
+
Written into every checkpoint at training time and checked when one is
|
|
318
|
+
loaded, so a model can never be served by an encoder it was not fitted on.
|
|
319
|
+
"""
|
|
320
|
+
clip = encode_clip(_fingerprint_clip(), T=FINGERPRINT_T, dominant="right", with_velocity=True)
|
|
321
|
+
# +0.0 normalises -0.0, which hashes differently from 0.0 for no useful reason.
|
|
322
|
+
quantised = np.round(clip.astype(np.float64), FINGERPRINT_DECIMALS) + 0.0
|
|
323
|
+
return hashlib.sha256(quantised.tobytes()).hexdigest()[:16]
|