islkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- islkit/__init__.py +88 -0
- islkit/adapters.py +237 -0
- islkit/baseline.py +287 -0
- islkit/data.py +320 -0
- islkit/device.py +38 -0
- islkit/domain.py +187 -0
- islkit/features.py +323 -0
- islkit/infer.py +911 -0
- islkit/labels.py +213 -0
- islkit/metrics.py +81 -0
- islkit/model.py +623 -0
- islkit/pipeline.py +717 -0
- islkit/plotting.py +131 -0
- islkit/seeding.py +19 -0
- islkit/server.py +246 -0
- islkit/view.py +287 -0
- islkit/viz.py +435 -0
- islkit-0.1.0.dist-info/METADATA +200 -0
- islkit-0.1.0.dist-info/RECORD +21 -0
- islkit-0.1.0.dist-info/WHEEL +4 -0
- islkit-0.1.0.dist-info/licenses/LICENSE +21 -0
islkit/viz.py
ADDED
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
"""Replay a stick figure from *normalised* features.
|
|
2
|
+
|
|
3
|
+
A replay instrument. A normalisation bug does not show up in a loss
|
|
4
|
+
curve — the model just learns something slightly wrong and converges anyway.
|
|
5
|
+
It shows up instantly as a skeleton that shears, drifts, or grows when the
|
|
6
|
+
signer steps back. So the reconstruction here deliberately reads only the
|
|
7
|
+
encoded 183-vector, never the raw landmarks: if the encoded vector has lost
|
|
8
|
+
information or tangled two frames together, the figure has to be wrong too.
|
|
9
|
+
|
|
10
|
+
Everything is drawn in the **body frame** — origin at the shoulder midpoint,
|
|
11
|
+
one unit = one shoulder width. Two consequences worth knowing before reading a
|
|
12
|
+
replay:
|
|
13
|
+
|
|
14
|
+
* All panels share fixed axis limits. Per-panel autoscale would silently
|
|
15
|
+
rescale a figure that had drifted, which is the exact bug being hunted.
|
|
16
|
+
* Absolute hand size is *not* in the feature vector, by design: `_hand_local`
|
|
17
|
+
divides it out so handshape survives camera distance. Hands are therefore
|
|
18
|
+
drawn at a fixed nominal size. A hand that changes size between replays is a
|
|
19
|
+
bug in the visualiser, not in the encoder.
|
|
20
|
+
|
|
21
|
+
Only x and y are drawn. MediaPipe's z is a weak relative-depth estimate and
|
|
22
|
+
plotting it would imply a precision it does not have.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from collections.abc import Mapping
|
|
28
|
+
from dataclasses import dataclass
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
import matplotlib
|
|
32
|
+
|
|
33
|
+
matplotlib.use("Agg") # no display; must be set before pyplot is imported
|
|
34
|
+
import matplotlib.pyplot as plt # noqa: E402
|
|
35
|
+
import numpy as np # noqa: E402
|
|
36
|
+
from matplotlib.animation import FuncAnimation, PillowWriter # noqa: E402
|
|
37
|
+
|
|
38
|
+
from islkit.features import DIM_FRAME, DIM_GEOM, N_POSE_SEL # noqa: E402
|
|
39
|
+
from islkit.plotting import RUNS_DIR # noqa: E402
|
|
40
|
+
|
|
41
|
+
# --------------------------------------------------------------------------
|
|
42
|
+
# Skeleton topology
|
|
43
|
+
# --------------------------------------------------------------------------
|
|
44
|
+
|
|
45
|
+
# Indices into the 11 kept pose landmarks, i.e. positions within POSE_SEL:
|
|
46
|
+
# 0 nose 1 ear_l 2 ear_r 3 sh_l 4 sh_r 5 el_l 6 el_r
|
|
47
|
+
# 7 wr_l 8 wr_r 9 hip_l 10 hip_r
|
|
48
|
+
NOSE, EAR_L, EAR_R = 0, 1, 2
|
|
49
|
+
SH_L, SH_R, EL_L, EL_R, WR_L, WR_R, HIP_L, HIP_R = 3, 4, 5, 6, 7, 8, 9, 10
|
|
50
|
+
|
|
51
|
+
POSE_EDGES = [
|
|
52
|
+
(SH_L, SH_R), # shoulder line — the body-frame baseline, always 1.0 long
|
|
53
|
+
(SH_L, EL_L),
|
|
54
|
+
(EL_L, WR_L),
|
|
55
|
+
(SH_R, EL_R),
|
|
56
|
+
(EL_R, WR_R),
|
|
57
|
+
(SH_L, HIP_L),
|
|
58
|
+
(SH_R, HIP_R),
|
|
59
|
+
(HIP_L, HIP_R),
|
|
60
|
+
(EAR_L, NOSE),
|
|
61
|
+
(NOSE, EAR_R),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
HAND_EDGES = [
|
|
65
|
+
(0, 1), (1, 2), (2, 3), (3, 4), # thumb
|
|
66
|
+
(0, 5), (5, 6), (6, 7), (7, 8), # index
|
|
67
|
+
(9, 10), (10, 11), (11, 12), # middle
|
|
68
|
+
(13, 14), (14, 15), (15, 16), # ring
|
|
69
|
+
(0, 17), (17, 18), (18, 19), (19, 20), # pinky
|
|
70
|
+
(5, 9), (9, 13), (13, 17), # knuckle line
|
|
71
|
+
] # fmt: skip
|
|
72
|
+
|
|
73
|
+
# Wrist-to-middle-MCP as a fraction of shoulder width. Roughly anatomical
|
|
74
|
+
# (~9 cm against a ~40 cm shoulder span). The encoder discards the true value,
|
|
75
|
+
# so this is a drawing constant and nothing more.
|
|
76
|
+
HAND_SCALE = 0.22
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
# --------------------------------------------------------------------------
|
|
80
|
+
# Reconstruction
|
|
81
|
+
# --------------------------------------------------------------------------
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass
|
|
85
|
+
class Skeleton:
|
|
86
|
+
"""One frame of features, unpacked back into drawable geometry.
|
|
87
|
+
|
|
88
|
+
All coordinates are in the body frame. `usable` is false for a frame the
|
|
89
|
+
encoder rejected (no shoulders, so no frame of reference at all).
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
pose: np.ndarray # (11, 3)
|
|
93
|
+
wrists: np.ndarray # (2, 3) — slot 0 is the dominant hand
|
|
94
|
+
hands: np.ndarray # (2, 21, 3) placed at the wrists, nominal size
|
|
95
|
+
hand_valid: np.ndarray # (2,) bool
|
|
96
|
+
pose_vis: np.ndarray # (11,) float, MediaPipe visibility
|
|
97
|
+
face_valid: bool
|
|
98
|
+
nonmanual: np.ndarray # (4,) brow raise, brow furrow, mouth open, mouth width
|
|
99
|
+
usable: bool
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def reconstruct(feat: np.ndarray) -> Skeleton:
|
|
103
|
+
"""Encoded frame -> `Skeleton`. Accepts a 183- or 352-vector.
|
|
104
|
+
|
|
105
|
+
The velocity half of a 352-vector is ignored: it is a difference between
|
|
106
|
+
consecutive frames, and animating the positions already shows it.
|
|
107
|
+
"""
|
|
108
|
+
feat = np.asarray(feat, np.float32).ravel()
|
|
109
|
+
if feat.shape[0] < DIM_FRAME:
|
|
110
|
+
raise ValueError(f"expected at least {DIM_FRAME} features, got {feat.shape[0]}")
|
|
111
|
+
feat = feat[:DIM_FRAME]
|
|
112
|
+
|
|
113
|
+
mask = feat[DIM_GEOM:]
|
|
114
|
+
hand_valid = mask[:2] > 0.5
|
|
115
|
+
pose_vis = mask[2 : 2 + N_POSE_SEL].copy()
|
|
116
|
+
face_valid = bool(mask[-1] > 0.5)
|
|
117
|
+
|
|
118
|
+
shapes = feat[:126].reshape(2, 21, 3)
|
|
119
|
+
wrists = feat[126:132].reshape(2, 3)
|
|
120
|
+
pose = feat[132 : 132 + N_POSE_SEL * 3].reshape(N_POSE_SEL, 3).copy()
|
|
121
|
+
|
|
122
|
+
# Hand-local shapes are unit-scaled on wrist-to-middle-MCP. Re-inflate to a
|
|
123
|
+
# fixed nominal size and hang each off its own wrist position, which is the
|
|
124
|
+
# separate block the decomposition kept precisely so this can be done.
|
|
125
|
+
hands = wrists[:, None, :] + shapes * HAND_SCALE
|
|
126
|
+
hands[~hand_valid] = 0.0
|
|
127
|
+
|
|
128
|
+
return Skeleton(
|
|
129
|
+
pose=pose,
|
|
130
|
+
wrists=wrists.copy(),
|
|
131
|
+
hands=hands.astype(np.float32),
|
|
132
|
+
hand_valid=hand_valid,
|
|
133
|
+
pose_vis=pose_vis,
|
|
134
|
+
face_valid=face_valid,
|
|
135
|
+
nonmanual=feat[165:169].copy(),
|
|
136
|
+
usable=bool(feat.any()),
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _face_glyph(sk: Skeleton) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
141
|
+
"""Brow lines and a mouth outline from the four non-manual scalars.
|
|
142
|
+
|
|
143
|
+
Indicative, not anatomical — the scalars are ratios, so the glyph shows
|
|
144
|
+
whether they *move sensibly*, which is the only thing worth seeing here.
|
|
145
|
+
Returns (left brow, right brow, mouth outline), each (N, 2).
|
|
146
|
+
"""
|
|
147
|
+
nose = sk.pose[NOSE, :2]
|
|
148
|
+
face_w = float(np.linalg.norm(sk.pose[EAR_L, :2] - sk.pose[EAR_R, :2])) * 1.15
|
|
149
|
+
if face_w < 1e-6:
|
|
150
|
+
face_w = 0.45
|
|
151
|
+
face_h = face_w * 1.3
|
|
152
|
+
raise_, furrow, mouth_open, mouth_w = sk.nonmanual
|
|
153
|
+
|
|
154
|
+
# y grows downward (MediaPipe convention), so "up" is a subtraction.
|
|
155
|
+
brow_y = nose[1] - 0.18 * face_h - float(raise_) * face_h
|
|
156
|
+
inner = float(furrow) * face_w / 2
|
|
157
|
+
outer = inner + 0.20 * face_w
|
|
158
|
+
brow_l = np.array([[nose[0] + inner, brow_y], [nose[0] + outer, brow_y]], np.float32)
|
|
159
|
+
brow_r = np.array([[nose[0] - inner, brow_y], [nose[0] - outer, brow_y]], np.float32)
|
|
160
|
+
|
|
161
|
+
theta = np.linspace(0, 2 * np.pi, 33)
|
|
162
|
+
mouth = np.stack(
|
|
163
|
+
[
|
|
164
|
+
nose[0] + np.cos(theta) * float(mouth_w) * face_w / 2,
|
|
165
|
+
nose[1] + 0.30 * face_h + np.sin(theta) * max(float(mouth_open), 0.01) * face_h / 2,
|
|
166
|
+
],
|
|
167
|
+
axis=1,
|
|
168
|
+
).astype(np.float32)
|
|
169
|
+
return brow_l, brow_r, mouth
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# --------------------------------------------------------------------------
|
|
173
|
+
# Drawing
|
|
174
|
+
# --------------------------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
class SkeletonArtist:
|
|
178
|
+
"""Line artists for one skeleton on one axis, updated in place per frame."""
|
|
179
|
+
|
|
180
|
+
def __init__(
|
|
181
|
+
self,
|
|
182
|
+
ax: plt.Axes,
|
|
183
|
+
color: str = "#1f77b4",
|
|
184
|
+
lw: float = 1.6,
|
|
185
|
+
alpha: float = 1.0,
|
|
186
|
+
joints: bool = True,
|
|
187
|
+
face: bool = True,
|
|
188
|
+
zorder: float = 2.0,
|
|
189
|
+
) -> None:
|
|
190
|
+
self.face = face
|
|
191
|
+
kw = {
|
|
192
|
+
"color": color,
|
|
193
|
+
"lw": lw,
|
|
194
|
+
"alpha": alpha,
|
|
195
|
+
"solid_capstyle": "round",
|
|
196
|
+
"zorder": zorder,
|
|
197
|
+
}
|
|
198
|
+
# The neck is not a landmark; it is drawn from the shoulder midpoint,
|
|
199
|
+
# which in the body frame is the origin by construction. If it ever
|
|
200
|
+
# renders anywhere but between the shoulders, the frame is wrong.
|
|
201
|
+
self.body = [ax.plot([], [], **kw)[0] for _ in POSE_EDGES]
|
|
202
|
+
self.neck = ax.plot([], [], **kw)[0]
|
|
203
|
+
self.head = ax.plot([], [], **{**kw, "lw": lw * 0.8})[0]
|
|
204
|
+
self.hands = [[ax.plot([], [], **kw)[0] for _ in HAND_EDGES] for _ in range(2)]
|
|
205
|
+
self.dots = ax.plot(
|
|
206
|
+
[],
|
|
207
|
+
[],
|
|
208
|
+
"o",
|
|
209
|
+
color=color,
|
|
210
|
+
ms=3.0 if joints else 0.0,
|
|
211
|
+
alpha=alpha,
|
|
212
|
+
ls="none",
|
|
213
|
+
zorder=zorder,
|
|
214
|
+
)[0]
|
|
215
|
+
if face:
|
|
216
|
+
fkw = {**kw, "lw": lw * 0.8}
|
|
217
|
+
self.brows = [ax.plot([], [], **fkw)[0] for _ in range(2)]
|
|
218
|
+
self.mouth = ax.plot([], [], **fkw)[0]
|
|
219
|
+
|
|
220
|
+
def artists(self):
|
|
221
|
+
out = [*self.body, self.neck, self.head, self.dots, *self.hands[0], *self.hands[1]]
|
|
222
|
+
if self.face:
|
|
223
|
+
out += [*self.brows, self.mouth]
|
|
224
|
+
return out
|
|
225
|
+
|
|
226
|
+
def clear(self):
|
|
227
|
+
for artist in self.artists():
|
|
228
|
+
artist.set_data([], [])
|
|
229
|
+
return self.artists()
|
|
230
|
+
|
|
231
|
+
def update(self, sk: Skeleton):
|
|
232
|
+
if not sk.usable:
|
|
233
|
+
return self.clear()
|
|
234
|
+
|
|
235
|
+
for line, (a, b) in zip(self.body, POSE_EDGES, strict=True):
|
|
236
|
+
line.set_data(sk.pose[[a, b], 0], sk.pose[[a, b], 1])
|
|
237
|
+
centre = (sk.pose[EAR_L, :2] + sk.pose[EAR_R, :2]) / 2
|
|
238
|
+
radius = float(np.linalg.norm(sk.pose[EAR_L, :2] - sk.pose[EAR_R, :2])) * 0.72
|
|
239
|
+
self.neck.set_data([0.0, float(centre[0])], [0.0, float(centre[1] + radius * 0.6)])
|
|
240
|
+
theta = np.linspace(0, 2 * np.pi, 49)
|
|
241
|
+
self.head.set_data(centre[0] + radius * np.cos(theta), centre[1] + radius * np.sin(theta))
|
|
242
|
+
self.dots.set_data(sk.pose[:, 0], sk.pose[:, 1])
|
|
243
|
+
|
|
244
|
+
for slot in range(2):
|
|
245
|
+
pts, valid = sk.hands[slot], sk.hand_valid[slot]
|
|
246
|
+
for line, (a, b) in zip(self.hands[slot], HAND_EDGES, strict=True):
|
|
247
|
+
# A missing hand is drawn as nothing, never as a hand at the
|
|
248
|
+
# origin — that distinction is the whole point of the mask.
|
|
249
|
+
if valid:
|
|
250
|
+
line.set_data(pts[[a, b], 0], pts[[a, b], 1])
|
|
251
|
+
else:
|
|
252
|
+
line.set_data([], [])
|
|
253
|
+
|
|
254
|
+
if self.face:
|
|
255
|
+
if sk.face_valid:
|
|
256
|
+
brow_l, brow_r, mouth = _face_glyph(sk)
|
|
257
|
+
self.brows[0].set_data(brow_l[:, 0], brow_l[:, 1])
|
|
258
|
+
self.brows[1].set_data(brow_r[:, 0], brow_r[:, 1])
|
|
259
|
+
self.mouth.set_data(mouth[:, 0], mouth[:, 1])
|
|
260
|
+
else:
|
|
261
|
+
for artist in (*self.brows, self.mouth):
|
|
262
|
+
artist.set_data([], [])
|
|
263
|
+
return self.artists()
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _limits(clips: Mapping[str, np.ndarray], margin: float = 0.25) -> tuple[tuple, tuple]:
|
|
267
|
+
"""One set of axis limits for every panel.
|
|
268
|
+
|
|
269
|
+
Shared, not per-panel: autoscaling each panel would rescale away exactly
|
|
270
|
+
the divergence this figure exists to reveal.
|
|
271
|
+
"""
|
|
272
|
+
pts = []
|
|
273
|
+
for clip in clips.values():
|
|
274
|
+
for feat in clip:
|
|
275
|
+
sk = reconstruct(feat)
|
|
276
|
+
if not sk.usable:
|
|
277
|
+
continue
|
|
278
|
+
pts.append(sk.pose[:, :2])
|
|
279
|
+
for slot in range(2):
|
|
280
|
+
if sk.hand_valid[slot]:
|
|
281
|
+
pts.append(sk.hands[slot][:, :2])
|
|
282
|
+
if not pts:
|
|
283
|
+
return (-1.5, 1.5), (1.5, -1.5)
|
|
284
|
+
p = np.concatenate(pts)
|
|
285
|
+
lo, hi = p.min(0) - margin, p.max(0) + margin
|
|
286
|
+
# Equal aspect: pad the shorter axis rather than squashing the figure.
|
|
287
|
+
span = float(max(hi - lo))
|
|
288
|
+
mid = (lo + hi) / 2
|
|
289
|
+
x = (float(mid[0] - span / 2), float(mid[0] + span / 2))
|
|
290
|
+
y = (float(mid[1] + span / 2), float(mid[1] - span / 2)) # inverted: y is down
|
|
291
|
+
return x, y
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _panel(ax, xlim, ylim, title: str = "", title_color: str = "#333333") -> None:
|
|
295
|
+
# Widen x to whatever the axes box actually is, keeping y fixed, so the
|
|
296
|
+
# panel fills its slot without matplotlib rescaling anything. The aspect
|
|
297
|
+
# stays equal throughout: a squashed axis would hide a skeleton that had
|
|
298
|
+
# genuinely sheared, which is one of the bugs this figure exists to catch.
|
|
299
|
+
box = ax.get_position()
|
|
300
|
+
fig_w, fig_h = ax.figure.get_size_inches()
|
|
301
|
+
aspect = (box.width * fig_w) / (box.height * fig_h)
|
|
302
|
+
y_span = abs(ylim[1] - ylim[0])
|
|
303
|
+
x_mid = (xlim[0] + xlim[1]) / 2
|
|
304
|
+
ax.set_xlim(x_mid - y_span * aspect / 2, x_mid + y_span * aspect / 2)
|
|
305
|
+
ax.set_ylim(*ylim)
|
|
306
|
+
ax.set_aspect("equal")
|
|
307
|
+
ax.set_xticks([])
|
|
308
|
+
ax.set_yticks([])
|
|
309
|
+
for spine in ax.spines.values():
|
|
310
|
+
spine.set_color("#dddddd")
|
|
311
|
+
if title:
|
|
312
|
+
ax.set_title(title, fontsize=9, color=title_color)
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _build_figure(clips: Mapping[str, np.ndarray], title: str, reference: str | None):
|
|
316
|
+
"""Panel per clip on top, all clips overlaid underneath.
|
|
317
|
+
|
|
318
|
+
The overlay is the acceptance test: the reference is drawn as a thick pale
|
|
319
|
+
band and every clip as a thin coloured line on top of it. Identical
|
|
320
|
+
normalisation means every colour stays inside the band. Anything that
|
|
321
|
+
diverges leaves it, and you cannot miss it.
|
|
322
|
+
"""
|
|
323
|
+
names = list(clips)
|
|
324
|
+
n = len(names)
|
|
325
|
+
ref = reference if reference in clips else names[0]
|
|
326
|
+
xlim, ylim = _limits(clips)
|
|
327
|
+
colors = [plt.get_cmap("tab10")(i % 10) for i in range(n)]
|
|
328
|
+
|
|
329
|
+
fig = plt.figure(figsize=(max(2.4 * n, 8.0), 7.6))
|
|
330
|
+
gs = fig.add_gridspec(
|
|
331
|
+
2, n, height_ratios=[1.0, 1.9],
|
|
332
|
+
top=0.90, bottom=0.09, left=0.03, right=0.97, hspace=0.14, wspace=0.05,
|
|
333
|
+
) # fmt: skip
|
|
334
|
+
|
|
335
|
+
panels = []
|
|
336
|
+
for i, name in enumerate(names):
|
|
337
|
+
ax = fig.add_subplot(gs[0, i])
|
|
338
|
+
_panel(ax, xlim, ylim, name, title_color=colors[i])
|
|
339
|
+
panels.append(SkeletonArtist(ax, color=colors[i]))
|
|
340
|
+
|
|
341
|
+
# The overlay is centred over a subset of the columns rather than spanning
|
|
342
|
+
# the lot: equal aspect on a very wide box would just pad it with empty
|
|
343
|
+
# space and shrink the figure that has to be read closely.
|
|
344
|
+
span = n if n <= 3 else 3
|
|
345
|
+
start = (n - span) // 2
|
|
346
|
+
ax_ov = fig.add_subplot(gs[1, start : start + span])
|
|
347
|
+
_panel(ax_ov, xlim, ylim, f"all {n} overlaid — grey band is “{ref}”")
|
|
348
|
+
band = SkeletonArtist(ax_ov, color="#cfcfcf", lw=8.0, joints=False, zorder=1.0)
|
|
349
|
+
overlay = [
|
|
350
|
+
SkeletonArtist(ax_ov, color=colors[i], lw=1.3, alpha=0.9, joints=False, zorder=2.0)
|
|
351
|
+
for i in range(n)
|
|
352
|
+
]
|
|
353
|
+
|
|
354
|
+
if title:
|
|
355
|
+
fig.suptitle(title, fontsize=11)
|
|
356
|
+
handles = [plt.Line2D([], [], color=colors[i], lw=2.5) for i in range(n)]
|
|
357
|
+
fig.legend(
|
|
358
|
+
handles, names, loc="lower center", bbox_to_anchor=(0.5, 0.035),
|
|
359
|
+
ncols=min(n, 6), frameon=False, fontsize=8,
|
|
360
|
+
) # fmt: skip
|
|
361
|
+
stamp = fig.text(0.5, 0.012, "", ha="center", fontsize=8, color="#888888")
|
|
362
|
+
seqs = [np.asarray(clips[name], np.float32) for name in names]
|
|
363
|
+
ref_seq = np.asarray(clips[ref], np.float32)
|
|
364
|
+
|
|
365
|
+
def update(t: int):
|
|
366
|
+
artists = []
|
|
367
|
+
for i, seq in enumerate(seqs):
|
|
368
|
+
sk = reconstruct(seq[t])
|
|
369
|
+
artists += panels[i].update(sk)
|
|
370
|
+
artists += overlay[i].update(sk)
|
|
371
|
+
artists += band.update(reconstruct(ref_seq[t]))
|
|
372
|
+
sk0 = reconstruct(seqs[0][t])
|
|
373
|
+
stamp.set_text(
|
|
374
|
+
f"frame {t + 1}/{len(seqs[0])} "
|
|
375
|
+
f"hands {int(sk0.hand_valid[0])}{int(sk0.hand_valid[1])} "
|
|
376
|
+
f"face {int(sk0.face_valid)} body frame: 1 unit = shoulder width"
|
|
377
|
+
)
|
|
378
|
+
return [*artists, stamp]
|
|
379
|
+
|
|
380
|
+
return fig, update, len(seqs[0])
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
# --------------------------------------------------------------------------
|
|
384
|
+
# Entry points
|
|
385
|
+
# --------------------------------------------------------------------------
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def animate_replay(
|
|
389
|
+
clips: Mapping[str, np.ndarray],
|
|
390
|
+
experiment: str,
|
|
391
|
+
filename: str = "replay.gif",
|
|
392
|
+
fps: int = 12,
|
|
393
|
+
title: str = "",
|
|
394
|
+
reference: str | None = None,
|
|
395
|
+
) -> Path:
|
|
396
|
+
"""Animate one or more encoded clips side by side and save a GIF.
|
|
397
|
+
|
|
398
|
+
`clips` maps a label to a (T, 183) or (T, 352) array; all must share T.
|
|
399
|
+
Writes to runs/<experiment>/<filename> and returns the path.
|
|
400
|
+
"""
|
|
401
|
+
if not clips:
|
|
402
|
+
raise ValueError("nothing to replay")
|
|
403
|
+
lengths = {len(np.asarray(c)) for c in clips.values()}
|
|
404
|
+
if len(lengths) != 1:
|
|
405
|
+
raise ValueError(f"clips must share a frame count, got {sorted(lengths)}")
|
|
406
|
+
|
|
407
|
+
out_dir = RUNS_DIR / experiment
|
|
408
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
409
|
+
out_path = out_dir / filename
|
|
410
|
+
|
|
411
|
+
fig, update, n_frames = _build_figure(clips, title, reference)
|
|
412
|
+
anim = FuncAnimation(fig, update, frames=n_frames, blit=False)
|
|
413
|
+
anim.save(out_path, writer=PillowWriter(fps=fps))
|
|
414
|
+
plt.close(fig)
|
|
415
|
+
return out_path
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def still_replay(
|
|
419
|
+
clips: Mapping[str, np.ndarray],
|
|
420
|
+
experiment: str,
|
|
421
|
+
filename: str = "replay.png",
|
|
422
|
+
frame: int = 0,
|
|
423
|
+
title: str = "",
|
|
424
|
+
reference: str | None = None,
|
|
425
|
+
) -> Path:
|
|
426
|
+
"""One frame of the same figure, as a PNG. For pasting into a write-up."""
|
|
427
|
+
out_dir = RUNS_DIR / experiment
|
|
428
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
429
|
+
out_path = out_dir / filename
|
|
430
|
+
|
|
431
|
+
fig, update, n_frames = _build_figure(clips, title, reference)
|
|
432
|
+
update(int(np.clip(frame, 0, n_frames - 1)))
|
|
433
|
+
fig.savefig(out_path, dpi=130)
|
|
434
|
+
plt.close(fig)
|
|
435
|
+
return out_path
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: islkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Indian Sign Language recognition from MediaPipe landmarks: features, a dual-branch TCN, and a live recognition service
|
|
5
|
+
Project-URL: Homepage, https://github.com/jbrathwa/islkit
|
|
6
|
+
Project-URL: Issues, https://github.com/jbrathwa/islkit/issues
|
|
7
|
+
Author: Jayraj
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: accessibility,isl,landmarks,mediapipe,sign-language,tcn
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
18
|
+
Requires-Python: <3.13,>=3.12
|
|
19
|
+
Requires-Dist: mediapipe==0.10.18
|
|
20
|
+
Requires-Dist: numpy<2,>=1.26
|
|
21
|
+
Requires-Dist: torch>=2.5
|
|
22
|
+
Provides-Extra: train
|
|
23
|
+
Requires-Dist: pandas>=3.0.5; extra == 'train'
|
|
24
|
+
Requires-Dist: pyarrow>=25.0.1; extra == 'train'
|
|
25
|
+
Requires-Dist: xgboost>=3.4.1; extra == 'train'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# islkit
|
|
29
|
+
|
|
30
|
+
Indian Sign Language (ISL) recognition from MediaPipe Holistic landmarks.
|
|
31
|
+
|
|
32
|
+
islkit turns camera frames into glosses. It covers every step: a landmark
|
|
33
|
+
feature encoder, a dual-branch temporal convolutional network (TCN), a loader
|
|
34
|
+
for the INCLUDE dataset, live inference with a confidence-gated decline, and a
|
|
35
|
+
small HTTP/SSE recognition service. It runs on CPU and is built to work on small
|
|
36
|
+
ARM boards as well as laptops.
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
camera → MediaPipe Holistic → RawFrame → encode_clip (T×352) → TCN → gloss | None
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Install
|
|
43
|
+
|
|
44
|
+
```sh
|
|
45
|
+
pip install islkit # inference and the recognition service
|
|
46
|
+
pip install "islkit[train]" # + pandas, pyarrow, xgboost for INCLUDE and the baseline
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Requires **Python 3.12**. `mediapipe` is pinned to `0.10.18`, which also pins
|
|
50
|
+
`numpy<2`. That is the last MediaPipe release that keeps `mp.solutions` and
|
|
51
|
+
still runs on ARMv8.0-A (Cortex-A53) boards.
|
|
52
|
+
|
|
53
|
+
On macOS, XGBoost needs `brew install libomp`. Don't import `torch` and
|
|
54
|
+
`xgboost` in the same process: each bundles its own libomp, and the process
|
|
55
|
+
aborts with `OMP: Error #15`. `import islkit` keeps torch lazy for this reason.
|
|
56
|
+
|
|
57
|
+
## Design
|
|
58
|
+
|
|
59
|
+
**Features, not raw landmarks.** Holistic's flattened output has 1,662 values,
|
|
60
|
+
and 85% of them are face mesh. The encoder reduces each frame to **183 floats**
|
|
61
|
+
(352 with velocity):
|
|
62
|
+
|
|
63
|
+
| Block | Dims |
|
|
64
|
+
|--------------------------------|------|
|
|
65
|
+
| Hand-local shapes (2 × 21 × 3) | 126 |
|
|
66
|
+
| Wrist positions in body frame | 6 |
|
|
67
|
+
| Upper-body pose (11 landmarks) | 33 |
|
|
68
|
+
| Non-manual scalars from face | 4 |
|
|
69
|
+
| Validity mask | 14 |
|
|
70
|
+
|
|
71
|
+
The encoder uses three coordinate frames rather than one normalisation. Body
|
|
72
|
+
frame: origin at the shoulder midpoint, scaled by shoulder width. Hand-local
|
|
73
|
+
frame: origin at the wrist, scaled by the wrist-to-middle-MCP distance. Wrist
|
|
74
|
+
position is recovered separately in the body frame. This keeps handshape
|
|
75
|
+
separate from location.
|
|
76
|
+
|
|
77
|
+
**Principles the code enforces:**
|
|
78
|
+
|
|
79
|
+
- **Hand slots are geometric.** Slot 0 is the hand nearer the dominant-side
|
|
80
|
+
shoulder. MediaPipe's handedness label flips under occlusion and is never read.
|
|
81
|
+
- **Mask, never zero-fill.** A missing hand is not a hand at the origin. A
|
|
82
|
+
validity bit is carried per part and multiplied through the network.
|
|
83
|
+
- **The label map is frozen.** `LabelMap` is saved beside the weights. Class
|
|
84
|
+
order rebuilt from a directory listing silently shifts indices.
|
|
85
|
+
- **Checkpoints are tied to the encoder.** A checkpoint records an encoder
|
|
86
|
+
fingerprint, and `SignRecogniser` refuses to serve it through a different
|
|
87
|
+
encoder.
|
|
88
|
+
- **Store raw landmarks.** Recordings hold raw MediaPipe output, so they can
|
|
89
|
+
be re-encoded when the normalisation changes.
|
|
90
|
+
- **The model can say "I don't know".** Below the confidence threshold a
|
|
91
|
+
prediction is `None`. For an accessibility device, silence is better than a
|
|
92
|
+
confident wrong answer.
|
|
93
|
+
|
|
94
|
+
## Usage
|
|
95
|
+
|
|
96
|
+
### Encode a clip
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from islkit import encode_clip
|
|
100
|
+
from islkit.infer import HolisticExtractor
|
|
101
|
+
|
|
102
|
+
frames = []
|
|
103
|
+
with HolisticExtractor() as extractor:
|
|
104
|
+
for frame_bgr in video_frames: # BGR numpy arrays, e.g. from cv2
|
|
105
|
+
_, raw, _ = extractor.process(frame_bgr)
|
|
106
|
+
frames.append(raw)
|
|
107
|
+
|
|
108
|
+
clip = encode_clip(frames, T=48) # (48, 352) float32
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Recognise a sign
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from islkit import SignRecogniser
|
|
115
|
+
|
|
116
|
+
recogniser = SignRecogniser("classifier.pt", threshold=0.6) # labels_*.json beside it
|
|
117
|
+
prediction = recogniser.classify(frames)
|
|
118
|
+
print(prediction.gloss, prediction.confidence, prediction.top3)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
`prediction.gloss` is `None` when the model declines.
|
|
122
|
+
|
|
123
|
+
### Train
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
from islkit import build_model, load_include
|
|
127
|
+
from islkit.model import fit
|
|
128
|
+
|
|
129
|
+
data = load_include("path/to/isl-mediapipe-holistic-landmarks") # data.X: (N, 48, 352)
|
|
130
|
+
model = build_model(n_classes=len(data.label_map))
|
|
131
|
+
fit(model, data.X, data.y)
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
`load_include` reads the Kaggle
|
|
135
|
+
[`indian-sign-language-mediapipe-holistic-landmarks`](https://www.kaggle.com/datasets/swaptr/indian-sign-language-mediapipe-holistic-landmarks)
|
|
136
|
+
dump of INCLUDE and caches the encoded arrays. That dump records no signer or
|
|
137
|
+
session, so a class-stratified k-fold over whole clips is the only honest split
|
|
138
|
+
available. It still leaks signers, so **report numbers from it as optimistic**.
|
|
139
|
+
For your own recordings, use leave-one-session-out: `build_dataset` returns a
|
|
140
|
+
`sessions` array for this.
|
|
141
|
+
|
|
142
|
+
`replace_head` and `freeze_backbone` support fine-tuning a pretrained backbone
|
|
143
|
+
on a smaller vocabulary. Always report per-class F1 (`per_class_f1`), not just
|
|
144
|
+
aggregate accuracy.
|
|
145
|
+
|
|
146
|
+
### Run the recognition service
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
import threading
|
|
150
|
+
from islkit.infer import ClipStore, HolisticExtractor, SignRecogniser
|
|
151
|
+
from islkit.pipeline import CameraSource, RecognitionPipeline
|
|
152
|
+
from islkit.server import EventHub, make_server
|
|
153
|
+
|
|
154
|
+
recogniser = SignRecogniser("classifier.pt")
|
|
155
|
+
hub = EventHub()
|
|
156
|
+
pipeline = RecognitionPipeline(
|
|
157
|
+
recogniser=recogniser,
|
|
158
|
+
source_factory=lambda: CameraSource(0, "1280x720"),
|
|
159
|
+
on_event=hub.publish,
|
|
160
|
+
extractor_factory=HolisticExtractor,
|
|
161
|
+
store=ClipStore(),
|
|
162
|
+
on_pause=hub.clear_take,
|
|
163
|
+
)
|
|
164
|
+
threading.Thread(target=pipeline.run, daemon=True).start()
|
|
165
|
+
make_server(pipeline, hub).serve_forever() # 127.0.0.1:9978
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
| Route | Does |
|
|
169
|
+
|-----------------|----------------------------------------|
|
|
170
|
+
| `GET /results` | Server-sent events: tracking, takes, predictions |
|
|
171
|
+
| `POST /capture` | Start or pause capture |
|
|
172
|
+
| `GET /health` | Pipeline and subscriber status |
|
|
173
|
+
|
|
174
|
+
`islkit.view.make_view_server` serves an optional annotated MJPEG debug view
|
|
175
|
+
on a separate port.
|
|
176
|
+
|
|
177
|
+
Pretrained weights are not shipped with the package.
|
|
178
|
+
|
|
179
|
+
## Guides and experiments
|
|
180
|
+
|
|
181
|
+
- [`docs/`](docs/README.md): how the features, dataset, baseline, pretraining and fine-tuning
|
|
182
|
+
work, each argued from committed results.
|
|
183
|
+
- [`experiments/`](experiments/README.md): the scripts that produce those results, runnable from
|
|
184
|
+
a clone of this repository.
|
|
185
|
+
|
|
186
|
+
## Development
|
|
187
|
+
|
|
188
|
+
```sh
|
|
189
|
+
uv sync --extra train
|
|
190
|
+
uv run pytest
|
|
191
|
+
uv run ruff check .
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
The tests cover the properties that would silently break recognition: encoder
|
|
195
|
+
invariances, mask gating, label-map and checkpoint round-trips, and the
|
|
196
|
+
saved-clip layout.
|
|
197
|
+
|
|
198
|
+
## License
|
|
199
|
+
|
|
200
|
+
MIT. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
islkit/__init__.py,sha256=TE3jMbwfw67n1Cw057y7dUvaIZyzOjsxpgCBRFrd78w,2676
|
|
2
|
+
islkit/adapters.py,sha256=vrFHC00P4EeLHWc46JfK5AaBWE2Q-CXtOMJhx5OU9_Y,8604
|
|
3
|
+
islkit/baseline.py,sha256=qlU50HvkyPlV8KdB0VaAxnKL_KunJmXgHJS7SeoVxcM,12309
|
|
4
|
+
islkit/data.py,sha256=9ZWLYRMFE6UOhCUOInYhoAvZqcqQ6wBAlk80ilSpJAQ,12494
|
|
5
|
+
islkit/device.py,sha256=9fdnYIJCAwzo9UR-77JPbCQ1x9P_J1I0B_bCFRMppHc,1170
|
|
6
|
+
islkit/domain.py,sha256=eYargvj4nmqnkPOht90hmnQMT72zOnDb5PKuOYBKusU,7989
|
|
7
|
+
islkit/features.py,sha256=D_KtoSQP2fP4Ln86rkzPxUfU5ZaymJIfvFQLv4oISO8,11896
|
|
8
|
+
islkit/infer.py,sha256=Xghyd3uen61GAo1kD2mhUD0w3FTEcvxfp0n9PAb0KiA,39820
|
|
9
|
+
islkit/labels.py,sha256=DXh0W3NCe6jTQ6WFHmzYL9toOA9plcrQbrDkKhfm_gc,7982
|
|
10
|
+
islkit/metrics.py,sha256=ZvuRUzlh8Y9-PczIK-qB_uU7C0ZvNACIec05rG-hpqw,2870
|
|
11
|
+
islkit/model.py,sha256=gNFdpHk9aG1FCxRcTll0rhSxEzmR2hAiY_581OI2-lk,25764
|
|
12
|
+
islkit/pipeline.py,sha256=-XIkfDfYxUKDfE-97calQaM2E7z7fnY3aqMvOtV2nLk,28811
|
|
13
|
+
islkit/plotting.py,sha256=UXtY5aOlBxewXNpoozkC4mnwOamI_eg1bqzc55nZ88g,4285
|
|
14
|
+
islkit/seeding.py,sha256=1PG-oZdGpesg7HaR8N5Ofh47F8wi6hB1syEOcNN8nnQ,489
|
|
15
|
+
islkit/server.py,sha256=MsOa30XPbSZtw3P-LVLSgwa1Bou_rCEQr74sAi6HArU,9664
|
|
16
|
+
islkit/view.py,sha256=_ZzUmEsW15Kx4fsAANNHHS3V-VC0bHq0Lj6UQULqVUc,10929
|
|
17
|
+
islkit/viz.py,sha256=3ZCJC7VzqDcTKC2hfSlzRo_PACWa0A5cVnqS18c1qwE,16541
|
|
18
|
+
islkit-0.1.0.dist-info/METADATA,sha256=JSHY0JswdvGbwSPlOxx1VjK0g3Z3owW-gtG4WY2U5Oo,7433
|
|
19
|
+
islkit-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
20
|
+
islkit-0.1.0.dist-info/licenses/LICENSE,sha256=V2ZUbrXhmO0lJ5gEnhAVHTfopcs9h8ev1E0IDQTX-0k,1063
|
|
21
|
+
islkit-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jayraj
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|