islkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
islkit/labels.py ADDED
@@ -0,0 +1,213 @@
1
+ """Gloss <-> index mapping, dataset assembly, and prediction decoding.
2
+
3
+ Split from features.py: that module owns the geometry,
4
+ this owns everything in gloss space. One-way dependency, features.py never
5
+ imports from here.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from pathlib import Path
12
+ from typing import TYPE_CHECKING
13
+
14
+ import numpy as np
15
+
16
+ from islkit.features import RawFrame, encode_clip
17
+
18
+ if TYPE_CHECKING:
19
+ from islkit.data import IncludeData
20
+
21
+ # --------------------------------------------------------------------------
22
+ # Labels
23
+ # --------------------------------------------------------------------------
24
+
25
+
26
+ class LabelMap:
27
+ """Gloss <-> index, frozen at training time and shipped with the weights.
28
+
29
+ Rebuilding this from a directory listing at deploy time is the single
30
+ most common silent-failure bug in this pipeline: add one sign and every
31
+ index above it shifts, so the model predicts confidently and wrongly.
32
+ """
33
+
34
+ def __init__(self, glosses: list[str]):
35
+ self.glosses = list(glosses)
36
+ self._to_idx = {g: i for i, g in enumerate(self.glosses)}
37
+
38
+ @classmethod
39
+ def from_directory(cls, root: str | Path) -> LabelMap:
40
+ root = Path(root)
41
+ glosses = sorted(
42
+ p.name for p in root.iterdir() if p.is_dir() and not p.name.startswith(".")
43
+ )
44
+ return cls(glosses)
45
+
46
+ def encode(self, gloss: str) -> int:
47
+ return self._to_idx[gloss]
48
+
49
+ def decode(self, index: int) -> str:
50
+ return self.glosses[index]
51
+
52
+ def save(self, path: str | Path) -> None:
53
+ Path(path).write_text(
54
+ json.dumps({"glosses": self.glosses, "n_classes": len(self.glosses)}, indent=2)
55
+ )
56
+
57
+ @classmethod
58
+ def load(cls, path: str | Path) -> LabelMap:
59
+ return cls(json.loads(Path(path).read_text())["glosses"])
60
+
61
+ def __len__(self) -> int:
62
+ return len(self.glosses)
63
+
64
+
65
+ def read_take(path: str | Path) -> list[RawFrame]:
66
+ """One take_NN.npz -> list[RawFrame], the layout `ClipStore.save` writes.
67
+
68
+ Split out of `build_dataset` so a leave-one-session-out eval can read a
69
+ single held-out take's raw frames straight through `SignRecogniser`, the
70
+ exact path live inference uses, instead of the pre-encoded arrays
71
+ `build_dataset` returns.
72
+ """
73
+ d = np.load(Path(path), allow_pickle=True)
74
+ n = len(d["pose"])
75
+ return [
76
+ RawFrame(
77
+ pose=d["pose"][i] if d["pose"][i] is not None else None,
78
+ face=d["face"][i] if d["face"][i] is not None else None,
79
+ hand_left=d["hand_left"][i],
80
+ hand_right=d["hand_right"][i],
81
+ )
82
+ for i in range(n)
83
+ ]
84
+
85
+
86
+ def build_dataset(root: str | Path, T: int = 48, dominant: str = "right"):
87
+ """Read data/<GLOSS>/<session>/take_NN.npz -> X, y, sessions, label_map.
88
+
89
+ Each .npz holds raw landmark arrays saved at capture time, never the
90
+ encoded features — normalisation will change several times and
91
+ re-recording is not an option.
92
+ """
93
+ root = Path(root)
94
+ label_map = LabelMap.from_directory(root)
95
+ X, y, sessions = [], [], []
96
+
97
+ for gloss in label_map.glosses:
98
+ for take in sorted((root / gloss).rglob("*.npz")):
99
+ frames = read_take(take)
100
+ X.append(encode_clip(frames, T=T, dominant=dominant))
101
+ y.append(label_map.encode(gloss))
102
+ # Session id drives leave-one-session-out splitting. A random
103
+ # split leaks near-duplicate frames and inflates accuracy by
104
+ # 10-15 points.
105
+ sessions.append(take.parent.name)
106
+
107
+ return (np.stack(X), np.array(y, np.int64), np.array(sessions), label_map)
108
+
109
+
110
+ # --------------------------------------------------------------------------
111
+ # Fine-tuning data assembly
112
+ # --------------------------------------------------------------------------
113
+
114
+
115
+ def remap_labels(y: np.ndarray, source_map: LabelMap, target_map: LabelMap) -> np.ndarray:
116
+ """Re-encode `y` from `source_map`'s class order onto `target_map`'s.
117
+
118
+ Two label maps only agree on an index by coincidence — each is built by
119
+ sorting whatever glosses its own source happens to contain, so index 3 in
120
+ one is not index 3 in the other. Every remap in this pipeline has to go
121
+ through the gloss string, never the position (an unfrozen label map's failure mode is a
122
+ silent index shift; this is the same bug one hop removed from a directory
123
+ listing).
124
+
125
+ Raises if `y` contains a class `target_map` does not have: a class that
126
+ quietly has no target slot must not train as index noise.
127
+ """
128
+ glosses = [source_map.decode(int(c)) for c in y]
129
+ unknown = sorted(set(glosses) - set(target_map.glosses))
130
+ if unknown:
131
+ raise ValueError(f"labels outside the target vocabulary {target_map.glosses}: {unknown}")
132
+ return np.array([target_map.encode(g) for g in glosses], dtype=np.int64)
133
+
134
+
135
+ def pool_finetune_data(
136
+ include: IncludeData,
137
+ own_X: np.ndarray,
138
+ own_y: np.ndarray,
139
+ own_label_map: LabelMap,
140
+ target_label_map: LabelMap,
141
+ ) -> tuple[np.ndarray, np.ndarray]:
142
+ """INCLUDE clips (filtered to `target_label_map`) plus own recordings, all
143
+ on `target_label_map`'s class order. Stage 1's training set.
144
+
145
+ Safe to pool INCLUDE here specifically because stage 1 freezes the
146
+ backbone: a
147
+ frozen backbone cannot re-anchor to INCLUDE's signer mix, it only gives
148
+ the new head more label-aligned examples in the pretrained feature space.
149
+ Stage 2 must NOT reuse this — it fine-tunes on own recordings alone.
150
+ """
151
+ own_target_y = remap_labels(own_y, own_label_map, target_label_map)
152
+
153
+ target_set = set(target_label_map.glosses)
154
+ include_glosses = np.array([include.label_map.decode(int(c)) for c in include.y])
155
+ keep = np.array([g in target_set for g in include_glosses])
156
+ include_target_y = np.array(
157
+ [target_label_map.encode(g) for g in include_glosses[keep]], dtype=np.int64
158
+ )
159
+
160
+ X = np.concatenate([include.X[keep], own_X], axis=0)
161
+ y = np.concatenate([include_target_y, own_target_y], axis=0)
162
+ return X, y
163
+
164
+
165
+ # --------------------------------------------------------------------------
166
+ # Inference
167
+ # --------------------------------------------------------------------------
168
+
169
+
170
+ def decode_prediction(
171
+ probs: np.ndarray, label_map: LabelMap, threshold: float = 0.6
172
+ ) -> tuple[str | None, float]:
173
+ """Softmax vector -> (gloss, confidence). None below threshold.
174
+
175
+ Returning None is a feature. A device that admits uncertainty is more
176
+ trustworthy than one that guesses at a stranger on the user's behalf.
177
+ """
178
+ idx = int(np.argmax(probs))
179
+ conf = float(probs[idx])
180
+ return (label_map.decode(idx) if conf >= threshold else None), conf
181
+
182
+
183
+ class VoteBuffer:
184
+ """2-of-3 agreement before emitting. Kills flicker for ~100 ms latency."""
185
+
186
+ def __init__(self, window: int = 3, agree: int = 2):
187
+ self.window, self.agree, self.buf = window, agree, []
188
+
189
+ def push(self, gloss: str | None) -> str | None:
190
+ self.buf.append(gloss)
191
+ if len(self.buf) > self.window:
192
+ self.buf.pop(0)
193
+ for g in set(self.buf):
194
+ if g is not None and self.buf.count(g) >= self.agree:
195
+ self.buf.clear()
196
+ return g
197
+ return None
198
+
199
+
200
+ def normalise_gloss(text: str) -> str | None:
201
+ """Fold a typed label to the dump's convention: lowercase, alphanumeric only.
202
+
203
+ Folder names become class labels, so a label typed with a space or a capital
204
+ is a different class from the one intended — "Good morning" is not
205
+ "goodmorning", and the mistake produces no error at all. It costs twice: the
206
+ INCLUDE clips for that sign can no longer be pooled with the new recordings,
207
+ and two people spelling it differently produce two classes for one sign.
208
+
209
+ Returns None when nothing survives, so an empty prompt is an absent label
210
+ rather than a class named "".
211
+ """
212
+ cleaned = "".join(ch for ch in text.lower() if ch.isalnum())
213
+ return cleaned or None
islkit/metrics.py ADDED
@@ -0,0 +1,81 @@
1
+ """Classification metrics, written out rather than imported from sklearn.
2
+
3
+ Aggregate accuracy hides a class that never works. Per-class F1 is the metric
4
+ that tells you the truth, so it lives here and gets printed by default.
5
+ """
6
+
7
+ import numpy as np
8
+
9
+
10
+ def _as_numpy(a) -> np.ndarray:
11
+ """Accept torch tensors (on any device) or anything array-like."""
12
+ if hasattr(a, "detach"):
13
+ a = a.detach().cpu()
14
+ return np.asarray(a)
15
+
16
+
17
+ def accuracy(pred, true) -> float:
18
+ pred, true = _as_numpy(pred), _as_numpy(true)
19
+ if len(true) == 0:
20
+ return 0.0
21
+ return float((pred == true).mean())
22
+
23
+
24
+ def per_class_f1(pred, true, n_classes: int) -> list[float]:
25
+ """F1 for each class index in range(n_classes).
26
+
27
+ A class with no predictions and no true labels scores 0.0 rather than
28
+ raising — that case shows up whenever a split misses a rare class.
29
+ """
30
+ pred, true = _as_numpy(pred), _as_numpy(true)
31
+ scores = []
32
+ for c in range(n_classes):
33
+ tp = int(np.sum((pred == c) & (true == c)))
34
+ fp = int(np.sum((pred == c) & (true != c)))
35
+ fn = int(np.sum((pred != c) & (true == c)))
36
+ denom = 2 * tp + fp + fn
37
+ scores.append(2 * tp / denom if denom else 0.0)
38
+ return scores
39
+
40
+
41
+ def format_per_class_f1(pred, true, n_classes: int, names: list[str] | None = None) -> str:
42
+ scores = per_class_f1(pred, true, n_classes)
43
+ labels = names or [f"class {c}" for c in range(n_classes)]
44
+ width = max(len(label) for label in labels)
45
+ return "\n".join(
46
+ f" {label:<{width}} : {s:.3f}" for label, s in zip(labels, scores, strict=True)
47
+ )
48
+
49
+
50
+ def confusion_matrix(pred, true, n_classes: int) -> np.ndarray:
51
+ """(n_classes, n_classes) counts, rows = true class, columns = predicted."""
52
+ pred, true = _as_numpy(pred).astype(int), _as_numpy(true).astype(int)
53
+ m = np.zeros((n_classes, n_classes), dtype=np.int64)
54
+ np.add.at(m, (true, pred), 1)
55
+ return m
56
+
57
+
58
+ def top_confusions(
59
+ cm: np.ndarray, names: list[str] | None = None, k: int = 25
60
+ ) -> list[tuple[str, str, int, float]]:
61
+ """The k worst off-diagonal cells: (true, predicted, count, share of true).
62
+
63
+ A 262x262 confusion image is a texture, not a finding. This is the part of
64
+ it anyone can act on — which specific pairs the model cannot tell apart,
65
+ which is also how the demo vocabulary gets pruned of near-collisions.
66
+ """
67
+ cm = np.asarray(cm)
68
+ labels = names or [f"class {c}" for c in range(len(cm))]
69
+ off = cm.copy()
70
+ np.fill_diagonal(off, 0)
71
+ support = cm.sum(axis=1)
72
+
73
+ flat = np.argsort(off, axis=None)[::-1][:k]
74
+ out = []
75
+ for f in flat:
76
+ t, p = int(f // len(cm)), int(f % len(cm))
77
+ if off[t, p] == 0:
78
+ break
79
+ share = float(off[t, p] / support[t]) if support[t] else 0.0
80
+ out.append((labels[t], labels[p], int(off[t, p]), share))
81
+ return out