islkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- islkit/__init__.py +88 -0
- islkit/adapters.py +237 -0
- islkit/baseline.py +287 -0
- islkit/data.py +320 -0
- islkit/device.py +38 -0
- islkit/domain.py +187 -0
- islkit/features.py +323 -0
- islkit/infer.py +911 -0
- islkit/labels.py +213 -0
- islkit/metrics.py +81 -0
- islkit/model.py +623 -0
- islkit/pipeline.py +717 -0
- islkit/plotting.py +131 -0
- islkit/seeding.py +19 -0
- islkit/server.py +246 -0
- islkit/view.py +287 -0
- islkit/viz.py +435 -0
- islkit-0.1.0.dist-info/METADATA +200 -0
- islkit-0.1.0.dist-info/RECORD +21 -0
- islkit-0.1.0.dist-info/WHEEL +4 -0
- islkit-0.1.0.dist-info/licenses/LICENSE +21 -0
islkit/labels.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Gloss <-> index mapping, dataset assembly, and prediction decoding.
|
|
2
|
+
|
|
3
|
+
Split from features.py: that module owns the geometry,
|
|
4
|
+
this owns everything in gloss space. One-way dependency, features.py never
|
|
5
|
+
imports from here.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import TYPE_CHECKING
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
from islkit.features import RawFrame, encode_clip
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from islkit.data import IncludeData
|
|
20
|
+
|
|
21
|
+
# --------------------------------------------------------------------------
|
|
22
|
+
# Labels
|
|
23
|
+
# --------------------------------------------------------------------------
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class LabelMap:
|
|
27
|
+
"""Gloss <-> index, frozen at training time and shipped with the weights.
|
|
28
|
+
|
|
29
|
+
Rebuilding this from a directory listing at deploy time is the single
|
|
30
|
+
most common silent-failure bug in this pipeline: add one sign and every
|
|
31
|
+
index above it shifts, so the model predicts confidently and wrongly.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(self, glosses: list[str]):
|
|
35
|
+
self.glosses = list(glosses)
|
|
36
|
+
self._to_idx = {g: i for i, g in enumerate(self.glosses)}
|
|
37
|
+
|
|
38
|
+
@classmethod
|
|
39
|
+
def from_directory(cls, root: str | Path) -> LabelMap:
|
|
40
|
+
root = Path(root)
|
|
41
|
+
glosses = sorted(
|
|
42
|
+
p.name for p in root.iterdir() if p.is_dir() and not p.name.startswith(".")
|
|
43
|
+
)
|
|
44
|
+
return cls(glosses)
|
|
45
|
+
|
|
46
|
+
def encode(self, gloss: str) -> int:
|
|
47
|
+
return self._to_idx[gloss]
|
|
48
|
+
|
|
49
|
+
def decode(self, index: int) -> str:
|
|
50
|
+
return self.glosses[index]
|
|
51
|
+
|
|
52
|
+
def save(self, path: str | Path) -> None:
|
|
53
|
+
Path(path).write_text(
|
|
54
|
+
json.dumps({"glosses": self.glosses, "n_classes": len(self.glosses)}, indent=2)
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def load(cls, path: str | Path) -> LabelMap:
|
|
59
|
+
return cls(json.loads(Path(path).read_text())["glosses"])
|
|
60
|
+
|
|
61
|
+
def __len__(self) -> int:
|
|
62
|
+
return len(self.glosses)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def read_take(path: str | Path) -> list[RawFrame]:
|
|
66
|
+
"""One take_NN.npz -> list[RawFrame], the layout `ClipStore.save` writes.
|
|
67
|
+
|
|
68
|
+
Split out of `build_dataset` so a leave-one-session-out eval can read a
|
|
69
|
+
single held-out take's raw frames straight through `SignRecogniser`, the
|
|
70
|
+
exact path live inference uses, instead of the pre-encoded arrays
|
|
71
|
+
`build_dataset` returns.
|
|
72
|
+
"""
|
|
73
|
+
d = np.load(Path(path), allow_pickle=True)
|
|
74
|
+
n = len(d["pose"])
|
|
75
|
+
return [
|
|
76
|
+
RawFrame(
|
|
77
|
+
pose=d["pose"][i] if d["pose"][i] is not None else None,
|
|
78
|
+
face=d["face"][i] if d["face"][i] is not None else None,
|
|
79
|
+
hand_left=d["hand_left"][i],
|
|
80
|
+
hand_right=d["hand_right"][i],
|
|
81
|
+
)
|
|
82
|
+
for i in range(n)
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def build_dataset(root: str | Path, T: int = 48, dominant: str = "right"):
|
|
87
|
+
"""Read data/<GLOSS>/<session>/take_NN.npz -> X, y, sessions, label_map.
|
|
88
|
+
|
|
89
|
+
Each .npz holds raw landmark arrays saved at capture time, never the
|
|
90
|
+
encoded features — normalisation will change several times and
|
|
91
|
+
re-recording is not an option.
|
|
92
|
+
"""
|
|
93
|
+
root = Path(root)
|
|
94
|
+
label_map = LabelMap.from_directory(root)
|
|
95
|
+
X, y, sessions = [], [], []
|
|
96
|
+
|
|
97
|
+
for gloss in label_map.glosses:
|
|
98
|
+
for take in sorted((root / gloss).rglob("*.npz")):
|
|
99
|
+
frames = read_take(take)
|
|
100
|
+
X.append(encode_clip(frames, T=T, dominant=dominant))
|
|
101
|
+
y.append(label_map.encode(gloss))
|
|
102
|
+
# Session id drives leave-one-session-out splitting. A random
|
|
103
|
+
# split leaks near-duplicate frames and inflates accuracy by
|
|
104
|
+
# 10-15 points.
|
|
105
|
+
sessions.append(take.parent.name)
|
|
106
|
+
|
|
107
|
+
return (np.stack(X), np.array(y, np.int64), np.array(sessions), label_map)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# --------------------------------------------------------------------------
|
|
111
|
+
# Fine-tuning data assembly
|
|
112
|
+
# --------------------------------------------------------------------------
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def remap_labels(y: np.ndarray, source_map: LabelMap, target_map: LabelMap) -> np.ndarray:
|
|
116
|
+
"""Re-encode `y` from `source_map`'s class order onto `target_map`'s.
|
|
117
|
+
|
|
118
|
+
Two label maps only agree on an index by coincidence — each is built by
|
|
119
|
+
sorting whatever glosses its own source happens to contain, so index 3 in
|
|
120
|
+
one is not index 3 in the other. Every remap in this pipeline has to go
|
|
121
|
+
through the gloss string, never the position (an unfrozen label map's failure mode is a
|
|
122
|
+
silent index shift; this is the same bug one hop removed from a directory
|
|
123
|
+
listing).
|
|
124
|
+
|
|
125
|
+
Raises if `y` contains a class `target_map` does not have: a class that
|
|
126
|
+
quietly has no target slot must not train as index noise.
|
|
127
|
+
"""
|
|
128
|
+
glosses = [source_map.decode(int(c)) for c in y]
|
|
129
|
+
unknown = sorted(set(glosses) - set(target_map.glosses))
|
|
130
|
+
if unknown:
|
|
131
|
+
raise ValueError(f"labels outside the target vocabulary {target_map.glosses}: {unknown}")
|
|
132
|
+
return np.array([target_map.encode(g) for g in glosses], dtype=np.int64)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def pool_finetune_data(
|
|
136
|
+
include: IncludeData,
|
|
137
|
+
own_X: np.ndarray,
|
|
138
|
+
own_y: np.ndarray,
|
|
139
|
+
own_label_map: LabelMap,
|
|
140
|
+
target_label_map: LabelMap,
|
|
141
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
142
|
+
"""INCLUDE clips (filtered to `target_label_map`) plus own recordings, all
|
|
143
|
+
on `target_label_map`'s class order. Stage 1's training set.
|
|
144
|
+
|
|
145
|
+
Safe to pool INCLUDE here specifically because stage 1 freezes the
|
|
146
|
+
backbone: a
|
|
147
|
+
frozen backbone cannot re-anchor to INCLUDE's signer mix, it only gives
|
|
148
|
+
the new head more label-aligned examples in the pretrained feature space.
|
|
149
|
+
Stage 2 must NOT reuse this — it fine-tunes on own recordings alone.
|
|
150
|
+
"""
|
|
151
|
+
own_target_y = remap_labels(own_y, own_label_map, target_label_map)
|
|
152
|
+
|
|
153
|
+
target_set = set(target_label_map.glosses)
|
|
154
|
+
include_glosses = np.array([include.label_map.decode(int(c)) for c in include.y])
|
|
155
|
+
keep = np.array([g in target_set for g in include_glosses])
|
|
156
|
+
include_target_y = np.array(
|
|
157
|
+
[target_label_map.encode(g) for g in include_glosses[keep]], dtype=np.int64
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
X = np.concatenate([include.X[keep], own_X], axis=0)
|
|
161
|
+
y = np.concatenate([include_target_y, own_target_y], axis=0)
|
|
162
|
+
return X, y
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# --------------------------------------------------------------------------
|
|
166
|
+
# Inference
|
|
167
|
+
# --------------------------------------------------------------------------
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def decode_prediction(
|
|
171
|
+
probs: np.ndarray, label_map: LabelMap, threshold: float = 0.6
|
|
172
|
+
) -> tuple[str | None, float]:
|
|
173
|
+
"""Softmax vector -> (gloss, confidence). None below threshold.
|
|
174
|
+
|
|
175
|
+
Returning None is a feature. A device that admits uncertainty is more
|
|
176
|
+
trustworthy than one that guesses at a stranger on the user's behalf.
|
|
177
|
+
"""
|
|
178
|
+
idx = int(np.argmax(probs))
|
|
179
|
+
conf = float(probs[idx])
|
|
180
|
+
return (label_map.decode(idx) if conf >= threshold else None), conf
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
class VoteBuffer:
|
|
184
|
+
"""2-of-3 agreement before emitting. Kills flicker for ~100 ms latency."""
|
|
185
|
+
|
|
186
|
+
def __init__(self, window: int = 3, agree: int = 2):
|
|
187
|
+
self.window, self.agree, self.buf = window, agree, []
|
|
188
|
+
|
|
189
|
+
def push(self, gloss: str | None) -> str | None:
|
|
190
|
+
self.buf.append(gloss)
|
|
191
|
+
if len(self.buf) > self.window:
|
|
192
|
+
self.buf.pop(0)
|
|
193
|
+
for g in set(self.buf):
|
|
194
|
+
if g is not None and self.buf.count(g) >= self.agree:
|
|
195
|
+
self.buf.clear()
|
|
196
|
+
return g
|
|
197
|
+
return None
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def normalise_gloss(text: str) -> str | None:
|
|
201
|
+
"""Fold a typed label to the dump's convention: lowercase, alphanumeric only.
|
|
202
|
+
|
|
203
|
+
Folder names become class labels, so a label typed with a space or a capital
|
|
204
|
+
is a different class from the one intended — "Good morning" is not
|
|
205
|
+
"goodmorning", and the mistake produces no error at all. It costs twice: the
|
|
206
|
+
INCLUDE clips for that sign can no longer be pooled with the new recordings,
|
|
207
|
+
and two people spelling it differently produce two classes for one sign.
|
|
208
|
+
|
|
209
|
+
Returns None when nothing survives, so an empty prompt is an absent label
|
|
210
|
+
rather than a class named "".
|
|
211
|
+
"""
|
|
212
|
+
cleaned = "".join(ch for ch in text.lower() if ch.isalnum())
|
|
213
|
+
return cleaned or None
|
islkit/metrics.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Classification metrics, written out rather than imported from sklearn.
|
|
2
|
+
|
|
3
|
+
Aggregate accuracy hides a class that never works. Per-class F1 is the metric
|
|
4
|
+
that tells you the truth, so it lives here and gets printed by default.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _as_numpy(a) -> np.ndarray:
|
|
11
|
+
"""Accept torch tensors (on any device) or anything array-like."""
|
|
12
|
+
if hasattr(a, "detach"):
|
|
13
|
+
a = a.detach().cpu()
|
|
14
|
+
return np.asarray(a)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def accuracy(pred, true) -> float:
|
|
18
|
+
pred, true = _as_numpy(pred), _as_numpy(true)
|
|
19
|
+
if len(true) == 0:
|
|
20
|
+
return 0.0
|
|
21
|
+
return float((pred == true).mean())
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def per_class_f1(pred, true, n_classes: int) -> list[float]:
|
|
25
|
+
"""F1 for each class index in range(n_classes).
|
|
26
|
+
|
|
27
|
+
A class with no predictions and no true labels scores 0.0 rather than
|
|
28
|
+
raising — that case shows up whenever a split misses a rare class.
|
|
29
|
+
"""
|
|
30
|
+
pred, true = _as_numpy(pred), _as_numpy(true)
|
|
31
|
+
scores = []
|
|
32
|
+
for c in range(n_classes):
|
|
33
|
+
tp = int(np.sum((pred == c) & (true == c)))
|
|
34
|
+
fp = int(np.sum((pred == c) & (true != c)))
|
|
35
|
+
fn = int(np.sum((pred != c) & (true == c)))
|
|
36
|
+
denom = 2 * tp + fp + fn
|
|
37
|
+
scores.append(2 * tp / denom if denom else 0.0)
|
|
38
|
+
return scores
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def format_per_class_f1(pred, true, n_classes: int, names: list[str] | None = None) -> str:
|
|
42
|
+
scores = per_class_f1(pred, true, n_classes)
|
|
43
|
+
labels = names or [f"class {c}" for c in range(n_classes)]
|
|
44
|
+
width = max(len(label) for label in labels)
|
|
45
|
+
return "\n".join(
|
|
46
|
+
f" {label:<{width}} : {s:.3f}" for label, s in zip(labels, scores, strict=True)
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def confusion_matrix(pred, true, n_classes: int) -> np.ndarray:
|
|
51
|
+
"""(n_classes, n_classes) counts, rows = true class, columns = predicted."""
|
|
52
|
+
pred, true = _as_numpy(pred).astype(int), _as_numpy(true).astype(int)
|
|
53
|
+
m = np.zeros((n_classes, n_classes), dtype=np.int64)
|
|
54
|
+
np.add.at(m, (true, pred), 1)
|
|
55
|
+
return m
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def top_confusions(
|
|
59
|
+
cm: np.ndarray, names: list[str] | None = None, k: int = 25
|
|
60
|
+
) -> list[tuple[str, str, int, float]]:
|
|
61
|
+
"""The k worst off-diagonal cells: (true, predicted, count, share of true).
|
|
62
|
+
|
|
63
|
+
A 262x262 confusion image is a texture, not a finding. This is the part of
|
|
64
|
+
it anyone can act on — which specific pairs the model cannot tell apart,
|
|
65
|
+
which is also how the demo vocabulary gets pruned of near-collisions.
|
|
66
|
+
"""
|
|
67
|
+
cm = np.asarray(cm)
|
|
68
|
+
labels = names or [f"class {c}" for c in range(len(cm))]
|
|
69
|
+
off = cm.copy()
|
|
70
|
+
np.fill_diagonal(off, 0)
|
|
71
|
+
support = cm.sum(axis=1)
|
|
72
|
+
|
|
73
|
+
flat = np.argsort(off, axis=None)[::-1][:k]
|
|
74
|
+
out = []
|
|
75
|
+
for f in flat:
|
|
76
|
+
t, p = int(f // len(cm)), int(f % len(cm))
|
|
77
|
+
if off[t, p] == 0:
|
|
78
|
+
break
|
|
79
|
+
share = float(off[t, p] / support[t]) if support[t] else 0.0
|
|
80
|
+
out.append((labels[t], labels[p], int(off[t, p]), share))
|
|
81
|
+
return out
|