islkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- islkit/__init__.py +88 -0
- islkit/adapters.py +237 -0
- islkit/baseline.py +287 -0
- islkit/data.py +320 -0
- islkit/device.py +38 -0
- islkit/domain.py +187 -0
- islkit/features.py +323 -0
- islkit/infer.py +911 -0
- islkit/labels.py +213 -0
- islkit/metrics.py +81 -0
- islkit/model.py +623 -0
- islkit/pipeline.py +717 -0
- islkit/plotting.py +131 -0
- islkit/seeding.py +19 -0
- islkit/server.py +246 -0
- islkit/view.py +287 -0
- islkit/viz.py +435 -0
- islkit-0.1.0.dist-info/METADATA +200 -0
- islkit-0.1.0.dist-info/RECORD +21 -0
- islkit-0.1.0.dist-info/WHEEL +4 -0
- islkit-0.1.0.dist-info/licenses/LICENSE +21 -0
islkit/pipeline.py
ADDED
|
@@ -0,0 +1,717 @@
|
|
|
1
|
+
"""Headless capture: camera frames in, event dicts out. No HTTP in this file.
|
|
2
|
+
|
|
3
|
+
This composes HolisticExtractor, AutoTake, SignRecogniser, ClipStore and
|
|
4
|
+
check_tracking into a loop with no window and no keyboard, because an embedded
|
|
5
|
+
board has neither.
|
|
6
|
+
|
|
7
|
+
`cv2` is imported inside `CameraSource.__init__` so this module can be imported
|
|
8
|
+
— and tested — without a camera, without OpenCV and without MediaPipe.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
import shutil
|
|
15
|
+
import threading
|
|
16
|
+
import time
|
|
17
|
+
from collections import deque
|
|
18
|
+
from dataclasses import asdict, dataclass
|
|
19
|
+
from typing import Protocol
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
|
|
23
|
+
from islkit.infer import (
|
|
24
|
+
AutoTake,
|
|
25
|
+
ClipTooShort,
|
|
26
|
+
Prediction,
|
|
27
|
+
TakeQuality,
|
|
28
|
+
Tracking,
|
|
29
|
+
check_tracking,
|
|
30
|
+
take_quality,
|
|
31
|
+
)
|
|
32
|
+
from islkit.view import annotate
|
|
33
|
+
|
|
34
|
+
log = logging.getLogger("islkit.pipeline")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class CameraUnavailable(RuntimeError):
|
|
38
|
+
"""The camera would not open, or opened and delivered nothing.
|
|
39
|
+
|
|
40
|
+
A RuntimeError rather than SystemExit on purpose: live_demo can exit here
|
|
41
|
+
because a person is watching it, but this is a device service and the
|
|
42
|
+
caller's job is to emit a fault, back off and retry.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# AutoTake's defaults were tuned by hand at this rate on a Mac. Everything below
|
|
47
|
+
# is expressed as a ratio against it rather than as seconds, so the conversion is
|
|
48
|
+
# exactly the identity here — seconds would reintroduce float drift (0.96 * 12.5
|
|
49
|
+
# is 12.000000000000002, and a ceil of that is 13).
|
|
50
|
+
REFERENCE_FPS = 12.5
|
|
51
|
+
_REFERENCE_COUNTS = {
|
|
52
|
+
"pre_roll": 12, # 0.96 s of leading rest seeded into every take
|
|
53
|
+
"rest_to_arm": 6, # 0.48 s at rest before we will accept a sign
|
|
54
|
+
"raised_to_start": 6, # 0.48 s raised before a take opens
|
|
55
|
+
"rest_to_close": 8, # 0.64 s at rest closes the take
|
|
56
|
+
"max_frames": 400, # 32 s runaway cap
|
|
57
|
+
}
|
|
58
|
+
# A one-frame run is a MediaPipe dropout, not a state change.
|
|
59
|
+
_FLOORS = {"pre_roll": 1, "max_frames": 30}
|
|
60
|
+
_DEFAULT_FLOOR = 2
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen=True)
|
|
64
|
+
class TakeTiming:
|
|
65
|
+
"""AutoTake's thresholds at some particular frame rate."""
|
|
66
|
+
|
|
67
|
+
pre_roll: int
|
|
68
|
+
rest_to_arm: int
|
|
69
|
+
raised_to_start: int
|
|
70
|
+
rest_to_close: int
|
|
71
|
+
max_frames: int
|
|
72
|
+
|
|
73
|
+
def as_dict(self) -> dict:
|
|
74
|
+
return asdict(self)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def derive_timing(fps: float) -> TakeTiming:
|
|
78
|
+
"""Scale AutoTake's Mac-tuned frame counts to the rate actually observed.
|
|
79
|
+
|
|
80
|
+
At 3 fps a literal `rest_to_arm=6` becomes a two-second wait before the
|
|
81
|
+
device will accept a sign, and a literal `max_frames=400` lets someone who
|
|
82
|
+
raises their hands and holds still record for over two minutes.
|
|
83
|
+
|
|
84
|
+
`min_frames` is deliberately absent. It is a resampling floor rather than a
|
|
85
|
+
duration — `encode_clip` resamples instead of padding, so too few frames
|
|
86
|
+
would classify without complaint — and picking a board value for it without
|
|
87
|
+
the board's real rate would be a guess dressed as a decision.
|
|
88
|
+
"""
|
|
89
|
+
ratio = max(fps, 0.1) / REFERENCE_FPS
|
|
90
|
+
counts = {
|
|
91
|
+
name: max(_FLOORS.get(name, _DEFAULT_FLOOR), round(reference * ratio))
|
|
92
|
+
for name, reference in _REFERENCE_COUNTS.items()
|
|
93
|
+
}
|
|
94
|
+
return TakeTiming(**counts)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class FpsMeter:
|
|
98
|
+
"""Rolling frame rate over the last `window` frames.
|
|
99
|
+
|
|
100
|
+
Reported rather than assumed: `/health` publishing a real measured rate is
|
|
101
|
+
how a 2 fps board becomes visible instead of inferred.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
def __init__(self, window: int = 60):
|
|
105
|
+
self._times: deque[float] = deque(maxlen=window)
|
|
106
|
+
|
|
107
|
+
def tick(self, now: float | None = None) -> None:
|
|
108
|
+
self._times.append(time.perf_counter() if now is None else now)
|
|
109
|
+
|
|
110
|
+
@property
|
|
111
|
+
def fps(self) -> float:
|
|
112
|
+
if len(self._times) < 2:
|
|
113
|
+
return 0.0
|
|
114
|
+
span = self._times[-1] - self._times[0]
|
|
115
|
+
return (len(self._times) - 1) / span if span > 0 else 0.0
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class FrameSource(Protocol):
|
|
119
|
+
"""Where frames come from. One implementation reads a camera, one is a test."""
|
|
120
|
+
|
|
121
|
+
def read(self) -> np.ndarray | None:
|
|
122
|
+
"""The next frame, or None if the source is not delivering."""
|
|
123
|
+
|
|
124
|
+
def close(self) -> None: ...
|
|
125
|
+
|
|
126
|
+
def describe(self) -> dict:
|
|
127
|
+
"""Identity and geometry, for /health and for every take's sidecar."""
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class CameraSource:
|
|
131
|
+
"""cv2.VideoCapture, with live_demo's hard-won opening discipline.
|
|
132
|
+
|
|
133
|
+
`isOpened()` is not enough: a device can enumerate, report a resolution and
|
|
134
|
+
still return nothing from every read() — which is what a disconnected or
|
|
135
|
+
in-use webcam looks like. The only honest check is to pull a frame.
|
|
136
|
+
"""
|
|
137
|
+
|
|
138
|
+
def __init__(self, index: int = 0, resolution: str = "1280x720"):
|
|
139
|
+
import cv2
|
|
140
|
+
|
|
141
|
+
self.index = index
|
|
142
|
+
self._cap = cv2.VideoCapture(index)
|
|
143
|
+
if not self._cap.isOpened() or not any(self._cap.read()[0] for _ in range(5)):
|
|
144
|
+
self._cap.release()
|
|
145
|
+
raise CameraUnavailable(f"camera {index} opens but returns no frames")
|
|
146
|
+
|
|
147
|
+
want_w, want_h = (int(v) for v in resolution.lower().split("x"))
|
|
148
|
+
self._cap.set(cv2.CAP_PROP_FRAME_WIDTH, want_w)
|
|
149
|
+
self._cap.set(cv2.CAP_PROP_FRAME_HEIGHT, want_h)
|
|
150
|
+
self.width = int(self._cap.get(cv2.CAP_PROP_FRAME_WIDTH))
|
|
151
|
+
self.height = int(self._cap.get(cv2.CAP_PROP_FRAME_HEIGHT))
|
|
152
|
+
# Reported, never enforced. Aspect changes the encoded geometry, so takes
|
|
153
|
+
# recorded at different aspects are not comparable — but refusing to start
|
|
154
|
+
# over it would be worse than saying so.
|
|
155
|
+
self.requested = (want_w, want_h)
|
|
156
|
+
|
|
157
|
+
def read(self) -> np.ndarray | None:
|
|
158
|
+
ok, frame = self._cap.read()
|
|
159
|
+
return frame if ok else None
|
|
160
|
+
|
|
161
|
+
def close(self) -> None:
|
|
162
|
+
self._cap.release()
|
|
163
|
+
|
|
164
|
+
def describe(self) -> dict:
|
|
165
|
+
return {
|
|
166
|
+
"kind": "camera",
|
|
167
|
+
"index": self.index,
|
|
168
|
+
"width": self.width,
|
|
169
|
+
"height": self.height,
|
|
170
|
+
"aspect": round(self.width / self.height, 4) if self.height else 0.0,
|
|
171
|
+
"requested": f"{self.requested[0]}x{self.requested[1]}",
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class FakeFrameSource:
|
|
176
|
+
"""Frames that are not pictures of anything, for tests.
|
|
177
|
+
|
|
178
|
+
The pipeline hands whatever this returns straight to the extractor, and tests
|
|
179
|
+
inject an extractor that ignores it, so the content is irrelevant and the
|
|
180
|
+
shape is only there to keep `describe()` honest.
|
|
181
|
+
"""
|
|
182
|
+
|
|
183
|
+
def __init__(self, n_frames: int | None = None, shape: tuple[int, int] = (4, 4)):
|
|
184
|
+
self.n_frames = n_frames
|
|
185
|
+
self.shape = shape
|
|
186
|
+
self.served = 0
|
|
187
|
+
self.closed = False
|
|
188
|
+
|
|
189
|
+
def read(self) -> np.ndarray | None:
|
|
190
|
+
if self.n_frames is not None and self.served >= self.n_frames:
|
|
191
|
+
return None
|
|
192
|
+
self.served += 1
|
|
193
|
+
return np.zeros((*self.shape, 3), np.uint8)
|
|
194
|
+
|
|
195
|
+
def close(self) -> None:
|
|
196
|
+
self.closed = True
|
|
197
|
+
|
|
198
|
+
def describe(self) -> dict:
|
|
199
|
+
return {"kind": "fake", "height": self.shape[0], "width": self.shape[1]}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
# check_tracking words its verdicts for a human ("no pose", "Hands not visible").
|
|
203
|
+
# The wire carries a stable machine enum instead: the English belongs to the
|
|
204
|
+
# orchestrator's phrase table, because it is display language and this package must
|
|
205
|
+
# not be in the path of a wording or translation change.
|
|
206
|
+
TRACKING_STATUS = {
|
|
207
|
+
"no pose": "absent",
|
|
208
|
+
"clipped": "clipped",
|
|
209
|
+
"no hands": "hands_hidden",
|
|
210
|
+
"ok": "ok",
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class TrackingDebouncer:
|
|
215
|
+
"""Turns per-frame tracking verdicts into rare, stable status changes.
|
|
216
|
+
|
|
217
|
+
MediaPipe loses hands constantly — especially held still against the body —
|
|
218
|
+
so an un-debounced status would flap every few frames and strobe whatever is
|
|
219
|
+
rendering it. A new status has to hold before it is believed.
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
def __init__(self, hold_frames: int):
|
|
223
|
+
self.hold_frames = max(1, hold_frames)
|
|
224
|
+
self.status: str | None = None # last status actually emitted
|
|
225
|
+
self.raw: str | None = None # what the current run is of
|
|
226
|
+
self.run = 0 # how many consecutive frames it has held
|
|
227
|
+
|
|
228
|
+
def update(self, tracking: Tracking) -> str | None:
|
|
229
|
+
"""Feed one frame's verdict. Returns a status only when it changes."""
|
|
230
|
+
status = TRACKING_STATUS[tracking.status]
|
|
231
|
+
self.run = self.run + 1 if status == self.raw else 1
|
|
232
|
+
self.raw = status
|
|
233
|
+
|
|
234
|
+
if status != self.status and self.run >= self.hold_frames:
|
|
235
|
+
self.status = status
|
|
236
|
+
return status
|
|
237
|
+
return None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class RecognitionPipeline:
|
|
241
|
+
"""The capture loop. Camera in, event dicts out.
|
|
242
|
+
|
|
243
|
+
`classify()` runs inline on this thread: it costs ~34 ms and happens once per
|
|
244
|
+
deliberate gesture, so serialising it is free and keeps torch on exactly one
|
|
245
|
+
thread. Events leave through `on_event`, which is called from this thread and
|
|
246
|
+
must not block — the EventHub on the other end is built for that.
|
|
247
|
+
"""
|
|
248
|
+
|
|
249
|
+
def __init__(
|
|
250
|
+
self,
|
|
251
|
+
recogniser,
|
|
252
|
+
source_factory,
|
|
253
|
+
on_event,
|
|
254
|
+
*,
|
|
255
|
+
extractor_factory=None,
|
|
256
|
+
store=None,
|
|
257
|
+
warmup_frames: int = 30,
|
|
258
|
+
start_active: bool = False,
|
|
259
|
+
classifier_name: str = "",
|
|
260
|
+
classifier_sha256: str = "",
|
|
261
|
+
min_free_mb: int = 500,
|
|
262
|
+
drop_limit: int | None = None,
|
|
263
|
+
reopen_backoff: tuple[float, ...] = (1.0, 2.0, 4.0, 8.0),
|
|
264
|
+
max_reopen_attempts: int | None = None,
|
|
265
|
+
on_pause=None,
|
|
266
|
+
view=None,
|
|
267
|
+
):
|
|
268
|
+
self._recogniser = recogniser
|
|
269
|
+
self._source_factory = source_factory
|
|
270
|
+
self._on_event = on_event
|
|
271
|
+
self._extractor_factory = extractor_factory or self._default_extractor
|
|
272
|
+
self._store = store
|
|
273
|
+
# Called with no arguments when capture pauses. The pipeline does not
|
|
274
|
+
# know what an EventHub is — serve.py binds this to hub.clear_take so
|
|
275
|
+
# the retained take-state slot doesn't outlive the take it describes.
|
|
276
|
+
self._on_pause = on_pause
|
|
277
|
+
self._warmup_frames = max(1, warmup_frames)
|
|
278
|
+
self._classifier_name = classifier_name
|
|
279
|
+
self._classifier_sha256 = classifier_sha256
|
|
280
|
+
|
|
281
|
+
self.dominant = recogniser.dominant
|
|
282
|
+
self._lock = threading.Lock()
|
|
283
|
+
self._capture = start_active
|
|
284
|
+
self._stop = threading.Event()
|
|
285
|
+
|
|
286
|
+
self._source = None
|
|
287
|
+
self._extractor = None
|
|
288
|
+
self._fps = FpsMeter()
|
|
289
|
+
self._timing: TakeTiming | None = None
|
|
290
|
+
self._auto: AutoTake | None = None
|
|
291
|
+
self._tracker: TrackingDebouncer | None = None
|
|
292
|
+
self._take_state: str | None = None
|
|
293
|
+
self._started = time.monotonic()
|
|
294
|
+
self._frames_seen = 0
|
|
295
|
+
self._min_free_mb = min_free_mb
|
|
296
|
+
self._drop_limit = drop_limit
|
|
297
|
+
self._reopen_backoff = reopen_backoff
|
|
298
|
+
self._max_reopen_attempts = max_reopen_attempts
|
|
299
|
+
self._fault: dict | None = None
|
|
300
|
+
self._absent_disarmed = False
|
|
301
|
+
# Optional ViewSink. None, or nobody watching, means the frame path
|
|
302
|
+
# below costs one boolean read. cv2 stays lazily imported for the same
|
|
303
|
+
# reason CameraSource defers it: this module must import without OpenCV.
|
|
304
|
+
self._view = view
|
|
305
|
+
self._cv2 = None
|
|
306
|
+
self._drops = 0
|
|
307
|
+
|
|
308
|
+
@staticmethod
|
|
309
|
+
def _default_extractor():
|
|
310
|
+
from islkit.infer import HolisticExtractor
|
|
311
|
+
|
|
312
|
+
return HolisticExtractor()
|
|
313
|
+
|
|
314
|
+
# -- control surface, called from the HTTP threads ---------------------
|
|
315
|
+
|
|
316
|
+
def set_capture(self, active: bool) -> bool:
|
|
317
|
+
"""Returns the state now in force.
|
|
318
|
+
|
|
319
|
+
Pausing disarms in the same locked step. Setting the flag alone would
|
|
320
|
+
leave a half-finished take in the buffer, which then completes when
|
|
321
|
+
capture resumes — a sign nobody made, attributed to whoever is in frame.
|
|
322
|
+
"""
|
|
323
|
+
with self._lock:
|
|
324
|
+
self._capture = bool(active)
|
|
325
|
+
paused = not self._capture
|
|
326
|
+
if paused and self._auto is not None:
|
|
327
|
+
self._auto.disarm()
|
|
328
|
+
self._take_state = None
|
|
329
|
+
in_force = self._capture
|
|
330
|
+
|
|
331
|
+
# Outside the lock deliberately: on_pause (bound to hub.clear_take)
|
|
332
|
+
# must not be called while holding this non-reentrant lock, and
|
|
333
|
+
# in_force is a local snapshot rather than a second read of
|
|
334
|
+
# self._capture, which a concurrent set_capture() could have already
|
|
335
|
+
# moved on from.
|
|
336
|
+
if paused and self._on_pause is not None:
|
|
337
|
+
self._on_pause()
|
|
338
|
+
return in_force
|
|
339
|
+
|
|
340
|
+
def stop(self) -> None:
|
|
341
|
+
self._stop.set()
|
|
342
|
+
|
|
343
|
+
def health(self) -> dict:
|
|
344
|
+
# Measured before the lock: self._store never changes after
|
|
345
|
+
# construction, and a disk_usage() syscall has no business holding a
|
|
346
|
+
# lock the capture thread also wants.
|
|
347
|
+
disk_free_mb = self._measure_disk_free_mb()
|
|
348
|
+
with self._lock:
|
|
349
|
+
# Bound to locals rather than read twice (once for the truthiness
|
|
350
|
+
# check, once for the attribute access): run() and _note_drop()
|
|
351
|
+
# both set self._source/self._tracker to None from the capture
|
|
352
|
+
# thread without this lock, so a poll landing between the two
|
|
353
|
+
# reads used to see a live object turn into None mid-expression
|
|
354
|
+
# and raise AttributeError instead of a clean 500.
|
|
355
|
+
source = self._source
|
|
356
|
+
tracker = self._tracker
|
|
357
|
+
return {
|
|
358
|
+
"ok": self._fault is None,
|
|
359
|
+
"capture": self._capture,
|
|
360
|
+
"fps": round(self._fps.fps, 2),
|
|
361
|
+
"classes": len(self._recogniser.label_map),
|
|
362
|
+
"classifier": self._classifier_name,
|
|
363
|
+
"sha256": self._classifier_sha256,
|
|
364
|
+
"tracking": tracker.status if tracker else None,
|
|
365
|
+
"state": self._take_state,
|
|
366
|
+
"uptime_s": round(time.monotonic() - self._started, 1),
|
|
367
|
+
"frames": self._frames_seen,
|
|
368
|
+
"source": source.describe() if source else None,
|
|
369
|
+
"autotake": self._timing.as_dict() if self._timing else None,
|
|
370
|
+
"min_frames": self._recogniser.min_frames,
|
|
371
|
+
"threshold": self._recogniser.threshold,
|
|
372
|
+
"disk_free_mb": disk_free_mb,
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
# -- the loop ----------------------------------------------------------
|
|
376
|
+
|
|
377
|
+
def run(self) -> None:
|
|
378
|
+
"""Loop until stop(). A camera or model failure is a fault and a retry,
|
|
379
|
+
never an exit — that includes building the extractor itself. A missing
|
|
380
|
+
MediaPipe asset or a bad wheel raises out of the factory exactly like a
|
|
381
|
+
dead camera raises out of source_factory, and must be survived the same
|
|
382
|
+
way: emit `fault model_error`, latch, back off, retry."""
|
|
383
|
+
attempts = 0 # backoff-ladder position: resets on a healthy reopen
|
|
384
|
+
total_attempts = 0 # give-up counter: never resets within one run()
|
|
385
|
+
# The extractor gets its own ladder position and give-up counter,
|
|
386
|
+
# independent of the source's: they open at different times (the
|
|
387
|
+
# extractor once, up front; the source on every camera loss) and
|
|
388
|
+
# sharing one counter would make an extractor retry eat into the
|
|
389
|
+
# source's retry budget for no reason.
|
|
390
|
+
ext_attempts = 0
|
|
391
|
+
ext_total_attempts = 0
|
|
392
|
+
try:
|
|
393
|
+
while not self._stop.is_set():
|
|
394
|
+
if self._extractor is None:
|
|
395
|
+
if self._max_reopen_attempts is not None:
|
|
396
|
+
if ext_total_attempts >= self._max_reopen_attempts:
|
|
397
|
+
return
|
|
398
|
+
ext_attempts += 1
|
|
399
|
+
ext_total_attempts += 1
|
|
400
|
+
try:
|
|
401
|
+
self._extractor = self._extractor_factory()
|
|
402
|
+
self._fault = None
|
|
403
|
+
ext_attempts = 0
|
|
404
|
+
except Exception as exc: # noqa: BLE001 - a bad wheel or a
|
|
405
|
+
# missing model asset must not be able to exit this thread.
|
|
406
|
+
self._fault_out("model_error", f"{type(exc).__name__}: {exc}")
|
|
407
|
+
self._backoff(ext_attempts)
|
|
408
|
+
continue
|
|
409
|
+
|
|
410
|
+
if self._source is None:
|
|
411
|
+
if self._max_reopen_attempts is not None:
|
|
412
|
+
if total_attempts >= self._max_reopen_attempts:
|
|
413
|
+
return
|
|
414
|
+
attempts += 1
|
|
415
|
+
total_attempts += 1
|
|
416
|
+
try:
|
|
417
|
+
self._source = self._source_factory()
|
|
418
|
+
self._fault = None
|
|
419
|
+
attempts = 0 # a healthy reopen must not pin the backoff ladder
|
|
420
|
+
except CameraUnavailable as exc:
|
|
421
|
+
self._fault_out("camera_open_failed", str(exc))
|
|
422
|
+
self._backoff(attempts)
|
|
423
|
+
continue
|
|
424
|
+
|
|
425
|
+
try:
|
|
426
|
+
image = self._source.read()
|
|
427
|
+
if image is None:
|
|
428
|
+
if self._note_drop():
|
|
429
|
+
self._backoff(attempts)
|
|
430
|
+
continue
|
|
431
|
+
|
|
432
|
+
self._drops = 0
|
|
433
|
+
self._on_frame(image)
|
|
434
|
+
except Exception as exc: # noqa: BLE001 - nothing here may exit
|
|
435
|
+
# A cv2 read or a MediaPipe process() that starts raising has
|
|
436
|
+
# stopped working, same as a camera that stopped returning
|
|
437
|
+
# frames — reuse camera_lost rather than inventing a fourth code.
|
|
438
|
+
self._fault_out("camera_lost", f"{type(exc).__name__}: {exc}")
|
|
439
|
+
if self._auto is not None:
|
|
440
|
+
with self._lock:
|
|
441
|
+
self._auto.disarm()
|
|
442
|
+
self._take_state = None
|
|
443
|
+
self._close_source_quietly()
|
|
444
|
+
self._drops = 0
|
|
445
|
+
self._backoff(attempts)
|
|
446
|
+
continue
|
|
447
|
+
finally:
|
|
448
|
+
try:
|
|
449
|
+
if self._extractor is not None:
|
|
450
|
+
self._extractor.close()
|
|
451
|
+
finally:
|
|
452
|
+
if self._source is not None:
|
|
453
|
+
self._source.close()
|
|
454
|
+
|
|
455
|
+
def _note_drop(self) -> bool:
|
|
456
|
+
"""Returns True once the drop run is long enough to call the camera lost."""
|
|
457
|
+
self._drops += 1
|
|
458
|
+
limit = self._drop_limit if self._drop_limit is not None else self._current_drop_limit()
|
|
459
|
+
if self._drops < limit:
|
|
460
|
+
return False
|
|
461
|
+
self._fault_out(
|
|
462
|
+
"camera_lost",
|
|
463
|
+
f"camera stopped returning frames after {self._frames_seen} frames",
|
|
464
|
+
)
|
|
465
|
+
if self._auto is not None:
|
|
466
|
+
with self._lock:
|
|
467
|
+
self._auto.disarm()
|
|
468
|
+
self._take_state = None
|
|
469
|
+
self._close_source_quietly()
|
|
470
|
+
self._drops = 0
|
|
471
|
+
return True
|
|
472
|
+
|
|
473
|
+
def _close_source_quietly(self) -> None:
|
|
474
|
+
"""close() itself raising must not be able to exit run() either — the
|
|
475
|
+
same governing rule as the read/process failure this is usually called
|
|
476
|
+
from, just on a colder path."""
|
|
477
|
+
source, self._source = self._source, None
|
|
478
|
+
if source is None:
|
|
479
|
+
return
|
|
480
|
+
try:
|
|
481
|
+
source.close()
|
|
482
|
+
except Exception as exc: # noqa: BLE001 - nothing here may exit
|
|
483
|
+
log.warning("source.close() raised: %s", exc)
|
|
484
|
+
|
|
485
|
+
def _current_drop_limit(self) -> int:
|
|
486
|
+
"""~5 s of frames, at whatever rate we are actually running.
|
|
487
|
+
|
|
488
|
+
No fixed 30-frame floor: at 3 fps that was 10 s, twice the ~5 s budget. A
|
|
489
|
+
small floor still guards a dropout from tripping the fault before the
|
|
490
|
+
rate is even measured.
|
|
491
|
+
"""
|
|
492
|
+
if not self._fps.fps:
|
|
493
|
+
return 60
|
|
494
|
+
return max(2, round(5.0 * self._fps.fps))
|
|
495
|
+
|
|
496
|
+
def _fault_out(self, code: str, msg: str) -> None:
|
|
497
|
+
self._fault = {"e": "fault", "code": code, "msg": msg}
|
|
498
|
+
self._on_event(dict(self._fault))
|
|
499
|
+
|
|
500
|
+
def _backoff(self, attempts: int) -> None:
|
|
501
|
+
# attempts can be 0 right after a reset (see run()): treat that as "the
|
|
502
|
+
# first attempt" rather than wrapping to reopen_backoff[-1] — the whole
|
|
503
|
+
# point of the reset is that a fresh outage should not inherit the
|
|
504
|
+
# previous outage's backoff ceiling.
|
|
505
|
+
index = min(max(attempts, 1) - 1, len(self._reopen_backoff) - 1)
|
|
506
|
+
delay = self._reopen_backoff[index]
|
|
507
|
+
if delay:
|
|
508
|
+
self._stop.wait(delay)
|
|
509
|
+
|
|
510
|
+
def _on_frame(self, image) -> None:
|
|
511
|
+
frame, raw, results = self._extractor.process(image)
|
|
512
|
+
self._fps.tick()
|
|
513
|
+
self._frames_seen += 1
|
|
514
|
+
|
|
515
|
+
# Before the warmup return and before the capture gate, both
|
|
516
|
+
# deliberately: the view's whole purpose is aiming the camera, which
|
|
517
|
+
# happens while warming up and while paused. The overlay's tracking
|
|
518
|
+
# status is therefore one frame behind — immaterial at ~2 fps, and the
|
|
519
|
+
# alternative is publishing from three separate places.
|
|
520
|
+
self._publish_view(frame, results)
|
|
521
|
+
|
|
522
|
+
if self._auto is None:
|
|
523
|
+
if self._frames_seen < self._warmup_frames:
|
|
524
|
+
return
|
|
525
|
+
self._arm_segmenter()
|
|
526
|
+
|
|
527
|
+
self._emit_tracking(check_tracking(raw, self.dominant))
|
|
528
|
+
self._disarm_if_absent()
|
|
529
|
+
|
|
530
|
+
with self._lock:
|
|
531
|
+
if not self._capture:
|
|
532
|
+
return
|
|
533
|
+
take = self._auto.update(raw)
|
|
534
|
+
state = self._auto.state
|
|
535
|
+
previous = self._take_state
|
|
536
|
+
n_frames = self._auto.n_frames
|
|
537
|
+
self._take_state = state
|
|
538
|
+
|
|
539
|
+
if take is not None:
|
|
540
|
+
self._classify(take)
|
|
541
|
+
self._emit_take_state(state, previous, n_frames)
|
|
542
|
+
|
|
543
|
+
def _publish_view(self, frame, results) -> None:
|
|
544
|
+
"""Hand an annotated frame to the debug view, if one is being watched.
|
|
545
|
+
|
|
546
|
+
Reuses the landmarks recognition already extracted — this runs no
|
|
547
|
+
inference. Failures are swallowed inside `annotate`: a debug view is
|
|
548
|
+
never worth taking the capture loop down.
|
|
549
|
+
"""
|
|
550
|
+
if self._view is None or not self._view.watching:
|
|
551
|
+
return
|
|
552
|
+
if self._cv2 is None:
|
|
553
|
+
import cv2
|
|
554
|
+
|
|
555
|
+
self._cv2 = cv2
|
|
556
|
+
tracker = self._tracker
|
|
557
|
+
jpeg = annotate(
|
|
558
|
+
self._cv2,
|
|
559
|
+
self._extractor,
|
|
560
|
+
frame,
|
|
561
|
+
results,
|
|
562
|
+
{
|
|
563
|
+
"tracking": tracker.status if tracker else None,
|
|
564
|
+
"take_state": self._take_state,
|
|
565
|
+
"capture": self._capture,
|
|
566
|
+
"fps": self._fps.fps,
|
|
567
|
+
},
|
|
568
|
+
)
|
|
569
|
+
if jpeg is not None:
|
|
570
|
+
self._view.publish(jpeg)
|
|
571
|
+
|
|
572
|
+
def _disarm_if_absent(self) -> None:
|
|
573
|
+
"""Nobody in frame means no take.
|
|
574
|
+
|
|
575
|
+
AutoTake reads an untracked hand as rest — correct for its purpose, since
|
|
576
|
+
MediaPipe drops hands constantly — but it means an empty room accumulates
|
|
577
|
+
rest frames and leaves the segmenter armed at nobody. Worse, the first
|
|
578
|
+
hand to appear after that would open a take with an empty pre-roll.
|
|
579
|
+
"""
|
|
580
|
+
if self._tracker.raw != "absent":
|
|
581
|
+
self._absent_disarmed = False
|
|
582
|
+
return
|
|
583
|
+
if not self._absent_disarmed and self._tracker.run >= self._tracker.hold_frames:
|
|
584
|
+
with self._lock:
|
|
585
|
+
self._auto.disarm()
|
|
586
|
+
self._take_state = None
|
|
587
|
+
self._absent_disarmed = True
|
|
588
|
+
|
|
589
|
+
def _arm_segmenter(self) -> None:
|
|
590
|
+
"""Size AutoTake against the rate this machine actually achieved."""
|
|
591
|
+
self._timing = derive_timing(self._fps.fps)
|
|
592
|
+
self._auto = AutoTake(
|
|
593
|
+
dominant=self.dominant,
|
|
594
|
+
pre_roll=self._timing.pre_roll,
|
|
595
|
+
rest_to_arm=self._timing.rest_to_arm,
|
|
596
|
+
raised_to_start=self._timing.raised_to_start,
|
|
597
|
+
rest_to_close=self._timing.rest_to_close,
|
|
598
|
+
max_frames=self._timing.max_frames,
|
|
599
|
+
)
|
|
600
|
+
# ~0.5 s of agreement before a framing change is believed.
|
|
601
|
+
self._tracker = TrackingDebouncer(hold_frames=max(2, self._timing.rest_to_arm))
|
|
602
|
+
|
|
603
|
+
def _emit_tracking(self, tracking: Tracking) -> None:
|
|
604
|
+
status = self._tracker.update(tracking)
|
|
605
|
+
if status is not None:
|
|
606
|
+
self._on_event(
|
|
607
|
+
{
|
|
608
|
+
"e": "tracking",
|
|
609
|
+
"status": status,
|
|
610
|
+
"hands": tracking.hands,
|
|
611
|
+
"clipped": tracking.clipped,
|
|
612
|
+
}
|
|
613
|
+
)
|
|
614
|
+
|
|
615
|
+
def _emit_take_state(self, state: str, previous: str | None, n_frames: int) -> None:
|
|
616
|
+
"""`recording` fires every frame; `armed` only on the transition into it."""
|
|
617
|
+
if state == "recording":
|
|
618
|
+
self._on_event({"e": "recording", "frames": n_frames})
|
|
619
|
+
elif state == "armed" and previous != "armed":
|
|
620
|
+
self._on_event({"e": "armed"})
|
|
621
|
+
|
|
622
|
+
def _classify(self, take: list) -> None:
|
|
623
|
+
self._on_event({"e": "classifying"})
|
|
624
|
+
prediction: Prediction | None = None
|
|
625
|
+
try:
|
|
626
|
+
prediction = self._recogniser.classify(take)
|
|
627
|
+
except ClipTooShort:
|
|
628
|
+
self._on_event(
|
|
629
|
+
{
|
|
630
|
+
"e": "unclear",
|
|
631
|
+
"conf": 0.0,
|
|
632
|
+
"top3": [],
|
|
633
|
+
"reason": "too_short",
|
|
634
|
+
"n_frames": len(take),
|
|
635
|
+
}
|
|
636
|
+
)
|
|
637
|
+
except Exception as exc: # noqa: BLE001 - one bad take must not end the service
|
|
638
|
+
self._fault_out("model_error", f"{type(exc).__name__}: {exc}")
|
|
639
|
+
else:
|
|
640
|
+
# A watchdog polls /health: one bad take must not mark the service
|
|
641
|
+
# down forever once a later take proves the model still works.
|
|
642
|
+
self._fault = None
|
|
643
|
+
|
|
644
|
+
quality = take_quality(take, self.dominant)
|
|
645
|
+
if prediction is not None:
|
|
646
|
+
self._on_event(self._prediction_event(prediction, quality))
|
|
647
|
+
self._save(take, prediction, quality)
|
|
648
|
+
|
|
649
|
+
def _prediction_event(self, prediction: Prediction, quality: TakeQuality) -> dict:
|
|
650
|
+
top3 = [[gloss, float(p)] for gloss, p in prediction.top3]
|
|
651
|
+
if prediction.gloss is None:
|
|
652
|
+
return {"e": "unclear", "conf": float(prediction.confidence), "top3": top3}
|
|
653
|
+
return {
|
|
654
|
+
"e": "recognised",
|
|
655
|
+
"gloss": prediction.gloss,
|
|
656
|
+
"conf": float(prediction.confidence),
|
|
657
|
+
"top3": top3,
|
|
658
|
+
"n_frames": prediction.n_frames,
|
|
659
|
+
"usable_frames": prediction.usable_frames,
|
|
660
|
+
"encode_ms": round(prediction.encode_ms, 1),
|
|
661
|
+
"forward_ms": round(prediction.forward_ms, 1),
|
|
662
|
+
# The difference between "the model was wrong" and "the camera never
|
|
663
|
+
# saw you". Invisible in a gloss, and field recordings proved you need it.
|
|
664
|
+
"take_usable": quality.usable,
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
def _measure_disk_free_mb(self) -> int | None:
|
|
668
|
+
"""None if there is no store to measure, or the measurement itself fails.
|
|
669
|
+
|
|
670
|
+
Shared by `_save` (which gates on it) and `health()` (which can then
|
|
671
|
+
warn about a full disk before any gesture has ever been recorded) so
|
|
672
|
+
the two never compute it two different ways.
|
|
673
|
+
"""
|
|
674
|
+
if self._store is None:
|
|
675
|
+
return None
|
|
676
|
+
try:
|
|
677
|
+
root = self._store.unlabelled_root
|
|
678
|
+
while not root.exists() and root != root.parent:
|
|
679
|
+
root = root.parent
|
|
680
|
+
usage = shutil.disk_usage(root)
|
|
681
|
+
return usage.free // (1024 * 1024)
|
|
682
|
+
except OSError:
|
|
683
|
+
# A disk_usage() failure is not a camera problem — it must not
|
|
684
|
+
# surface as camera_lost and force a pointless reopen.
|
|
685
|
+
return None
|
|
686
|
+
|
|
687
|
+
def _save(self, take: list, prediction: Prediction | None, quality: TakeQuality) -> None:
|
|
688
|
+
if self._store is None:
|
|
689
|
+
return
|
|
690
|
+
disk_free_mb = self._measure_disk_free_mb()
|
|
691
|
+
if disk_free_mb is not None and disk_free_mb < self._min_free_mb:
|
|
692
|
+
return # keep recognising; stop writing. /health says why.
|
|
693
|
+
try:
|
|
694
|
+
self._store.save(
|
|
695
|
+
take,
|
|
696
|
+
None, # nobody is at a keyboard to label this
|
|
697
|
+
prediction,
|
|
698
|
+
self._recogniser.threshold,
|
|
699
|
+
meta={
|
|
700
|
+
"trigger": "auto",
|
|
701
|
+
"source": "service",
|
|
702
|
+
"take_usable": quality.usable,
|
|
703
|
+
"take_hand_fraction": round(quality.hand_fraction, 3),
|
|
704
|
+
"classifier": self._classifier_name,
|
|
705
|
+
"classifier_sha256": self._classifier_sha256,
|
|
706
|
+
# build_dataset re-encodes from raw landmarks, but a take is
|
|
707
|
+
# only interpretable later against the encoder that
|
|
708
|
+
# classified it.
|
|
709
|
+
"encoder": getattr(self._recogniser, "meta", {}).get("encoder"),
|
|
710
|
+
"fps": round(self._fps.fps, 2),
|
|
711
|
+
**(self._source.describe() if self._source else {}),
|
|
712
|
+
},
|
|
713
|
+
)
|
|
714
|
+
except OSError as exc:
|
|
715
|
+
# A failed recording is not a recognition failure and must not take
|
|
716
|
+
# the service down with it.
|
|
717
|
+
log.warning("failed to save take: %s", exc)
|