@camstack/addon-pipeline 1.2.177 → 1.2.179

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/audio-analyzer/index.js +2 -2
  2. package/dist/audio-analyzer/index.mjs +2 -2
  3. package/dist/detection-pipeline/index.js +130 -22
  4. package/dist/detection-pipeline/index.mjs +129 -21
  5. package/dist/{dist-DxA44rNd.js → dist-Bbu27T_N.js} +42 -0
  6. package/dist/{dist-D7aPrnxq.mjs → dist-pMHigtQu.mjs} +42 -0
  7. package/dist/{lazy-sharp-B9gAzlWk.js → lazy-sharp-V_rih5vD.js} +1 -1
  8. package/dist/{local-frame-registry-CZGAexUl.mjs → local-frame-registry-C7uCSrQG.mjs} +1 -1
  9. package/dist/{local-frame-registry-8Hzf6pPx.js → local-frame-registry-DK-Z1Ev1.js} +1 -1
  10. package/dist/motion-wasm/index.js +2 -2
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node--gufH5K3.mjs → node-CmPPt-uJ.mjs} +1 -1
  13. package/dist/{node-D8a2tmPR.js → node-DMTe6WuM.js} +1 -1
  14. package/dist/pipeline-runner/index.js +5 -5
  15. package/dist/pipeline-runner/index.mjs +4 -4
  16. package/dist/{process-memory-D2ik9mux.mjs → process-memory-BSR1vUsp.mjs} +1 -1
  17. package/dist/{process-memory-C6gjUbUI.js → process-memory-DFA42X6A.js} +1 -1
  18. package/dist/recorder/index.js +2 -2
  19. package/dist/recorder/index.mjs +2 -2
  20. package/dist/{segment-demux-js-CUrDbmvm.mjs → segment-demux-js-D-yE6_2O.mjs} +1 -1
  21. package/dist/{segment-demux-js-CRPLwKUP.js → segment-demux-js-jj19HZoq.js} +1 -1
  22. package/dist/session-decode/decode-worker-child.js +2 -2
  23. package/dist/session-decode/decode-worker-child.mjs +1 -1
  24. package/dist/stream-broker/_stub.js +1 -1
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-DVfYOF91.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-CPhBUByO.mjs} +3 -3
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-BpHWJmKV.mjs +26 -0
  27. package/dist/stream-broker/demux-worker-child.js +1 -1
  28. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  29. package/dist/stream-broker/{hostInit-BaMXeegw.mjs → hostInit-9unPdu_p.mjs} +3 -3
  30. package/dist/stream-broker/index.js +3 -3
  31. package/dist/stream-broker/index.mjs +3 -3
  32. package/dist/stream-broker/remoteEntry.js +1 -1
  33. package/dist/{worker-protocol-BZQMaDfm.mjs → worker-protocol-BqiZM7J3.mjs} +1 -1
  34. package/dist/{worker-protocol-DX7eACGC.js → worker-protocol-D_44Incp.js} +1 -1
  35. package/package.json +1 -1
  36. package/python/__pycache__/inference_pool.cpython-314.pyc +0 -0
  37. package/python/__pycache__/test_inference_pool_deadline.cpython-314.pyc +0 -0
  38. package/python/inference_pool.py +225 -28
  39. package/python/test_inference_pool_deadline.py +253 -0
  40. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-B_HZNoVq.mjs +0 -26
@@ -36,6 +36,21 @@ Runtime protocol:
36
36
  0x02 — infer_raw payload = [1B model_idx][4B width][4B height]
37
37
  [1B fmt 0=RGB,1=BGR,2=GRAY][pixels]
38
38
 
39
+ Wire-protocol versioning (D350): the startup config may declare
40
+ `protocolVersion`; the effective wire version is `min(PROTOCOL_VERSION,
41
+ declared)` and the ready reply reports this worker's own PROTOCOL_VERSION so
42
+ the Node side can negotiate the same minimum. At v2, every SHEDDABLE request
43
+ (infer_jpeg / infer_raw / infer_batch / infer_cached) prefixes its payload
44
+ with the request's ABSOLUTE deadline — [8B epoch-ms uint64 LE] — and this
45
+ worker checks it immediately before predict, answering
46
+ `{"dropped": true, "shedReason": "deadline-expired"}` instead of executing
47
+ work whose caller has already abandoned it (the 2026-09-04 storm: 4h15m of
48
+ the accelerator at 100% on replies discarded as `unknown request id`).
49
+ Commands and model loads NEVER carry a deadline at any version — a shed
50
+ command desynchronizes model state. A host that declares nothing is an OLD
51
+ host and gets byte-identical v1 behaviour; skew in either direction is the
52
+ NORMAL state during a rolling deploy.
53
+
39
54
  Commands: load, unload, replace, reconfigure, status, uncache_frame, mem_stats.
40
55
  """
41
56
  from __future__ import annotations
@@ -87,6 +102,93 @@ RAW_FMT_RGB = 0x00
87
102
  RAW_FMT_BGR = 0x01
88
103
  RAW_FMT_GRAY = 0x02
89
104
 
105
+ # ---------------------------------------------------------------------------
106
+ # Wire-protocol versioning + the deadline on the wire (D350)
107
+ # ---------------------------------------------------------------------------
108
+
109
+ # Highest wire version THIS worker implements. Reported in the ready reply;
110
+ # the effective version per connection is min() of both sides' declarations,
111
+ # so a worker and a host at different versions (normal during a rolling
112
+ # deploy) always agree on the layout.
113
+ PROTOCOL_VERSION = 2
114
+
115
+ # Opcodes that may carry (and be shed by) a deadline. Mirrors
116
+ # SHEDDABLE_MSG_TYPES in shared-inference-pool.ts — commands and cacheFrame
117
+ # are absent on BOTH sides: a shed command desynchronizes model state, a shed
118
+ # cacheFrame strands the inferCached that follows it.
119
+ SHEDDABLE_MSG_TYPES = frozenset({MSG_INFER_JPEG, MSG_INFER_RAW, MSG_INFER_BATCH, MSG_INFER_CACHED})
120
+
121
+ # Bytes of the absolute-deadline prefix a v2 sheddable payload carries.
122
+ DEADLINE_PREFIX_LEN = 8
123
+
124
+ # Sampling stride for the deadline-expired shed stderr line. Under a storm
125
+ # this fires at frame rate; one line per shed buried every other signal the
126
+ # last time a shed path logged unsampled (45 741 rows/hour, 2026-08-01). The
127
+ # counter (PoolCounters.shed_deadline_expired) keeps the exact total.
128
+ DEADLINE_SHED_LOG_EVERY = 500
129
+
130
+
131
+ def _negotiate_wire_version(config: dict) -> int:
132
+ """Effective wire version for this connection: min(ours, host declared).
133
+
134
+ No declaration (an OLD host) or a garbage value negotiates v1 — the
135
+ byte-identical historical layout. Pure; unit-tested.
136
+ """
137
+ try:
138
+ declared = int(config.get("protocolVersion", 1) or 1)
139
+ except (TypeError, ValueError):
140
+ declared = 1
141
+ return max(1, min(PROTOCOL_VERSION, declared))
142
+
143
+
144
+ def _split_deadline(msg_type: int, payload: bytes, wire_version: int) -> "tuple[int, bytes]":
145
+ """Split a request payload into (deadline_ms, rest) by the NEGOTIATED
146
+ version. v1, and every non-sheddable opcode at any version, passes the
147
+ payload through untouched with deadline 0 (= no deadline, never sheds).
148
+ Raises ValueError on a v2 sheddable payload too short to carry its
149
+ prefix. Pure; unit-tested.
150
+ """
151
+ if wire_version < 2 or msg_type not in SHEDDABLE_MSG_TYPES:
152
+ return 0, payload
153
+ if len(payload) < DEADLINE_PREFIX_LEN:
154
+ raise ValueError(
155
+ f"truncated deadline prefix on msg_type={msg_type} "
156
+ f"({len(payload)} bytes, need {DEADLINE_PREFIX_LEN})"
157
+ )
158
+ (deadline_ms,) = struct.unpack("<Q", payload[:DEADLINE_PREFIX_LEN])
159
+ return deadline_ms, payload[DEADLINE_PREFIX_LEN:]
160
+
161
+
162
+ def _deadline_expired(deadline_ms: int, now_ms: float) -> bool:
163
+ """0 means "no deadline" and never expires (v1 host / unstamped)."""
164
+ return deadline_ms > 0 and now_ms > deadline_ms
165
+
166
+
167
+ def _dropped_deadline_payload() -> dict:
168
+ """The shed reply for an expired request — same shape the TS side already
169
+ recognises as flow control (`poolResultToEngineOutput`)."""
170
+ return {
171
+ "kind": "detections",
172
+ "detections": [],
173
+ "inferenceMs": 0,
174
+ "dropped": True,
175
+ "shedReason": "deadline-expired",
176
+ }
177
+
178
+
179
+ def _log_deadline_shed(total: int, model_idx: int, late_ms: float) -> None:
180
+ """Sampled stderr line for a deadline shed — first occurrence and every
181
+ DEADLINE_SHED_LOG_EVERY-th, carrying the running total so nothing is
182
+ lost, only repeated. A branch that drops work silently is how the
183
+ 2026-09-04 collapse produced not one line in four hours."""
184
+ if total != 1 and total % DEADLINE_SHED_LOG_EVERY != 0:
185
+ return
186
+ sys.stderr.write(
187
+ f"deadline-expired shed: model {model_idx}, {late_ms:.0f}ms past deadline "
188
+ f"(total {total}, sampled every {DEADLINE_SHED_LOG_EVERY})\n"
189
+ )
190
+ sys.stderr.flush()
191
+
90
192
 
91
193
  # ---------------------------------------------------------------------------
92
194
  # Preprocessing (unchanged from v1 — same shapes/math, moved into helpers)
@@ -1891,6 +1993,10 @@ class PoolCounters:
1891
1993
  infer_cached: int = 0
1892
1994
  commands: int = 0
1893
1995
  frames_cached: int = 0
1996
+ # Requests answered `dropped` because their wire deadline had already
1997
+ # passed when they would have run — dead work NOT executed (D350). For a
1998
+ # batch, every item counts.
1999
+ shed_deadline_expired: int = 0
1894
2000
 
1895
2001
  def as_dict(self) -> dict:
1896
2002
  return {
@@ -1900,6 +2006,7 @@ class PoolCounters:
1900
2006
  "inferCached": self.infer_cached,
1901
2007
  "commands": self.commands,
1902
2008
  "framesCached": self.frames_cached,
2009
+ "shedDeadlineExpired": self.shed_deadline_expired,
1903
2010
  "totalInference": (
1904
2011
  self.infer_jpeg + self.infer_raw + self.infer_batch + self.infer_cached
1905
2012
  ),
@@ -2008,6 +2115,58 @@ def _mem_stats_payload(
2008
2115
  }
2009
2116
 
2010
2117
 
2118
+ # ---------------------------------------------------------------------------
2119
+ # Single-frame inference execution — module-level so the deadline gate is
2120
+ # unit-testable (test_inference_pool_deadline.py) without driving the whole
2121
+ # event loop. `handle_inference` in _run() delegates here; the caller's
2122
+ # `finally` still owns the backpressure slot release.
2123
+ # ---------------------------------------------------------------------------
2124
+
2125
+
2126
+ async def _execute_inference(
2127
+ req_id: int,
2128
+ img: Any,
2129
+ model_idx: int,
2130
+ deadline_ms: int,
2131
+ *,
2132
+ models: "list[ModelSlot]",
2133
+ dispatcher: Any,
2134
+ writer: Any,
2135
+ batch_mode: str,
2136
+ get_accumulator: "Optional[Callable[[int], Any]]",
2137
+ counters: "PoolCounters",
2138
+ ) -> None:
2139
+ """Deadline gate + predict for ONE admitted single-frame request.
2140
+
2141
+ The deadline is checked HERE — immediately before predict — rather than
2142
+ only at arrival, because this coroutine is entered both on admission and
2143
+ when the backpressure queue promotes a waiting frame: the wait is exactly
2144
+ where a request goes stale. The gate comes first (before the model-loaded
2145
+ check) on purpose — the cheapest question first, and an abandoned request
2146
+ deserves a shed, not a diagnosis. `dispatcher` and `writer` are
2147
+ duck-typed (RuntimeDispatcher / ResponseWriter in production).
2148
+ """
2149
+ now_ms = time.time() * 1000.0
2150
+ if _deadline_expired(deadline_ms, now_ms):
2151
+ counters.shed_deadline_expired += 1
2152
+ _log_deadline_shed(counters.shed_deadline_expired, model_idx, now_ms - deadline_ms)
2153
+ await writer.send(req_id, _dropped_deadline_payload())
2154
+ return
2155
+ if model_idx >= len(models) or not models[model_idx].loaded:
2156
+ await writer.send(req_id, {
2157
+ "error": f"Model {model_idx} not loaded",
2158
+ "kind": "detections",
2159
+ "detections": [],
2160
+ "inferenceMs": 0,
2161
+ })
2162
+ return
2163
+ if batch_mode == "window" and get_accumulator is not None:
2164
+ await get_accumulator(model_idx).submit(req_id, img)
2165
+ return
2166
+ result = await dispatcher.run(models[model_idx], img)
2167
+ await writer.send(req_id, result)
2168
+
2169
+
2011
2170
  # ---------------------------------------------------------------------------
2012
2171
  # IPC — binary framing with request_id multiplexing
2013
2172
  # ---------------------------------------------------------------------------
@@ -2192,6 +2351,10 @@ async def _run() -> None:
2192
2351
  config = json.loads(payload)
2193
2352
  runtime = config.get("runtime", "coreml")
2194
2353
  pool_device = str(config.get("device", "") or "")
2354
+ # Wire version for THIS connection (D350): an old host declares nothing
2355
+ # and negotiates v1 — byte-identical historical behaviour, no deadline
2356
+ # prefix expected on any payload.
2357
+ wire_version = _negotiate_wire_version(config)
2195
2358
  concurrency = int(config.get("concurrency", 1) or 1)
2196
2359
  batch_mode = str(config.get("batch_mode", "none"))
2197
2360
  if batch_mode not in ("none", "list", "window"):
@@ -2201,7 +2364,8 @@ async def _run() -> None:
2201
2364
 
2202
2365
  sys.stderr.write(
2203
2366
  f"Initializing runtime: {runtime} (concurrency={concurrency}, "
2204
- f"batch_mode={batch_mode}, window_ms={window_ms}, max_batch={max_batch_size})\n"
2367
+ f"batch_mode={batch_mode}, window_ms={window_ms}, max_batch={max_batch_size}, "
2368
+ f"wire_version={wire_version})\n"
2205
2369
  )
2206
2370
  sys.stderr.flush()
2207
2371
  _init_runtime(runtime, pool_device)
@@ -2266,6 +2430,10 @@ async def _run() -> None:
2266
2430
  "startupMs": startup_ms,
2267
2431
  "runtime": runtime,
2268
2432
  "workers": dispatcher.workers,
2433
+ # Our own capability, NOT the negotiated minimum: the Node side runs
2434
+ # the same min() against what it declared, so both ends derive the
2435
+ # identical effective version (D350).
2436
+ "protocolVersion": PROTOCOL_VERSION,
2269
2437
  })
2270
2438
 
2271
2439
  # ── Window accumulator (per-model) ──────────────────────────────
@@ -2358,21 +2526,23 @@ async def _run() -> None:
2358
2526
  return acc
2359
2527
 
2360
2528
  # ── Main loop ───────────────────────────────────────────────────
2361
- async def handle_inference(req_id: int, img: Image.Image, model_idx: int) -> None:
2529
+ async def handle_inference(
2530
+ req_id: int, img: Image.Image, model_idx: int, deadline_ms: int = 0,
2531
+ ) -> None:
2362
2532
  try:
2363
- if model_idx >= len(models) or not models[model_idx].loaded:
2364
- await writer.send(req_id, {
2365
- "error": f"Model {model_idx} not loaded",
2366
- "kind": "detections",
2367
- "detections": [],
2368
- "inferenceMs": 0,
2369
- })
2370
- return
2371
- if batch_mode == "window":
2372
- await get_accumulator(model_idx).submit(req_id, img)
2373
- return
2374
- result = await dispatcher.run(models[model_idx], img)
2375
- await writer.send(req_id, result)
2533
+ # The deadline gate lives in _execute_inference, immediately
2534
+ # before predict — it runs here on admission AND again when a
2535
+ # queued frame is promoted below, which is where a request goes
2536
+ # stale (D350).
2537
+ await _execute_inference(
2538
+ req_id, img, model_idx, deadline_ms,
2539
+ models=models,
2540
+ dispatcher=dispatcher,
2541
+ writer=writer,
2542
+ batch_mode=batch_mode,
2543
+ get_accumulator=get_accumulator,
2544
+ counters=counters,
2545
+ )
2376
2546
  except Exception as exc:
2377
2547
  sys.stderr.write(f"Inference error (model {model_idx}, req {req_id}): {exc}\n")
2378
2548
  sys.stderr.flush()
@@ -2385,8 +2555,10 @@ async def _run() -> None:
2385
2555
  finally:
2386
2556
  # Release this model's running slot; promote the next queued
2387
2557
  # frame (if any) into the freed slot immediately.
2388
- for next_req_id, next_img in backpressure.complete(model_idx):
2389
- asyncio.create_task(handle_inference(next_req_id, next_img, model_idx))
2558
+ for next_req_id, next_img, next_deadline_ms in backpressure.complete(model_idx):
2559
+ asyncio.create_task(
2560
+ handle_inference(next_req_id, next_img, model_idx, next_deadline_ms)
2561
+ )
2390
2562
 
2391
2563
  def _send_dropped(req_id: int) -> None:
2392
2564
  # Fast shed response — returned in microseconds instead of queuing the
@@ -2399,19 +2571,36 @@ async def _run() -> None:
2399
2571
  "dropped": True,
2400
2572
  }))
2401
2573
 
2402
- def _dispatch_inference(req_id: int, img: Image.Image, model_idx: int) -> None:
2574
+ def _dispatch_inference(
2575
+ req_id: int, img: Image.Image, model_idx: int, deadline_ms: int = 0,
2576
+ ) -> None:
2403
2577
  # Gate every single-frame inference through the per-model in-flight
2404
2578
  # bound. Slot accounting is synchronous (admit here, complete in the
2405
2579
  # handle_inference `finally`), so a slot can never leak past a task's
2406
2580
  # lifetime. Under overload the OLDEST queued frame is shed and answered
2407
2581
  # immediately so the caller never hangs on a frame we chose not to run.
2408
- to_run, to_drop = backpressure.admit(model_idx, (req_id, img))
2409
- for dropped_req_id, _dropped_img in to_drop:
2582
+ # The wire deadline rides the queued tuple so the gate in
2583
+ # _execute_inference sees it AFTER the wait, not just at arrival.
2584
+ to_run, to_drop = backpressure.admit(model_idx, (req_id, img, deadline_ms))
2585
+ for dropped_req_id, _dropped_img, _dropped_deadline in to_drop:
2410
2586
  _send_dropped(dropped_req_id)
2411
- for run_req_id, run_img in to_run:
2412
- asyncio.create_task(handle_inference(run_req_id, run_img, model_idx))
2413
-
2414
- async def handle_batch(req_id: int, model_idx: int, items: list[Image.Image]) -> None:
2587
+ for run_req_id, run_img, run_deadline_ms in to_run:
2588
+ asyncio.create_task(handle_inference(run_req_id, run_img, model_idx, run_deadline_ms))
2589
+
2590
+ async def handle_batch(
2591
+ req_id: int, model_idx: int, items: list[Image.Image], deadline_ms: int = 0,
2592
+ ) -> None:
2593
+ # Deadline gate first — the whole batch is one abandoned caller, so
2594
+ # every item is answered as a well-formed dropped record and none of
2595
+ # the work runs (D350).
2596
+ now_ms = time.time() * 1000.0
2597
+ if _deadline_expired(deadline_ms, now_ms):
2598
+ counters.shed_deadline_expired += len(items)
2599
+ _log_deadline_shed(counters.shed_deadline_expired, model_idx, now_ms - deadline_ms)
2600
+ await writer.send(req_id, {
2601
+ "results": [_dropped_deadline_payload() for _ in items],
2602
+ })
2603
+ return
2415
2604
  if model_idx >= len(models) or not models[model_idx].loaded:
2416
2605
  await writer.send(req_id, {
2417
2606
  "error": f"Model {model_idx} not loaded",
@@ -2459,6 +2648,14 @@ async def _run() -> None:
2459
2648
  except asyncio.IncompleteReadError:
2460
2649
  return
2461
2650
 
2651
+ # v2 sheddable payloads carry [8B absolute deadline] first; v1 (and
2652
+ # every command) passes through untouched with deadline 0 (D350).
2653
+ try:
2654
+ deadline_ms, payload = _split_deadline(msg_type, payload, wire_version)
2655
+ except ValueError as exc:
2656
+ await writer.send(req_id, {"error": str(exc)})
2657
+ continue
2658
+
2462
2659
  if msg_type == MSG_COMMAND:
2463
2660
  counters.commands += 1
2464
2661
  try:
@@ -2505,7 +2702,7 @@ async def _run() -> None:
2505
2702
  except Exception as exc:
2506
2703
  await writer.send(req_id, {"error": f"jpeg decode failed: {exc}"})
2507
2704
  continue
2508
- _dispatch_inference(req_id, img, model_idx)
2705
+ _dispatch_inference(req_id, img, model_idx, deadline_ms)
2509
2706
 
2510
2707
  elif msg_type == MSG_INFER_RAW:
2511
2708
  counters.infer_raw += 1
@@ -2521,7 +2718,7 @@ async def _run() -> None:
2521
2718
  except Exception as exc:
2522
2719
  await writer.send(req_id, {"error": f"raw wrap failed: {exc}"})
2523
2720
  continue
2524
- _dispatch_inference(req_id, img, model_idx)
2721
+ _dispatch_inference(req_id, img, model_idx, deadline_ms)
2525
2722
 
2526
2723
  elif msg_type == MSG_INFER_BATCH:
2527
2724
  # Header: [1B model_idx][1B count][4B frame_id]
@@ -2567,7 +2764,7 @@ async def _run() -> None:
2567
2764
  if parse_err is not None:
2568
2765
  await writer.send(req_id, {"error": parse_err, "results": []})
2569
2766
  continue
2570
- asyncio.create_task(handle_batch(req_id, model_idx, items))
2767
+ asyncio.create_task(handle_batch(req_id, model_idx, items, deadline_ms))
2571
2768
 
2572
2769
  elif msg_type == MSG_CACHE_FRAME:
2573
2770
  counters.frames_cached += 1
@@ -2605,7 +2802,7 @@ async def _run() -> None:
2605
2802
  if model_idx >= len(models) or not models[model_idx].loaded:
2606
2803
  await writer.send(req_id, {"error": f"model {model_idx} not loaded"})
2607
2804
  continue
2608
- _dispatch_inference(req_id, img, model_idx)
2805
+ _dispatch_inference(req_id, img, model_idx, deadline_ms)
2609
2806
 
2610
2807
  else:
2611
2808
  await writer.send(req_id, {"error": f"unknown msg_type: {msg_type}"})
@@ -0,0 +1,253 @@
1
+ """Unit tests for the deadline-on-the-wire shed (D350).
2
+
3
+ On a pool timeout the TS runner drops the frame and returns null — but the
4
+ Python worker still executed the abandoned request and its late reply was
5
+ discarded as `Response for unknown request id`. Measured 2026-09-04: a 4h15m
6
+ storm, 16 146 timeouts, the accelerator at 100% on results nobody read.
7
+
8
+ Each SHEDDABLE request (infer_jpeg / infer_raw / infer_batch / infer_cached)
9
+ now carries its ABSOLUTE deadline (epoch ms, uint64 LE) on the wire when both
10
+ sides negotiated protocol v2; the worker checks it immediately before predict
11
+ and answers `{dropped: true, shedReason: "deadline-expired"}` instead of
12
+ running dead work. Commands and model loads NEVER carry a deadline — a shed
13
+ command desynchronizes model state.
14
+
15
+ Pure-python: stubs numpy / PIL / postprocessors in sys.modules so the module
16
+ imports without the ML runtime installed (inference_pool uses
17
+ `from __future__ import annotations`, so annotations never touch the stubs).
18
+
19
+ Run: python3 -m unittest test_inference_pool_deadline -v
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import struct
24
+ import sys
25
+ import time
26
+ import types
27
+ import unittest
28
+
29
+ # ---------------------------------------------------------------------------
30
+ # Stub the heavy third-party imports BEFORE importing inference_pool.
31
+ # ---------------------------------------------------------------------------
32
+
33
+ try: # real numpy when the sandbox has it — a blind stub POISONS the whole
34
+ import numpy # noqa: F401 # pytest session for every sibling that needs it
35
+ except ImportError:
36
+ _np = types.ModuleType("numpy")
37
+ _np.float32 = float
38
+ _np.array = lambda values, dtype=None: values
39
+ sys.modules["numpy"] = _np
40
+
41
+ try: # real PIL when present, for the same reason
42
+ import PIL.Image # noqa: F401
43
+ except ImportError:
44
+ _pil = types.ModuleType("PIL")
45
+ _pil_image = types.ModuleType("PIL.Image")
46
+ _pil.Image = _pil_image
47
+ sys.modules["PIL"] = _pil
48
+ sys.modules["PIL.Image"] = _pil_image
49
+
50
+ if "postprocessors" not in sys.modules:
51
+ _pp = types.ModuleType("postprocessors")
52
+ _pp.POSTPROCESSORS = {}
53
+ sys.modules["postprocessors"] = _pp
54
+
55
+ from inference_pool import ( # noqa: E402
56
+ MSG_COMMAND,
57
+ MSG_INFER_BATCH,
58
+ MSG_INFER_CACHED,
59
+ MSG_INFER_JPEG,
60
+ MSG_INFER_RAW,
61
+ PROTOCOL_VERSION,
62
+ SHEDDABLE_MSG_TYPES,
63
+ ModelSlot,
64
+ PoolCounters,
65
+ _deadline_expired,
66
+ _execute_inference,
67
+ _negotiate_wire_version,
68
+ _split_deadline,
69
+ )
70
+
71
+
72
+ # ---------------------------------------------------------------------------
73
+ # Fakes — duck-typed stand-ins for the loop's writer / dispatcher.
74
+ # ---------------------------------------------------------------------------
75
+
76
+
77
+ class _FakeWriter:
78
+ """Collects (req_id, payload) pairs the loop would write to stdout."""
79
+
80
+ def __init__(self) -> None:
81
+ self.sent: list[tuple[int, dict]] = []
82
+
83
+ async def send(self, req_id: int, payload: dict) -> None:
84
+ self.sent.append((req_id, payload))
85
+
86
+
87
+ class _FakeDispatcher:
88
+ """Counts predict dispatches — the thing an expired request must NOT do."""
89
+
90
+ def __init__(self) -> None:
91
+ self.run_calls = 0
92
+
93
+ async def run(self, slot: object, img: object) -> dict:
94
+ self.run_calls += 1
95
+ return {"kind": "detections", "detections": [], "inferenceMs": 1.0}
96
+
97
+
98
+ def _loaded_models() -> list:
99
+ slot = ModelSlot()
100
+ slot.loaded = True
101
+ return [slot]
102
+
103
+
104
+ def _now_ms() -> int:
105
+ return int(time.time() * 1000)
106
+
107
+
108
+ # ---------------------------------------------------------------------------
109
+ # The gate: expired ⇒ dropped, NO predict. Fresh ⇒ predict runs.
110
+ # ---------------------------------------------------------------------------
111
+
112
+
113
+ class DeadlineGateTest(unittest.IsolatedAsyncioTestCase):
114
+ async def test_expired_request_is_answered_dropped_and_no_predict_ran(self) -> None:
115
+ writer = _FakeWriter()
116
+ dispatcher = _FakeDispatcher()
117
+ counters = PoolCounters()
118
+ await _execute_inference(
119
+ 7,
120
+ object(),
121
+ 0,
122
+ _now_ms() - 5_000, # expired 5s ago — the storm's steady state
123
+ models=_loaded_models(),
124
+ dispatcher=dispatcher,
125
+ writer=writer,
126
+ batch_mode="none",
127
+ get_accumulator=None,
128
+ counters=counters,
129
+ )
130
+ # THE fix: dead work is not executed.
131
+ self.assertEqual(dispatcher.run_calls, 0)
132
+ self.assertEqual(len(writer.sent), 1)
133
+ req_id, payload = writer.sent[0]
134
+ self.assertEqual(req_id, 7)
135
+ self.assertIs(payload.get("dropped"), True)
136
+ self.assertEqual(payload.get("shedReason"), "deadline-expired")
137
+ self.assertEqual(payload.get("kind"), "detections")
138
+ self.assertEqual(payload.get("detections"), [])
139
+ # The shed is COUNTABLE — a silent drop is what let the original
140
+ # collapse run for four hours.
141
+ self.assertEqual(counters.shed_deadline_expired, 1)
142
+ self.assertEqual(counters.as_dict().get("shedDeadlineExpired"), 1)
143
+
144
+ async def test_future_deadline_runs_normally(self) -> None:
145
+ writer = _FakeWriter()
146
+ dispatcher = _FakeDispatcher()
147
+ counters = PoolCounters()
148
+ await _execute_inference(
149
+ 8,
150
+ object(),
151
+ 0,
152
+ _now_ms() + 60_000,
153
+ models=_loaded_models(),
154
+ dispatcher=dispatcher,
155
+ writer=writer,
156
+ batch_mode="none",
157
+ get_accumulator=None,
158
+ counters=counters,
159
+ )
160
+ self.assertEqual(dispatcher.run_calls, 1)
161
+ req_id, payload = writer.sent[0]
162
+ self.assertEqual(req_id, 8)
163
+ self.assertNotIn("dropped", payload)
164
+ self.assertEqual(counters.shed_deadline_expired, 0)
165
+
166
+ async def test_no_deadline_runs_normally(self) -> None:
167
+ """deadline 0 = old host / no deadline — never sheds (skew safety)."""
168
+ writer = _FakeWriter()
169
+ dispatcher = _FakeDispatcher()
170
+ counters = PoolCounters()
171
+ await _execute_inference(
172
+ 9,
173
+ object(),
174
+ 0,
175
+ 0,
176
+ models=_loaded_models(),
177
+ dispatcher=dispatcher,
178
+ writer=writer,
179
+ batch_mode="none",
180
+ get_accumulator=None,
181
+ counters=counters,
182
+ )
183
+ self.assertEqual(dispatcher.run_calls, 1)
184
+ self.assertEqual(counters.shed_deadline_expired, 0)
185
+
186
+
187
+ # ---------------------------------------------------------------------------
188
+ # Wire parsing — both layouts, chosen by the negotiated version.
189
+ # ---------------------------------------------------------------------------
190
+
191
+
192
+ class SplitDeadlineTest(unittest.TestCase):
193
+ def test_v1_payload_passes_through_untouched(self) -> None:
194
+ """An OLD host never stamps a deadline — the payload is the old layout
195
+ and must not lose its first 8 bytes to a prefix that is not there."""
196
+ payload = bytes([3]) + b"jpegdata"
197
+ self.assertEqual(_split_deadline(MSG_INFER_JPEG, payload, 1), (0, payload))
198
+
199
+ def test_v2_sheddable_payload_strips_the_deadline_prefix(self) -> None:
200
+ deadline = 1_757_000_000_123
201
+ rest = bytes([3]) + b"jpegdata"
202
+ payload = struct.pack("<Q", deadline) + rest
203
+ self.assertEqual(_split_deadline(MSG_INFER_JPEG, payload, 2), (deadline, rest))
204
+
205
+ def test_v2_applies_to_every_sheddable_opcode(self) -> None:
206
+ deadline = 42
207
+ rest = b"\x01payload"
208
+ for msg_type in (MSG_INFER_JPEG, MSG_INFER_RAW, MSG_INFER_BATCH, MSG_INFER_CACHED):
209
+ payload = struct.pack("<Q", deadline) + rest
210
+ self.assertEqual(_split_deadline(msg_type, payload, 2), (deadline, rest))
211
+ self.assertIn(msg_type, SHEDDABLE_MSG_TYPES)
212
+
213
+ def test_commands_never_carry_a_deadline_even_at_v2(self) -> None:
214
+ """A shed command desynchronizes model state — commands keep the old
215
+ layout at every version."""
216
+ payload = b'{"cmd": "status"}'
217
+ self.assertEqual(_split_deadline(MSG_COMMAND, payload, 2), (0, payload))
218
+ self.assertNotIn(MSG_COMMAND, SHEDDABLE_MSG_TYPES)
219
+
220
+ def test_truncated_deadline_prefix_raises(self) -> None:
221
+ with self.assertRaises(ValueError):
222
+ _split_deadline(MSG_INFER_JPEG, b"\x00\x01\x02", 2)
223
+
224
+
225
+ class DeadlineExpiredTest(unittest.TestCase):
226
+ def test_zero_deadline_never_expires(self) -> None:
227
+ self.assertFalse(_deadline_expired(0, 1e15))
228
+
229
+ def test_future_deadline_is_not_expired(self) -> None:
230
+ self.assertFalse(_deadline_expired(2_000, 1_999.0))
231
+
232
+ def test_past_deadline_is_expired(self) -> None:
233
+ self.assertTrue(_deadline_expired(2_000, 2_001.0))
234
+
235
+
236
+ class NegotiateWireVersionTest(unittest.TestCase):
237
+ def test_old_host_without_declaration_negotiates_v1(self) -> None:
238
+ """A host at a different version is the NORMAL state during a deploy —
239
+ no declaration means the old layout, always."""
240
+ self.assertEqual(_negotiate_wire_version({}), 1)
241
+
242
+ def test_matching_host_negotiates_v2(self) -> None:
243
+ self.assertEqual(_negotiate_wire_version({"protocolVersion": 2}), 2)
244
+
245
+ def test_newer_host_is_capped_at_our_version(self) -> None:
246
+ self.assertEqual(_negotiate_wire_version({"protocolVersion": 99}), PROTOCOL_VERSION)
247
+
248
+ def test_garbage_declaration_falls_back_to_v1(self) -> None:
249
+ self.assertEqual(_negotiate_wire_version({"protocolVersion": "banana"}), 1)
250
+
251
+
252
+ if __name__ == "__main__":
253
+ unittest.main()