@camstack/addon-pipeline 1.2.294 → 1.2.296

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/THIRD_PARTY_MODELS.md +241 -0
  2. package/dist/audio-analyzer/index.js +2 -2
  3. package/dist/audio-analyzer/index.mjs +2 -2
  4. package/dist/{default-detection-model-0dPKRKUD.mjs → default-detection-model-Co578D8C.mjs} +181 -99
  5. package/dist/{default-detection-model-D24AJOTn.js → default-detection-model-D1daTtqT.js} +181 -99
  6. package/dist/detection-pipeline/index.js +1301 -531
  7. package/dist/detection-pipeline/index.mjs +1301 -531
  8. package/dist/{dist-CJR259Xf.js → dist-8up-f2TX.js} +3688 -2698
  9. package/dist/{dist-RXbmRAwP.mjs → dist-CCd0Q3nr.mjs} +3676 -2698
  10. package/dist/motion-wasm/index.js +1 -1
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node-atmRSHPk.mjs → node-DgMSXSWP.mjs} +1 -1
  13. package/dist/{node-DWg9zbY1.js → node-lpQgHes9.js} +1 -1
  14. package/dist/pipeline-runner/index.js +975 -270
  15. package/dist/pipeline-runner/index.mjs +975 -270
  16. package/dist/{process-memory-BJUXvTjd.js → process-memory-CX_92V_r.js} +1 -1
  17. package/dist/{process-memory-BgFOHFnx.mjs → process-memory-DFC_O5zE.mjs} +1 -1
  18. package/dist/recorder/index.js +14 -6
  19. package/dist/recorder/index.mjs +14 -6
  20. package/dist/{segment-demux-js-C_fPJub3.js → segment-demux-js-DzBx6NN2.js} +1 -1
  21. package/dist/{segment-demux-js-G7wFpHzn.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
  22. package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
  23. package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
  24. package/dist/stream-broker/_stub.js +2 -2
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-6IyM-BIn.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
  27. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
  28. package/dist/stream-broker/demux-worker-child.js +1 -1
  29. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  30. package/dist/stream-broker/{hostInit-BBYHWS3M.mjs → hostInit-BPtppL3W.mjs} +2 -2
  31. package/dist/stream-broker/index.js +4 -4
  32. package/dist/stream-broker/index.mjs +4 -4
  33. package/dist/stream-broker/remoteEntry.js +1 -1
  34. package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
  35. package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
  36. package/package.json +3 -2
  37. package/python/inference_pool.py +422 -64
  38. package/python/postprocessors/__init__.py +2 -0
  39. package/python/postprocessors/ssd.py +73 -17
  40. package/python/postprocessors/test_ssd.py +205 -0
  41. package/python/postprocessors/test_yunet.py +292 -0
  42. package/python/postprocessors/testdata/ssdlite_mobiledet_outputs.json +1 -0
  43. package/python/postprocessors/yunet.py +275 -0
  44. package/python/test_inference_pool_compile_off_loop.py +414 -0
  45. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-6IHzlLJ_.mjs +0 -26
  46. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DoyA71_q.mjs +0 -26
@@ -52,6 +52,9 @@ host and gets byte-identical v1 behaviour; skew in either direction is the
52
52
  NORMAL state during a rolling deploy.
53
53
 
54
54
  Commands: load, unload, replace, reconfigure, status, uncache_frame, mem_stats.
55
+ `load`/`replace` compile on their own thread under a hard bound and never block
56
+ the loop; a load past the bound answers `reason: "compile-timeout"` and poisons
57
+ the worker (D653, see `ModelLoader`).
55
58
  """
56
59
  from __future__ import annotations
57
60
 
@@ -68,6 +71,7 @@ import re
68
71
  import shutil
69
72
  import struct
70
73
  import sys
74
+ import threading
71
75
  import time
72
76
  from collections import deque
73
77
  from dataclasses import dataclass, field
@@ -2358,6 +2362,7 @@ def _mem_stats_payload(
2358
2362
  preprocess_cache_len: int,
2359
2363
  bench_cache_len: int,
2360
2364
  uptime_sec: int,
2365
+ model_load: Optional[dict] = None,
2361
2366
  ) -> dict:
2362
2367
  """Assemble the mem_stats reply. Pure given its inputs except the two
2363
2368
  process-wide reads (memory + arenas), so the shape is unit-testable."""
@@ -2374,6 +2379,9 @@ def _mem_stats_payload(
2374
2379
  "preprocessCacheEntries": preprocess_cache_len,
2375
2380
  "benchPreprocessCacheEntries": bench_cache_len,
2376
2381
  "uptimeSec": uptime_sec,
2382
+ # What is compiling, and whether a compile ever failed to return
2383
+ # (D653) — the one fact a host needs to tell "slow" from "poisoned".
2384
+ "modelLoad": model_load,
2377
2385
  }
2378
2386
 
2379
2387
 
@@ -2486,25 +2494,13 @@ RECONFIGURABLE_CONFIG_KEYS = frozenset({"nmsIouThreshold"})
2486
2494
 
2487
2495
 
2488
2496
  def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
2489
- action = cmd.get("cmd")
2490
- if action == "load":
2491
- index = cmd["index"]
2492
- config = cmd["config"]
2493
- while len(models) <= index:
2494
- models.append(ModelSlot())
2495
- slot = models[index]
2496
- if slot.loaded:
2497
- _unload_model(slot)
2498
- try:
2499
- t0 = time.perf_counter()
2500
- _load_model(slot, config)
2501
- load_ms = round((time.perf_counter() - t0) * 1000)
2502
- sys.stderr.write(f"Model {index} loaded: {config['path']} ({load_ms}ms)\n")
2503
- sys.stderr.flush()
2504
- return {"cmd": "load", "index": index, "status": "ok", "loadMs": load_ms}
2505
- except Exception as exc:
2506
- return {"cmd": "load", "index": index, "status": "error", "error": str(exc)}
2497
+ """The model commands that need no compile: unload, reconfigure, status.
2507
2498
 
2499
+ `load` and `replace` are NOT here — they compile, and a compile must never
2500
+ run on the event loop (D653). They go through `_handle_model_command`,
2501
+ which the `CommandRouter` awaits off the loop's reader.
2502
+ """
2503
+ action = cmd.get("cmd")
2508
2504
  if action == "unload":
2509
2505
  index = cmd["index"]
2510
2506
  if index < len(models) and models[index].loaded:
@@ -2550,24 +2546,6 @@ def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
2550
2546
  sys.stderr.flush()
2551
2547
  return {"cmd": "reconfigure", "index": index, "status": "ok"}
2552
2548
 
2553
- if action == "replace":
2554
- index = cmd["index"]
2555
- config = cmd["config"]
2556
- while len(models) <= index:
2557
- models.append(ModelSlot())
2558
- slot = models[index]
2559
- if slot.loaded:
2560
- _unload_model(slot)
2561
- try:
2562
- t0 = time.perf_counter()
2563
- _load_model(slot, config)
2564
- load_ms = round((time.perf_counter() - t0) * 1000)
2565
- sys.stderr.write(f"Model {index} replaced: {config['path']} ({load_ms}ms)\n")
2566
- sys.stderr.flush()
2567
- return {"cmd": "replace", "index": index, "status": "ok", "loadMs": load_ms}
2568
- except Exception as exc:
2569
- return {"cmd": "replace", "index": index, "status": "error", "error": str(exc)}
2570
-
2571
2549
  if action == "status":
2572
2550
  status = []
2573
2551
  for i, slot in enumerate(models):
@@ -2582,6 +2560,374 @@ def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
2582
2560
  return {"cmd": action or "unknown", "status": "error", "error": f"Unknown command: {action}"}
2583
2561
 
2584
2562
 
2563
+ # ---------------------------------------------------------------------------
2564
+ # Model loads run OFF the event loop, one at a time, under a hard bound (D653)
2565
+ # ---------------------------------------------------------------------------
2566
+ #
2567
+ # 2026-09-26, hub `openvino:gpu`: one `core.compile_model()` of
2568
+ # `camstack-yunet-2023mar.xml` for the iGPU never returned. It ran INLINE on this
2569
+ # loop, so from 18:33:48 until a container restart at 22:41 the worker read no
2570
+ # request and wrote no reply: 786 client-side timeouts on six cameras, 123 of
2571
+ # 123 `mem_stats` probes lost, and a process that looked perfectly alive.
2572
+ #
2573
+ # So a load now runs on its own DAEMON thread and the loop only awaits it:
2574
+ # inference and `mem_stats` keep being served while a compile runs. And it is
2575
+ # BOUNDED: past `MODEL_LOAD_TIMEOUT_SEC` the load answers `compile-timeout`,
2576
+ # naming the model, instead of never answering.
2577
+ #
2578
+ # A native thread stuck in a GPU compiler cannot be cancelled or killed from
2579
+ # Python. So a timeout POISONS the worker FOR NEW LOADS: every later load is
2580
+ # refused at once (`worker-poisoned`) rather than queued behind a thread that
2581
+ # may never free the compiler, and the models already loaded keep serving.
2582
+ #
2583
+ # The abandoned compile is NOT abandoned work (fix round 1): it keeps running,
2584
+ # so a slow but finite compile still writes its cache. When it comes back the
2585
+ # worker says so on the event channel (`compile-finished-late`) and is usable
2586
+ # for loads again — the next load of that model is a warm cache hit. If it
2587
+ # never comes back, the host kills the process at its HARD bound
2588
+ # (shared-inference-pool.ts `POOL_MODEL_LOAD_BOUNDS`). Daemon, not an executor
2589
+ # thread: `concurrent.futures` joins its threads at interpreter exit, so a stuck
2590
+ # compile would also have made the worker impossible to shut down.
2591
+
2592
+ # SOFT bound on ONE model load — compile included — in seconds. A cold GPU
2593
+ # compile of the whole model set on the N100 iGPU measured ~25 s; one model is
2594
+ # normally 1-3 s. The host declares its own value in the startup config
2595
+ # (`modelLoadTimeoutMs`) so its reply deadline and this bound cannot drift;
2596
+ # this is what an old host that declares nothing gets.
2597
+ MODEL_LOAD_TIMEOUT_SEC = 120.0
2598
+
2599
+ # Request id reserved for UNSOLICITED worker events (a reply to no request).
2600
+ # The host never allocates it (shared-inference-pool.ts `WORKER_EVENT_REQ_ID`).
2601
+ WORKER_EVENT_REQ_ID = 0xFFFFFFFF
2602
+
2603
+ # Commands answered by the loop immediately, never queued behind a load: the
2604
+ # probe the host uses to tell "slow" from "wedged" must never wait for the very
2605
+ # compile it is asking about.
2606
+ INLINE_COMMANDS = frozenset({"mem_stats", "uncache_frame"})
2607
+
2608
+ # Commands that compile a model — the only ones that leave the loop.
2609
+ LOAD_COMMANDS = frozenset({"load", "replace"})
2610
+
2611
+
2612
+ def _model_load_timeout_sec(config: dict) -> float:
2613
+ """The load bound for this connection: the host's, else the default."""
2614
+ raw = config.get("modelLoadTimeoutMs")
2615
+ if isinstance(raw, (int, float)) and not isinstance(raw, bool) and raw > 0:
2616
+ return float(raw) / 1000.0
2617
+ return MODEL_LOAD_TIMEOUT_SEC
2618
+
2619
+
2620
+ def _model_id_of(config: dict) -> str:
2621
+ """The model's file stem — what a log line and the host's reply carry."""
2622
+ path = str(config.get("path") or "")
2623
+ stem = os.path.splitext(os.path.basename(path))[0]
2624
+ return stem or "unknown"
2625
+
2626
+
2627
+ class ModelLoadTimeout(Exception):
2628
+ """A load did not finish inside its bound; new loads are refused until it does."""
2629
+
2630
+
2631
+ class WorkerPoisoned(Exception):
2632
+ """A previous load is still running past its bound; no new load starts."""
2633
+
2634
+
2635
+ class ModelLoader:
2636
+ """Runs `_load_model` on a daemon thread under a hard bound.
2637
+
2638
+ One load at a time is the CALLER's guarantee (the `CommandRouter` consumes
2639
+ model commands in order); this class only owns the thread, the bound and
2640
+ the poisoning. `load_fn` is injectable for tests; production passes
2641
+ nothing and gets the module's `_load_model`, looked up at call time.
2642
+ `notify` sends an unsolicited event to the host (the `compile-finished-late`
2643
+ line); absent, the event is only written to stderr.
2644
+ """
2645
+
2646
+ def __init__(
2647
+ self,
2648
+ timeout_sec: float = MODEL_LOAD_TIMEOUT_SEC,
2649
+ load_fn: "Optional[Callable[[ModelSlot, dict], None]]" = None,
2650
+ notify: "Optional[Callable[[dict], Awaitable[None]]]" = None,
2651
+ ) -> None:
2652
+ self.timeout_sec = timeout_sec
2653
+ self._load_fn = load_fn
2654
+ self._notify = notify
2655
+ self.finished_late = 0
2656
+ # Test seam: runs on the load thread right after it has claimed the
2657
+ # outcome — lets a test hold the thread exactly in the race window.
2658
+ self._after_claim: Optional[Callable[[], None]] = None
2659
+ # Monotonic start of the compile running past its bound, if any — the
2660
+ # host counts its HARD bound from here, not from a later refusal.
2661
+ self._abandoned_started: Optional[float] = None
2662
+ # `ensure_future` tasks are only weakly referenced by the loop; keep
2663
+ # each notification alive until it is sent.
2664
+ self._pending_notifications: "set[asyncio.Future[None]]" = set()
2665
+ self.in_flight: Optional[dict] = None
2666
+ self.poisoned_by: Optional[dict] = None
2667
+ self.completed = 0
2668
+ self.failed = 0
2669
+ self.timed_out = 0
2670
+
2671
+ def snapshot(self) -> dict:
2672
+ """What the `mem_stats` reply carries about loads."""
2673
+ in_flight = None
2674
+ if self.in_flight is not None:
2675
+ in_flight = {
2676
+ "index": self.in_flight["index"],
2677
+ "modelId": self.in_flight["modelId"],
2678
+ "elapsedMs": round((time.monotonic() - self.in_flight["startedAt"]) * 1000),
2679
+ }
2680
+ return {
2681
+ "inFlight": in_flight,
2682
+ "poisonedBy": dict(self.poisoned_by) if self.poisoned_by is not None else None,
2683
+ "boundMs": round(self.timeout_sec * 1000),
2684
+ "completed": self.completed,
2685
+ "failed": self.failed,
2686
+ "timedOut": self.timed_out,
2687
+ "finishedLate": self.finished_late,
2688
+ }
2689
+
2690
+ def _finish_abandoned(
2691
+ self, index: int, model_id: str, outcome: Optional[BaseException], elapsed_ms: int,
2692
+ ) -> None:
2693
+ """On the loop: an abandoned compile came back. The compiler is free
2694
+ again, so the worker loads again; a successful one has written its
2695
+ cache, so the host's next load of the model is warm. Its RESULT is
2696
+ still discarded — the host was told it failed and has moved on."""
2697
+ self.finished_late += 1
2698
+ self.poisoned_by = None
2699
+ self._abandoned_started = None
2700
+ payload = {
2701
+ "event": "compile-finished-late",
2702
+ "index": index,
2703
+ "modelId": model_id,
2704
+ "ok": outcome is None,
2705
+ "elapsedMs": elapsed_ms,
2706
+ }
2707
+ if outcome is not None:
2708
+ payload["error"] = str(outcome)
2709
+ if self._notify is not None:
2710
+ task = asyncio.ensure_future(self._notify(payload))
2711
+ self._pending_notifications.add(task)
2712
+ task.add_done_callback(self._pending_notifications.discard)
2713
+
2714
+ def abandoned_elapsed_ms(self) -> Optional[int]:
2715
+ """How long the compile running past its bound has been running."""
2716
+ if self._abandoned_started is None:
2717
+ return None
2718
+ return round((time.monotonic() - self._abandoned_started) * 1000)
2719
+
2720
+ async def load(self, slot: ModelSlot, config: dict, index: int) -> None:
2721
+ """Load `config` into the DETACHED `slot`, or raise.
2722
+
2723
+ Raises `WorkerPoisoned` without starting anything once a previous load
2724
+ timed out, `ModelLoadTimeout` when this one exceeds the bound, and the
2725
+ load's own exception when it fails in time.
2726
+ """
2727
+ model_id = _model_id_of(config)
2728
+ if self.poisoned_by is not None:
2729
+ raise WorkerPoisoned(
2730
+ f"worker poisoned for loads: the load of {self.poisoned_by['modelId']} is still "
2731
+ f"running past its bound ({self.poisoned_by['boundMs']}ms) — no new load starts "
2732
+ f"until it returns"
2733
+ )
2734
+ load_fn = self._load_fn if self._load_fn is not None else _load_model
2735
+ loop = asyncio.get_running_loop()
2736
+ done: "asyncio.Future[None]" = loop.create_future()
2737
+ # Exactly ONE of "the load returned in time" and "the bound fired"
2738
+ # wins, decided under this lock (release review). Without it the
2739
+ # thread could see "not abandoned", the bound fire before its settle
2740
+ # ran, and the worker stay poisoned forever for a load that finished.
2741
+ decision = threading.Lock()
2742
+ claimed: dict = {"by": None} # "thread" | "bound"
2743
+
2744
+ def settle(exc: Optional[BaseException]) -> None:
2745
+ if done.done():
2746
+ return
2747
+ if exc is None:
2748
+ done.set_result(None)
2749
+ else:
2750
+ done.set_exception(exc)
2751
+
2752
+ def target() -> None:
2753
+ outcome: Optional[BaseException] = None
2754
+ try:
2755
+ load_fn(slot, config)
2756
+ except BaseException as exc: # noqa: BLE001 — handed to the awaiting loop
2757
+ outcome = exc
2758
+ with decision:
2759
+ late = claimed["by"] == "bound"
2760
+ if not late:
2761
+ claimed["by"] = "thread"
2762
+ if self._after_claim is not None:
2763
+ self._after_claim()
2764
+ if late:
2765
+ # The bound already fired. Say that the native call DID come
2766
+ # back, and when — the only evidence of how long a slow
2767
+ # compile really takes — and hand the worker back its loads.
2768
+ elapsed_ms = round((time.monotonic() - started) * 1000)
2769
+ sys.stderr.write(
2770
+ f"Model {index} load of {model_id} returned AFTER its bound "
2771
+ f"in {elapsed_ms}ms ({'error: ' + str(outcome) if outcome else 'ok'}) "
2772
+ f"— result discarded, loads re-enabled\n"
2773
+ )
2774
+ sys.stderr.flush()
2775
+ try:
2776
+ loop.call_soon_threadsafe(
2777
+ self._finish_abandoned, index, model_id, outcome, elapsed_ms,
2778
+ )
2779
+ except RuntimeError:
2780
+ pass # loop closed: the process is shutting down
2781
+ return
2782
+ try:
2783
+ loop.call_soon_threadsafe(settle, outcome)
2784
+ except RuntimeError:
2785
+ pass # loop closed: the process is shutting down
2786
+
2787
+ started = time.monotonic()
2788
+ self.in_flight = {"index": index, "modelId": model_id, "startedAt": started}
2789
+ thread = threading.Thread(target=target, name=f"model-load-{index}", daemon=True)
2790
+ thread.start()
2791
+ try:
2792
+ try:
2793
+ await asyncio.wait_for(asyncio.shield(done), timeout=self.timeout_sec)
2794
+ except asyncio.TimeoutError:
2795
+ with decision:
2796
+ bound_won = claimed["by"] is None
2797
+ if bound_won:
2798
+ claimed["by"] = "bound"
2799
+ if bound_won:
2800
+ raise
2801
+ # The load returned AT the bound: its settle is already on
2802
+ # its way to this loop. It counts as in time.
2803
+ await done
2804
+ except asyncio.TimeoutError:
2805
+ self.timed_out += 1
2806
+ self._abandoned_started = started
2807
+ self.poisoned_by = {
2808
+ "index": index,
2809
+ "modelId": model_id,
2810
+ "boundMs": round(self.timeout_sec * 1000),
2811
+ "at": round(time.time() * 1000),
2812
+ }
2813
+ raise ModelLoadTimeout(
2814
+ f"load of {model_id} (index {index}) did not return within "
2815
+ f"{round(self.timeout_sec * 1000)}ms — it keeps running so its cache "
2816
+ f"can be written; no new load starts until it returns"
2817
+ ) from None
2818
+ except Exception:
2819
+ self.failed += 1
2820
+ raise
2821
+ finally:
2822
+ self.in_flight = None
2823
+ self.completed += 1
2824
+
2825
+
2826
+ async def _handle_model_command(
2827
+ models: "list[ModelSlot]", cmd: dict, loader: ModelLoader,
2828
+ ) -> dict:
2829
+ """Every model-management command, in order; `load`/`replace` off the loop.
2830
+
2831
+ The new model is loaded into a DETACHED slot and installed only once its
2832
+ load completed inside the bound, so a compile that returns late — after the
2833
+ host was told `compile-timeout` — can never surface as a model the host
2834
+ believes failed. Whatever the index held before keeps serving until that
2835
+ swap: a `replace` that fails or times out leaves the old model in place,
2836
+ and the old slot object is only dropped, never mutated, so a predict
2837
+ already holding it finishes normally.
2838
+ """
2839
+ action = cmd.get("cmd")
2840
+ if action not in LOAD_COMMANDS:
2841
+ return _handle_command(models, cmd)
2842
+ index = cmd["index"]
2843
+ config = cmd["config"]
2844
+ model_id = _model_id_of(config)
2845
+ while len(models) <= index:
2846
+ models.append(ModelSlot())
2847
+ slot = ModelSlot()
2848
+ sys.stderr.write(
2849
+ f"Model {index} load started: {config['path']} "
2850
+ f"(bound {round(loader.timeout_sec * 1000)}ms)\n"
2851
+ )
2852
+ sys.stderr.flush()
2853
+ t0 = time.perf_counter()
2854
+ try:
2855
+ await loader.load(slot, config, index)
2856
+ except WorkerPoisoned as exc:
2857
+ return {
2858
+ "cmd": action, "index": index, "status": "error",
2859
+ "reason": "worker-poisoned", "modelId": model_id, "poisoned": True,
2860
+ # How long the EARLIER compile has run: the host's hard bound
2861
+ # counts from its start, not from this refusal.
2862
+ "compileElapsedMs": loader.abandoned_elapsed_ms(),
2863
+ "error": str(exc),
2864
+ }
2865
+ except ModelLoadTimeout as exc:
2866
+ sys.stderr.write(f"Model {index} load TIMED OUT: {exc}\n")
2867
+ sys.stderr.flush()
2868
+ return {
2869
+ "cmd": action, "index": index, "status": "error",
2870
+ "reason": "compile-timeout", "modelId": model_id, "poisoned": True,
2871
+ "timeoutMs": round(loader.timeout_sec * 1000),
2872
+ "compileElapsedMs": loader.abandoned_elapsed_ms(),
2873
+ "error": str(exc),
2874
+ }
2875
+ except Exception as exc:
2876
+ return {"cmd": action, "index": index, "status": "error", "error": str(exc)}
2877
+ models[index] = slot
2878
+ load_ms = round((time.perf_counter() - t0) * 1000)
2879
+ verb = "loaded" if action == "load" else "replaced"
2880
+ sys.stderr.write(f"Model {index} {verb}: {config['path']} ({load_ms}ms)\n")
2881
+ sys.stderr.flush()
2882
+ return {"cmd": action, "index": index, "status": "ok", "loadMs": load_ms}
2883
+
2884
+
2885
+ class CommandRouter:
2886
+ """Routes MSG_COMMAND frames without ever blocking the reader.
2887
+
2888
+ `INLINE_COMMANDS` are answered at once by `inline` (it needs the loop's own
2889
+ state: caches, backpressure, counters). Everything else is queued and
2890
+ consumed IN ORDER by one task — model slots are indices shared with the
2891
+ host, so a `status` or `unload` must never overtake the `load` sent before
2892
+ it.
2893
+ """
2894
+
2895
+ def __init__(
2896
+ self,
2897
+ models: "list[ModelSlot]",
2898
+ loader: ModelLoader,
2899
+ send: "Callable[[int, dict], Awaitable[None]]",
2900
+ inline: "Callable[[dict], dict]",
2901
+ ) -> None:
2902
+ self._models = models
2903
+ self._loader = loader
2904
+ self._send = send
2905
+ self._inline = inline
2906
+ self._queue: "asyncio.Queue[tuple[int, dict]]" = asyncio.Queue()
2907
+ self._consumer: "Optional[asyncio.Task[None]]" = None
2908
+
2909
+ async def submit(self, req_id: int, cmd: dict) -> None:
2910
+ if cmd.get("cmd") in INLINE_COMMANDS:
2911
+ try:
2912
+ response = self._inline(cmd)
2913
+ except Exception as exc:
2914
+ response = {"cmd": cmd.get("cmd") or "unknown", "status": "error", "error": str(exc)}
2915
+ await self._send(req_id, response)
2916
+ return
2917
+ self._queue.put_nowait((req_id, cmd))
2918
+ if self._consumer is None or self._consumer.done():
2919
+ self._consumer = asyncio.create_task(self._consume())
2920
+
2921
+ async def _consume(self) -> None:
2922
+ while not self._queue.empty():
2923
+ req_id, cmd = self._queue.get_nowait()
2924
+ try:
2925
+ response = await _handle_model_command(self._models, cmd, self._loader)
2926
+ except Exception as exc:
2927
+ response = {"cmd": cmd.get("cmd") or "unknown", "status": "error", "error": str(exc)}
2928
+ await self._send(req_id, response)
2929
+
2930
+
2585
2931
  # ---------------------------------------------------------------------------
2586
2932
  # Main async loop
2587
2933
  # ---------------------------------------------------------------------------
@@ -2631,6 +2977,13 @@ async def _run() -> None:
2631
2977
  )
2632
2978
  sys.stderr.flush()
2633
2979
  _init_runtime(runtime, pool_device)
2980
+ # Off-loop, bounded model loads (D653). Startup-config models below still
2981
+ # load inline: nothing is served before `ready`, and the host bounds that
2982
+ # handshake itself.
2983
+ model_loader = ModelLoader(
2984
+ timeout_sec=_model_load_timeout_sec(config),
2985
+ notify=lambda event: writer.send(WORKER_EVENT_REQ_ID, event),
2986
+ )
2634
2987
 
2635
2988
  models: list[ModelSlot] = []
2636
2989
  for i, mc in enumerate(config.get("models", [])):
@@ -2782,6 +3135,33 @@ async def _run() -> None:
2782
3135
  # Saves ~10ms/call (the PIL resize + numpy + tensor pack cost).
2783
3136
  _preprocess_cache: dict[tuple[int, int], tuple[dict, float, tuple[int, int], int, int]] = {}
2784
3137
 
3138
+ def inline_command(cmd: dict) -> dict:
3139
+ # Needs the loop's own state (caches, backpressure, counters), which
3140
+ # the model-command path never sees. Answered at once — this is the
3141
+ # host's liveness probe, and it must not wait for a compile.
3142
+ if cmd.get("cmd") == "mem_stats":
3143
+ return _mem_stats_payload(
3144
+ models,
3145
+ counters,
3146
+ backpressure,
3147
+ len(frame_cache),
3148
+ len(_preprocess_cache),
3149
+ len(_bench_preprocess_cache),
3150
+ round(time.monotonic() - pool_started_monotonic),
3151
+ model_load=model_loader.snapshot(),
3152
+ )
3153
+ fid = cmd.get("frameId", -1)
3154
+ frame_cache.pop(fid, None)
3155
+ # Purge preprocessed tensor caches for this frame
3156
+ for k in [k for k in _preprocess_cache if k[0] == fid]:
3157
+ del _preprocess_cache[k]
3158
+ # Also purge the bench preprocess cache
3159
+ for k in [k for k in _bench_preprocess_cache if k[0] == fid]:
3160
+ del _bench_preprocess_cache[k]
3161
+ return {"cmd": "uncache_frame", "status": "ok", "frameId": fid}
3162
+
3163
+ command_router = CommandRouter(models, model_loader, writer.send, inline_command)
3164
+
2785
3165
  def get_accumulator(model_idx: int) -> WindowAccumulator:
2786
3166
  acc = accumulators.get(model_idx)
2787
3167
  if acc is None:
@@ -2924,35 +3304,13 @@ async def _run() -> None:
2924
3304
  counters.commands += 1
2925
3305
  try:
2926
3306
  cmd = json.loads(payload)
2927
- # Handled inline (needs the loop's own state: caches,
2928
- # backpressure, counters) — _handle_command only sees models.
2929
- if cmd.get("cmd") == "mem_stats":
2930
- response = _mem_stats_payload(
2931
- models,
2932
- counters,
2933
- backpressure,
2934
- len(frame_cache),
2935
- len(_preprocess_cache),
2936
- len(_bench_preprocess_cache),
2937
- round(time.monotonic() - pool_started_monotonic),
2938
- )
2939
- elif cmd.get("cmd") == "uncache_frame":
2940
- fid = cmd.get("frameId", -1)
2941
- removed_img = frame_cache.pop(fid, None)
2942
- # Purge preprocessed tensor caches for this frame
2943
- to_del = [k for k in _preprocess_cache if k[0] == fid]
2944
- for k in to_del:
2945
- del _preprocess_cache[k]
2946
- # Also purge the bench preprocess cache
2947
- to_del2 = [k for k in _bench_preprocess_cache if k[0] == fid]
2948
- for k in to_del2:
2949
- del _bench_preprocess_cache[k]
2950
- response = {"cmd": "uncache_frame", "status": "ok", "frameId": fid}
2951
- else:
2952
- response = _handle_command(models, cmd)
2953
3307
  except Exception as exc:
2954
- response = {"cmd": "unknown", "status": "error", "error": str(exc)}
2955
- await writer.send(req_id, response)
3308
+ await writer.send(req_id, {"cmd": "unknown", "status": "error", "error": str(exc)})
3309
+ continue
3310
+ # NEVER awaits a model load: `load`/`replace` are queued and
3311
+ # compiled on their own thread under a bound (D653), so the next
3312
+ # frame is read while a GPU compile runs.
3313
+ await command_router.submit(req_id, cmd)
2956
3314
 
2957
3315
  elif msg_type == MSG_INFER_JPEG:
2958
3316
  counters.infer_jpeg += 1
@@ -16,6 +16,7 @@ from .yamnet import postprocess_yamnet
16
16
  from .ssd import postprocess_ssd
17
17
  from .rfdetr import postprocess_rfdetr
18
18
  from .yolonas import postprocess_yolonas
19
+ from .yunet import postprocess_yunet
19
20
 
20
21
  POSTPROCESSORS = {
21
22
  "yolo": postprocess_yolo,
@@ -31,4 +32,5 @@ POSTPROCESSORS = {
31
32
  "ssd": postprocess_ssd,
32
33
  "rfdetr": postprocess_rfdetr,
33
34
  "yolonas": postprocess_yolonas,
35
+ "yunet": postprocess_yunet,
34
36
  }