@camstack/addon-pipeline 1.2.294 → 1.2.296
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +241 -0
- package/dist/audio-analyzer/index.js +2 -2
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/{default-detection-model-0dPKRKUD.mjs → default-detection-model-Co578D8C.mjs} +181 -99
- package/dist/{default-detection-model-D24AJOTn.js → default-detection-model-D1daTtqT.js} +181 -99
- package/dist/detection-pipeline/index.js +1301 -531
- package/dist/detection-pipeline/index.mjs +1301 -531
- package/dist/{dist-CJR259Xf.js → dist-8up-f2TX.js} +3688 -2698
- package/dist/{dist-RXbmRAwP.mjs → dist-CCd0Q3nr.mjs} +3676 -2698
- package/dist/motion-wasm/index.js +1 -1
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/{node-atmRSHPk.mjs → node-DgMSXSWP.mjs} +1 -1
- package/dist/{node-DWg9zbY1.js → node-lpQgHes9.js} +1 -1
- package/dist/pipeline-runner/index.js +975 -270
- package/dist/pipeline-runner/index.mjs +975 -270
- package/dist/{process-memory-BJUXvTjd.js → process-memory-CX_92V_r.js} +1 -1
- package/dist/{process-memory-BgFOHFnx.mjs → process-memory-DFC_O5zE.mjs} +1 -1
- package/dist/recorder/index.js +14 -6
- package/dist/recorder/index.mjs +14 -6
- package/dist/{segment-demux-js-C_fPJub3.js → segment-demux-js-DzBx6NN2.js} +1 -1
- package/dist/{segment-demux-js-G7wFpHzn.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
- package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
- package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
- package/dist/stream-broker/_stub.js +2 -2
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-6IyM-BIn.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-BBYHWS3M.mjs → hostInit-BPtppL3W.mjs} +2 -2
- package/dist/stream-broker/index.js +4 -4
- package/dist/stream-broker/index.mjs +4 -4
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
- package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
- package/package.json +3 -2
- package/python/inference_pool.py +422 -64
- package/python/postprocessors/__init__.py +2 -0
- package/python/postprocessors/ssd.py +73 -17
- package/python/postprocessors/test_ssd.py +205 -0
- package/python/postprocessors/test_yunet.py +292 -0
- package/python/postprocessors/testdata/ssdlite_mobiledet_outputs.json +1 -0
- package/python/postprocessors/yunet.py +275 -0
- package/python/test_inference_pool_compile_off_loop.py +414 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-6IHzlLJ_.mjs +0 -26
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DoyA71_q.mjs +0 -26
package/python/inference_pool.py
CHANGED
|
@@ -52,6 +52,9 @@ host and gets byte-identical v1 behaviour; skew in either direction is the
|
|
|
52
52
|
NORMAL state during a rolling deploy.
|
|
53
53
|
|
|
54
54
|
Commands: load, unload, replace, reconfigure, status, uncache_frame, mem_stats.
|
|
55
|
+
`load`/`replace` compile on their own thread under a hard bound and never block
|
|
56
|
+
the loop; a load past the bound answers `reason: "compile-timeout"` and poisons
|
|
57
|
+
the worker (D653, see `ModelLoader`).
|
|
55
58
|
"""
|
|
56
59
|
from __future__ import annotations
|
|
57
60
|
|
|
@@ -68,6 +71,7 @@ import re
|
|
|
68
71
|
import shutil
|
|
69
72
|
import struct
|
|
70
73
|
import sys
|
|
74
|
+
import threading
|
|
71
75
|
import time
|
|
72
76
|
from collections import deque
|
|
73
77
|
from dataclasses import dataclass, field
|
|
@@ -2358,6 +2362,7 @@ def _mem_stats_payload(
|
|
|
2358
2362
|
preprocess_cache_len: int,
|
|
2359
2363
|
bench_cache_len: int,
|
|
2360
2364
|
uptime_sec: int,
|
|
2365
|
+
model_load: Optional[dict] = None,
|
|
2361
2366
|
) -> dict:
|
|
2362
2367
|
"""Assemble the mem_stats reply. Pure given its inputs except the two
|
|
2363
2368
|
process-wide reads (memory + arenas), so the shape is unit-testable."""
|
|
@@ -2374,6 +2379,9 @@ def _mem_stats_payload(
|
|
|
2374
2379
|
"preprocessCacheEntries": preprocess_cache_len,
|
|
2375
2380
|
"benchPreprocessCacheEntries": bench_cache_len,
|
|
2376
2381
|
"uptimeSec": uptime_sec,
|
|
2382
|
+
# What is compiling, and whether a compile ever failed to return
|
|
2383
|
+
# (D653) — the one fact a host needs to tell "slow" from "poisoned".
|
|
2384
|
+
"modelLoad": model_load,
|
|
2377
2385
|
}
|
|
2378
2386
|
|
|
2379
2387
|
|
|
@@ -2486,25 +2494,13 @@ RECONFIGURABLE_CONFIG_KEYS = frozenset({"nmsIouThreshold"})
|
|
|
2486
2494
|
|
|
2487
2495
|
|
|
2488
2496
|
def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
|
|
2489
|
-
|
|
2490
|
-
if action == "load":
|
|
2491
|
-
index = cmd["index"]
|
|
2492
|
-
config = cmd["config"]
|
|
2493
|
-
while len(models) <= index:
|
|
2494
|
-
models.append(ModelSlot())
|
|
2495
|
-
slot = models[index]
|
|
2496
|
-
if slot.loaded:
|
|
2497
|
-
_unload_model(slot)
|
|
2498
|
-
try:
|
|
2499
|
-
t0 = time.perf_counter()
|
|
2500
|
-
_load_model(slot, config)
|
|
2501
|
-
load_ms = round((time.perf_counter() - t0) * 1000)
|
|
2502
|
-
sys.stderr.write(f"Model {index} loaded: {config['path']} ({load_ms}ms)\n")
|
|
2503
|
-
sys.stderr.flush()
|
|
2504
|
-
return {"cmd": "load", "index": index, "status": "ok", "loadMs": load_ms}
|
|
2505
|
-
except Exception as exc:
|
|
2506
|
-
return {"cmd": "load", "index": index, "status": "error", "error": str(exc)}
|
|
2497
|
+
"""The model commands that need no compile: unload, reconfigure, status.
|
|
2507
2498
|
|
|
2499
|
+
`load` and `replace` are NOT here — they compile, and a compile must never
|
|
2500
|
+
run on the event loop (D653). They go through `_handle_model_command`,
|
|
2501
|
+
which the `CommandRouter` awaits off the loop's reader.
|
|
2502
|
+
"""
|
|
2503
|
+
action = cmd.get("cmd")
|
|
2508
2504
|
if action == "unload":
|
|
2509
2505
|
index = cmd["index"]
|
|
2510
2506
|
if index < len(models) and models[index].loaded:
|
|
@@ -2550,24 +2546,6 @@ def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
|
|
|
2550
2546
|
sys.stderr.flush()
|
|
2551
2547
|
return {"cmd": "reconfigure", "index": index, "status": "ok"}
|
|
2552
2548
|
|
|
2553
|
-
if action == "replace":
|
|
2554
|
-
index = cmd["index"]
|
|
2555
|
-
config = cmd["config"]
|
|
2556
|
-
while len(models) <= index:
|
|
2557
|
-
models.append(ModelSlot())
|
|
2558
|
-
slot = models[index]
|
|
2559
|
-
if slot.loaded:
|
|
2560
|
-
_unload_model(slot)
|
|
2561
|
-
try:
|
|
2562
|
-
t0 = time.perf_counter()
|
|
2563
|
-
_load_model(slot, config)
|
|
2564
|
-
load_ms = round((time.perf_counter() - t0) * 1000)
|
|
2565
|
-
sys.stderr.write(f"Model {index} replaced: {config['path']} ({load_ms}ms)\n")
|
|
2566
|
-
sys.stderr.flush()
|
|
2567
|
-
return {"cmd": "replace", "index": index, "status": "ok", "loadMs": load_ms}
|
|
2568
|
-
except Exception as exc:
|
|
2569
|
-
return {"cmd": "replace", "index": index, "status": "error", "error": str(exc)}
|
|
2570
|
-
|
|
2571
2549
|
if action == "status":
|
|
2572
2550
|
status = []
|
|
2573
2551
|
for i, slot in enumerate(models):
|
|
@@ -2582,6 +2560,374 @@ def _handle_command(models: list[ModelSlot], cmd: dict) -> dict:
|
|
|
2582
2560
|
return {"cmd": action or "unknown", "status": "error", "error": f"Unknown command: {action}"}
|
|
2583
2561
|
|
|
2584
2562
|
|
|
2563
|
+
# ---------------------------------------------------------------------------
|
|
2564
|
+
# Model loads run OFF the event loop, one at a time, under a hard bound (D653)
|
|
2565
|
+
# ---------------------------------------------------------------------------
|
|
2566
|
+
#
|
|
2567
|
+
# 2026-09-26, hub `openvino:gpu`: one `core.compile_model()` of
|
|
2568
|
+
# `camstack-yunet-2023mar.xml` for the iGPU never returned. It ran INLINE on this
|
|
2569
|
+
# loop, so from 18:33:48 until a container restart at 22:41 the worker read no
|
|
2570
|
+
# request and wrote no reply: 786 client-side timeouts on six cameras, 123 of
|
|
2571
|
+
# 123 `mem_stats` probes lost, and a process that looked perfectly alive.
|
|
2572
|
+
#
|
|
2573
|
+
# So a load now runs on its own DAEMON thread and the loop only awaits it:
|
|
2574
|
+
# inference and `mem_stats` keep being served while a compile runs. And it is
|
|
2575
|
+
# BOUNDED: past `MODEL_LOAD_TIMEOUT_SEC` the load answers `compile-timeout`,
|
|
2576
|
+
# naming the model, instead of never answering.
|
|
2577
|
+
#
|
|
2578
|
+
# A native thread stuck in a GPU compiler cannot be cancelled or killed from
|
|
2579
|
+
# Python. So a timeout POISONS the worker FOR NEW LOADS: every later load is
|
|
2580
|
+
# refused at once (`worker-poisoned`) rather than queued behind a thread that
|
|
2581
|
+
# may never free the compiler, and the models already loaded keep serving.
|
|
2582
|
+
#
|
|
2583
|
+
# The abandoned compile is NOT abandoned work (fix round 1): it keeps running,
|
|
2584
|
+
# so a slow but finite compile still writes its cache. When it comes back the
|
|
2585
|
+
# worker says so on the event channel (`compile-finished-late`) and is usable
|
|
2586
|
+
# for loads again — the next load of that model is a warm cache hit. If it
|
|
2587
|
+
# never comes back, the host kills the process at its HARD bound
|
|
2588
|
+
# (shared-inference-pool.ts `POOL_MODEL_LOAD_BOUNDS`). Daemon, not an executor
|
|
2589
|
+
# thread: `concurrent.futures` joins its threads at interpreter exit, so a stuck
|
|
2590
|
+
# compile would also have made the worker impossible to shut down.
|
|
2591
|
+
|
|
2592
|
+
# SOFT bound on ONE model load — compile included — in seconds. A cold GPU
|
|
2593
|
+
# compile of the whole model set on the N100 iGPU measured ~25 s; one model is
|
|
2594
|
+
# normally 1-3 s. The host declares its own value in the startup config
|
|
2595
|
+
# (`modelLoadTimeoutMs`) so its reply deadline and this bound cannot drift;
|
|
2596
|
+
# this is what an old host that declares nothing gets.
|
|
2597
|
+
MODEL_LOAD_TIMEOUT_SEC = 120.0
|
|
2598
|
+
|
|
2599
|
+
# Request id reserved for UNSOLICITED worker events (a reply to no request).
|
|
2600
|
+
# The host never allocates it (shared-inference-pool.ts `WORKER_EVENT_REQ_ID`).
|
|
2601
|
+
WORKER_EVENT_REQ_ID = 0xFFFFFFFF
|
|
2602
|
+
|
|
2603
|
+
# Commands answered by the loop immediately, never queued behind a load: the
|
|
2604
|
+
# probe the host uses to tell "slow" from "wedged" must never wait for the very
|
|
2605
|
+
# compile it is asking about.
|
|
2606
|
+
INLINE_COMMANDS = frozenset({"mem_stats", "uncache_frame"})
|
|
2607
|
+
|
|
2608
|
+
# Commands that compile a model — the only ones that leave the loop.
|
|
2609
|
+
LOAD_COMMANDS = frozenset({"load", "replace"})
|
|
2610
|
+
|
|
2611
|
+
|
|
2612
|
+
def _model_load_timeout_sec(config: dict) -> float:
|
|
2613
|
+
"""The load bound for this connection: the host's, else the default."""
|
|
2614
|
+
raw = config.get("modelLoadTimeoutMs")
|
|
2615
|
+
if isinstance(raw, (int, float)) and not isinstance(raw, bool) and raw > 0:
|
|
2616
|
+
return float(raw) / 1000.0
|
|
2617
|
+
return MODEL_LOAD_TIMEOUT_SEC
|
|
2618
|
+
|
|
2619
|
+
|
|
2620
|
+
def _model_id_of(config: dict) -> str:
|
|
2621
|
+
"""The model's file stem — what a log line and the host's reply carry."""
|
|
2622
|
+
path = str(config.get("path") or "")
|
|
2623
|
+
stem = os.path.splitext(os.path.basename(path))[0]
|
|
2624
|
+
return stem or "unknown"
|
|
2625
|
+
|
|
2626
|
+
|
|
2627
|
+
class ModelLoadTimeout(Exception):
|
|
2628
|
+
"""A load did not finish inside its bound; new loads are refused until it does."""
|
|
2629
|
+
|
|
2630
|
+
|
|
2631
|
+
class WorkerPoisoned(Exception):
|
|
2632
|
+
"""A previous load is still running past its bound; no new load starts."""
|
|
2633
|
+
|
|
2634
|
+
|
|
2635
|
+
class ModelLoader:
|
|
2636
|
+
"""Runs `_load_model` on a daemon thread under a hard bound.
|
|
2637
|
+
|
|
2638
|
+
One load at a time is the CALLER's guarantee (the `CommandRouter` consumes
|
|
2639
|
+
model commands in order); this class only owns the thread, the bound and
|
|
2640
|
+
the poisoning. `load_fn` is injectable for tests; production passes
|
|
2641
|
+
nothing and gets the module's `_load_model`, looked up at call time.
|
|
2642
|
+
`notify` sends an unsolicited event to the host (the `compile-finished-late`
|
|
2643
|
+
line); absent, the event is only written to stderr.
|
|
2644
|
+
"""
|
|
2645
|
+
|
|
2646
|
+
def __init__(
|
|
2647
|
+
self,
|
|
2648
|
+
timeout_sec: float = MODEL_LOAD_TIMEOUT_SEC,
|
|
2649
|
+
load_fn: "Optional[Callable[[ModelSlot, dict], None]]" = None,
|
|
2650
|
+
notify: "Optional[Callable[[dict], Awaitable[None]]]" = None,
|
|
2651
|
+
) -> None:
|
|
2652
|
+
self.timeout_sec = timeout_sec
|
|
2653
|
+
self._load_fn = load_fn
|
|
2654
|
+
self._notify = notify
|
|
2655
|
+
self.finished_late = 0
|
|
2656
|
+
# Test seam: runs on the load thread right after it has claimed the
|
|
2657
|
+
# outcome — lets a test hold the thread exactly in the race window.
|
|
2658
|
+
self._after_claim: Optional[Callable[[], None]] = None
|
|
2659
|
+
# Monotonic start of the compile running past its bound, if any — the
|
|
2660
|
+
# host counts its HARD bound from here, not from a later refusal.
|
|
2661
|
+
self._abandoned_started: Optional[float] = None
|
|
2662
|
+
# `ensure_future` tasks are only weakly referenced by the loop; keep
|
|
2663
|
+
# each notification alive until it is sent.
|
|
2664
|
+
self._pending_notifications: "set[asyncio.Future[None]]" = set()
|
|
2665
|
+
self.in_flight: Optional[dict] = None
|
|
2666
|
+
self.poisoned_by: Optional[dict] = None
|
|
2667
|
+
self.completed = 0
|
|
2668
|
+
self.failed = 0
|
|
2669
|
+
self.timed_out = 0
|
|
2670
|
+
|
|
2671
|
+
def snapshot(self) -> dict:
|
|
2672
|
+
"""What the `mem_stats` reply carries about loads."""
|
|
2673
|
+
in_flight = None
|
|
2674
|
+
if self.in_flight is not None:
|
|
2675
|
+
in_flight = {
|
|
2676
|
+
"index": self.in_flight["index"],
|
|
2677
|
+
"modelId": self.in_flight["modelId"],
|
|
2678
|
+
"elapsedMs": round((time.monotonic() - self.in_flight["startedAt"]) * 1000),
|
|
2679
|
+
}
|
|
2680
|
+
return {
|
|
2681
|
+
"inFlight": in_flight,
|
|
2682
|
+
"poisonedBy": dict(self.poisoned_by) if self.poisoned_by is not None else None,
|
|
2683
|
+
"boundMs": round(self.timeout_sec * 1000),
|
|
2684
|
+
"completed": self.completed,
|
|
2685
|
+
"failed": self.failed,
|
|
2686
|
+
"timedOut": self.timed_out,
|
|
2687
|
+
"finishedLate": self.finished_late,
|
|
2688
|
+
}
|
|
2689
|
+
|
|
2690
|
+
def _finish_abandoned(
|
|
2691
|
+
self, index: int, model_id: str, outcome: Optional[BaseException], elapsed_ms: int,
|
|
2692
|
+
) -> None:
|
|
2693
|
+
"""On the loop: an abandoned compile came back. The compiler is free
|
|
2694
|
+
again, so the worker loads again; a successful one has written its
|
|
2695
|
+
cache, so the host's next load of the model is warm. Its RESULT is
|
|
2696
|
+
still discarded — the host was told it failed and has moved on."""
|
|
2697
|
+
self.finished_late += 1
|
|
2698
|
+
self.poisoned_by = None
|
|
2699
|
+
self._abandoned_started = None
|
|
2700
|
+
payload = {
|
|
2701
|
+
"event": "compile-finished-late",
|
|
2702
|
+
"index": index,
|
|
2703
|
+
"modelId": model_id,
|
|
2704
|
+
"ok": outcome is None,
|
|
2705
|
+
"elapsedMs": elapsed_ms,
|
|
2706
|
+
}
|
|
2707
|
+
if outcome is not None:
|
|
2708
|
+
payload["error"] = str(outcome)
|
|
2709
|
+
if self._notify is not None:
|
|
2710
|
+
task = asyncio.ensure_future(self._notify(payload))
|
|
2711
|
+
self._pending_notifications.add(task)
|
|
2712
|
+
task.add_done_callback(self._pending_notifications.discard)
|
|
2713
|
+
|
|
2714
|
+
def abandoned_elapsed_ms(self) -> Optional[int]:
|
|
2715
|
+
"""How long the compile running past its bound has been running."""
|
|
2716
|
+
if self._abandoned_started is None:
|
|
2717
|
+
return None
|
|
2718
|
+
return round((time.monotonic() - self._abandoned_started) * 1000)
|
|
2719
|
+
|
|
2720
|
+
async def load(self, slot: ModelSlot, config: dict, index: int) -> None:
|
|
2721
|
+
"""Load `config` into the DETACHED `slot`, or raise.
|
|
2722
|
+
|
|
2723
|
+
Raises `WorkerPoisoned` without starting anything once a previous load
|
|
2724
|
+
timed out, `ModelLoadTimeout` when this one exceeds the bound, and the
|
|
2725
|
+
load's own exception when it fails in time.
|
|
2726
|
+
"""
|
|
2727
|
+
model_id = _model_id_of(config)
|
|
2728
|
+
if self.poisoned_by is not None:
|
|
2729
|
+
raise WorkerPoisoned(
|
|
2730
|
+
f"worker poisoned for loads: the load of {self.poisoned_by['modelId']} is still "
|
|
2731
|
+
f"running past its bound ({self.poisoned_by['boundMs']}ms) — no new load starts "
|
|
2732
|
+
f"until it returns"
|
|
2733
|
+
)
|
|
2734
|
+
load_fn = self._load_fn if self._load_fn is not None else _load_model
|
|
2735
|
+
loop = asyncio.get_running_loop()
|
|
2736
|
+
done: "asyncio.Future[None]" = loop.create_future()
|
|
2737
|
+
# Exactly ONE of "the load returned in time" and "the bound fired"
|
|
2738
|
+
# wins, decided under this lock (release review). Without it the
|
|
2739
|
+
# thread could see "not abandoned", the bound fire before its settle
|
|
2740
|
+
# ran, and the worker stay poisoned forever for a load that finished.
|
|
2741
|
+
decision = threading.Lock()
|
|
2742
|
+
claimed: dict = {"by": None} # "thread" | "bound"
|
|
2743
|
+
|
|
2744
|
+
def settle(exc: Optional[BaseException]) -> None:
|
|
2745
|
+
if done.done():
|
|
2746
|
+
return
|
|
2747
|
+
if exc is None:
|
|
2748
|
+
done.set_result(None)
|
|
2749
|
+
else:
|
|
2750
|
+
done.set_exception(exc)
|
|
2751
|
+
|
|
2752
|
+
def target() -> None:
|
|
2753
|
+
outcome: Optional[BaseException] = None
|
|
2754
|
+
try:
|
|
2755
|
+
load_fn(slot, config)
|
|
2756
|
+
except BaseException as exc: # noqa: BLE001 — handed to the awaiting loop
|
|
2757
|
+
outcome = exc
|
|
2758
|
+
with decision:
|
|
2759
|
+
late = claimed["by"] == "bound"
|
|
2760
|
+
if not late:
|
|
2761
|
+
claimed["by"] = "thread"
|
|
2762
|
+
if self._after_claim is not None:
|
|
2763
|
+
self._after_claim()
|
|
2764
|
+
if late:
|
|
2765
|
+
# The bound already fired. Say that the native call DID come
|
|
2766
|
+
# back, and when — the only evidence of how long a slow
|
|
2767
|
+
# compile really takes — and hand the worker back its loads.
|
|
2768
|
+
elapsed_ms = round((time.monotonic() - started) * 1000)
|
|
2769
|
+
sys.stderr.write(
|
|
2770
|
+
f"Model {index} load of {model_id} returned AFTER its bound "
|
|
2771
|
+
f"in {elapsed_ms}ms ({'error: ' + str(outcome) if outcome else 'ok'}) "
|
|
2772
|
+
f"— result discarded, loads re-enabled\n"
|
|
2773
|
+
)
|
|
2774
|
+
sys.stderr.flush()
|
|
2775
|
+
try:
|
|
2776
|
+
loop.call_soon_threadsafe(
|
|
2777
|
+
self._finish_abandoned, index, model_id, outcome, elapsed_ms,
|
|
2778
|
+
)
|
|
2779
|
+
except RuntimeError:
|
|
2780
|
+
pass # loop closed: the process is shutting down
|
|
2781
|
+
return
|
|
2782
|
+
try:
|
|
2783
|
+
loop.call_soon_threadsafe(settle, outcome)
|
|
2784
|
+
except RuntimeError:
|
|
2785
|
+
pass # loop closed: the process is shutting down
|
|
2786
|
+
|
|
2787
|
+
started = time.monotonic()
|
|
2788
|
+
self.in_flight = {"index": index, "modelId": model_id, "startedAt": started}
|
|
2789
|
+
thread = threading.Thread(target=target, name=f"model-load-{index}", daemon=True)
|
|
2790
|
+
thread.start()
|
|
2791
|
+
try:
|
|
2792
|
+
try:
|
|
2793
|
+
await asyncio.wait_for(asyncio.shield(done), timeout=self.timeout_sec)
|
|
2794
|
+
except asyncio.TimeoutError:
|
|
2795
|
+
with decision:
|
|
2796
|
+
bound_won = claimed["by"] is None
|
|
2797
|
+
if bound_won:
|
|
2798
|
+
claimed["by"] = "bound"
|
|
2799
|
+
if bound_won:
|
|
2800
|
+
raise
|
|
2801
|
+
# The load returned AT the bound: its settle is already on
|
|
2802
|
+
# its way to this loop. It counts as in time.
|
|
2803
|
+
await done
|
|
2804
|
+
except asyncio.TimeoutError:
|
|
2805
|
+
self.timed_out += 1
|
|
2806
|
+
self._abandoned_started = started
|
|
2807
|
+
self.poisoned_by = {
|
|
2808
|
+
"index": index,
|
|
2809
|
+
"modelId": model_id,
|
|
2810
|
+
"boundMs": round(self.timeout_sec * 1000),
|
|
2811
|
+
"at": round(time.time() * 1000),
|
|
2812
|
+
}
|
|
2813
|
+
raise ModelLoadTimeout(
|
|
2814
|
+
f"load of {model_id} (index {index}) did not return within "
|
|
2815
|
+
f"{round(self.timeout_sec * 1000)}ms — it keeps running so its cache "
|
|
2816
|
+
f"can be written; no new load starts until it returns"
|
|
2817
|
+
) from None
|
|
2818
|
+
except Exception:
|
|
2819
|
+
self.failed += 1
|
|
2820
|
+
raise
|
|
2821
|
+
finally:
|
|
2822
|
+
self.in_flight = None
|
|
2823
|
+
self.completed += 1
|
|
2824
|
+
|
|
2825
|
+
|
|
2826
|
+
async def _handle_model_command(
|
|
2827
|
+
models: "list[ModelSlot]", cmd: dict, loader: ModelLoader,
|
|
2828
|
+
) -> dict:
|
|
2829
|
+
"""Every model-management command, in order; `load`/`replace` off the loop.
|
|
2830
|
+
|
|
2831
|
+
The new model is loaded into a DETACHED slot and installed only once its
|
|
2832
|
+
load completed inside the bound, so a compile that returns late — after the
|
|
2833
|
+
host was told `compile-timeout` — can never surface as a model the host
|
|
2834
|
+
believes failed. Whatever the index held before keeps serving until that
|
|
2835
|
+
swap: a `replace` that fails or times out leaves the old model in place,
|
|
2836
|
+
and the old slot object is only dropped, never mutated, so a predict
|
|
2837
|
+
already holding it finishes normally.
|
|
2838
|
+
"""
|
|
2839
|
+
action = cmd.get("cmd")
|
|
2840
|
+
if action not in LOAD_COMMANDS:
|
|
2841
|
+
return _handle_command(models, cmd)
|
|
2842
|
+
index = cmd["index"]
|
|
2843
|
+
config = cmd["config"]
|
|
2844
|
+
model_id = _model_id_of(config)
|
|
2845
|
+
while len(models) <= index:
|
|
2846
|
+
models.append(ModelSlot())
|
|
2847
|
+
slot = ModelSlot()
|
|
2848
|
+
sys.stderr.write(
|
|
2849
|
+
f"Model {index} load started: {config['path']} "
|
|
2850
|
+
f"(bound {round(loader.timeout_sec * 1000)}ms)\n"
|
|
2851
|
+
)
|
|
2852
|
+
sys.stderr.flush()
|
|
2853
|
+
t0 = time.perf_counter()
|
|
2854
|
+
try:
|
|
2855
|
+
await loader.load(slot, config, index)
|
|
2856
|
+
except WorkerPoisoned as exc:
|
|
2857
|
+
return {
|
|
2858
|
+
"cmd": action, "index": index, "status": "error",
|
|
2859
|
+
"reason": "worker-poisoned", "modelId": model_id, "poisoned": True,
|
|
2860
|
+
# How long the EARLIER compile has run: the host's hard bound
|
|
2861
|
+
# counts from its start, not from this refusal.
|
|
2862
|
+
"compileElapsedMs": loader.abandoned_elapsed_ms(),
|
|
2863
|
+
"error": str(exc),
|
|
2864
|
+
}
|
|
2865
|
+
except ModelLoadTimeout as exc:
|
|
2866
|
+
sys.stderr.write(f"Model {index} load TIMED OUT: {exc}\n")
|
|
2867
|
+
sys.stderr.flush()
|
|
2868
|
+
return {
|
|
2869
|
+
"cmd": action, "index": index, "status": "error",
|
|
2870
|
+
"reason": "compile-timeout", "modelId": model_id, "poisoned": True,
|
|
2871
|
+
"timeoutMs": round(loader.timeout_sec * 1000),
|
|
2872
|
+
"compileElapsedMs": loader.abandoned_elapsed_ms(),
|
|
2873
|
+
"error": str(exc),
|
|
2874
|
+
}
|
|
2875
|
+
except Exception as exc:
|
|
2876
|
+
return {"cmd": action, "index": index, "status": "error", "error": str(exc)}
|
|
2877
|
+
models[index] = slot
|
|
2878
|
+
load_ms = round((time.perf_counter() - t0) * 1000)
|
|
2879
|
+
verb = "loaded" if action == "load" else "replaced"
|
|
2880
|
+
sys.stderr.write(f"Model {index} {verb}: {config['path']} ({load_ms}ms)\n")
|
|
2881
|
+
sys.stderr.flush()
|
|
2882
|
+
return {"cmd": action, "index": index, "status": "ok", "loadMs": load_ms}
|
|
2883
|
+
|
|
2884
|
+
|
|
2885
|
+
class CommandRouter:
|
|
2886
|
+
"""Routes MSG_COMMAND frames without ever blocking the reader.
|
|
2887
|
+
|
|
2888
|
+
`INLINE_COMMANDS` are answered at once by `inline` (it needs the loop's own
|
|
2889
|
+
state: caches, backpressure, counters). Everything else is queued and
|
|
2890
|
+
consumed IN ORDER by one task — model slots are indices shared with the
|
|
2891
|
+
host, so a `status` or `unload` must never overtake the `load` sent before
|
|
2892
|
+
it.
|
|
2893
|
+
"""
|
|
2894
|
+
|
|
2895
|
+
def __init__(
|
|
2896
|
+
self,
|
|
2897
|
+
models: "list[ModelSlot]",
|
|
2898
|
+
loader: ModelLoader,
|
|
2899
|
+
send: "Callable[[int, dict], Awaitable[None]]",
|
|
2900
|
+
inline: "Callable[[dict], dict]",
|
|
2901
|
+
) -> None:
|
|
2902
|
+
self._models = models
|
|
2903
|
+
self._loader = loader
|
|
2904
|
+
self._send = send
|
|
2905
|
+
self._inline = inline
|
|
2906
|
+
self._queue: "asyncio.Queue[tuple[int, dict]]" = asyncio.Queue()
|
|
2907
|
+
self._consumer: "Optional[asyncio.Task[None]]" = None
|
|
2908
|
+
|
|
2909
|
+
async def submit(self, req_id: int, cmd: dict) -> None:
|
|
2910
|
+
if cmd.get("cmd") in INLINE_COMMANDS:
|
|
2911
|
+
try:
|
|
2912
|
+
response = self._inline(cmd)
|
|
2913
|
+
except Exception as exc:
|
|
2914
|
+
response = {"cmd": cmd.get("cmd") or "unknown", "status": "error", "error": str(exc)}
|
|
2915
|
+
await self._send(req_id, response)
|
|
2916
|
+
return
|
|
2917
|
+
self._queue.put_nowait((req_id, cmd))
|
|
2918
|
+
if self._consumer is None or self._consumer.done():
|
|
2919
|
+
self._consumer = asyncio.create_task(self._consume())
|
|
2920
|
+
|
|
2921
|
+
async def _consume(self) -> None:
|
|
2922
|
+
while not self._queue.empty():
|
|
2923
|
+
req_id, cmd = self._queue.get_nowait()
|
|
2924
|
+
try:
|
|
2925
|
+
response = await _handle_model_command(self._models, cmd, self._loader)
|
|
2926
|
+
except Exception as exc:
|
|
2927
|
+
response = {"cmd": cmd.get("cmd") or "unknown", "status": "error", "error": str(exc)}
|
|
2928
|
+
await self._send(req_id, response)
|
|
2929
|
+
|
|
2930
|
+
|
|
2585
2931
|
# ---------------------------------------------------------------------------
|
|
2586
2932
|
# Main async loop
|
|
2587
2933
|
# ---------------------------------------------------------------------------
|
|
@@ -2631,6 +2977,13 @@ async def _run() -> None:
|
|
|
2631
2977
|
)
|
|
2632
2978
|
sys.stderr.flush()
|
|
2633
2979
|
_init_runtime(runtime, pool_device)
|
|
2980
|
+
# Off-loop, bounded model loads (D653). Startup-config models below still
|
|
2981
|
+
# load inline: nothing is served before `ready`, and the host bounds that
|
|
2982
|
+
# handshake itself.
|
|
2983
|
+
model_loader = ModelLoader(
|
|
2984
|
+
timeout_sec=_model_load_timeout_sec(config),
|
|
2985
|
+
notify=lambda event: writer.send(WORKER_EVENT_REQ_ID, event),
|
|
2986
|
+
)
|
|
2634
2987
|
|
|
2635
2988
|
models: list[ModelSlot] = []
|
|
2636
2989
|
for i, mc in enumerate(config.get("models", [])):
|
|
@@ -2782,6 +3135,33 @@ async def _run() -> None:
|
|
|
2782
3135
|
# Saves ~10ms/call (the PIL resize + numpy + tensor pack cost).
|
|
2783
3136
|
_preprocess_cache: dict[tuple[int, int], tuple[dict, float, tuple[int, int], int, int]] = {}
|
|
2784
3137
|
|
|
3138
|
+
def inline_command(cmd: dict) -> dict:
|
|
3139
|
+
# Needs the loop's own state (caches, backpressure, counters), which
|
|
3140
|
+
# the model-command path never sees. Answered at once — this is the
|
|
3141
|
+
# host's liveness probe, and it must not wait for a compile.
|
|
3142
|
+
if cmd.get("cmd") == "mem_stats":
|
|
3143
|
+
return _mem_stats_payload(
|
|
3144
|
+
models,
|
|
3145
|
+
counters,
|
|
3146
|
+
backpressure,
|
|
3147
|
+
len(frame_cache),
|
|
3148
|
+
len(_preprocess_cache),
|
|
3149
|
+
len(_bench_preprocess_cache),
|
|
3150
|
+
round(time.monotonic() - pool_started_monotonic),
|
|
3151
|
+
model_load=model_loader.snapshot(),
|
|
3152
|
+
)
|
|
3153
|
+
fid = cmd.get("frameId", -1)
|
|
3154
|
+
frame_cache.pop(fid, None)
|
|
3155
|
+
# Purge preprocessed tensor caches for this frame
|
|
3156
|
+
for k in [k for k in _preprocess_cache if k[0] == fid]:
|
|
3157
|
+
del _preprocess_cache[k]
|
|
3158
|
+
# Also purge the bench preprocess cache
|
|
3159
|
+
for k in [k for k in _bench_preprocess_cache if k[0] == fid]:
|
|
3160
|
+
del _bench_preprocess_cache[k]
|
|
3161
|
+
return {"cmd": "uncache_frame", "status": "ok", "frameId": fid}
|
|
3162
|
+
|
|
3163
|
+
command_router = CommandRouter(models, model_loader, writer.send, inline_command)
|
|
3164
|
+
|
|
2785
3165
|
def get_accumulator(model_idx: int) -> WindowAccumulator:
|
|
2786
3166
|
acc = accumulators.get(model_idx)
|
|
2787
3167
|
if acc is None:
|
|
@@ -2924,35 +3304,13 @@ async def _run() -> None:
|
|
|
2924
3304
|
counters.commands += 1
|
|
2925
3305
|
try:
|
|
2926
3306
|
cmd = json.loads(payload)
|
|
2927
|
-
# Handled inline (needs the loop's own state: caches,
|
|
2928
|
-
# backpressure, counters) — _handle_command only sees models.
|
|
2929
|
-
if cmd.get("cmd") == "mem_stats":
|
|
2930
|
-
response = _mem_stats_payload(
|
|
2931
|
-
models,
|
|
2932
|
-
counters,
|
|
2933
|
-
backpressure,
|
|
2934
|
-
len(frame_cache),
|
|
2935
|
-
len(_preprocess_cache),
|
|
2936
|
-
len(_bench_preprocess_cache),
|
|
2937
|
-
round(time.monotonic() - pool_started_monotonic),
|
|
2938
|
-
)
|
|
2939
|
-
elif cmd.get("cmd") == "uncache_frame":
|
|
2940
|
-
fid = cmd.get("frameId", -1)
|
|
2941
|
-
removed_img = frame_cache.pop(fid, None)
|
|
2942
|
-
# Purge preprocessed tensor caches for this frame
|
|
2943
|
-
to_del = [k for k in _preprocess_cache if k[0] == fid]
|
|
2944
|
-
for k in to_del:
|
|
2945
|
-
del _preprocess_cache[k]
|
|
2946
|
-
# Also purge the bench preprocess cache
|
|
2947
|
-
to_del2 = [k for k in _bench_preprocess_cache if k[0] == fid]
|
|
2948
|
-
for k in to_del2:
|
|
2949
|
-
del _bench_preprocess_cache[k]
|
|
2950
|
-
response = {"cmd": "uncache_frame", "status": "ok", "frameId": fid}
|
|
2951
|
-
else:
|
|
2952
|
-
response = _handle_command(models, cmd)
|
|
2953
3307
|
except Exception as exc:
|
|
2954
|
-
|
|
2955
|
-
|
|
3308
|
+
await writer.send(req_id, {"cmd": "unknown", "status": "error", "error": str(exc)})
|
|
3309
|
+
continue
|
|
3310
|
+
# NEVER awaits a model load: `load`/`replace` are queued and
|
|
3311
|
+
# compiled on their own thread under a bound (D653), so the next
|
|
3312
|
+
# frame is read while a GPU compile runs.
|
|
3313
|
+
await command_router.submit(req_id, cmd)
|
|
2956
3314
|
|
|
2957
3315
|
elif msg_type == MSG_INFER_JPEG:
|
|
2958
3316
|
counters.infer_jpeg += 1
|
|
@@ -16,6 +16,7 @@ from .yamnet import postprocess_yamnet
|
|
|
16
16
|
from .ssd import postprocess_ssd
|
|
17
17
|
from .rfdetr import postprocess_rfdetr
|
|
18
18
|
from .yolonas import postprocess_yolonas
|
|
19
|
+
from .yunet import postprocess_yunet
|
|
19
20
|
|
|
20
21
|
POSTPROCESSORS = {
|
|
21
22
|
"yolo": postprocess_yolo,
|
|
@@ -31,4 +32,5 @@ POSTPROCESSORS = {
|
|
|
31
32
|
"ssd": postprocess_ssd,
|
|
32
33
|
"rfdetr": postprocess_rfdetr,
|
|
33
34
|
"yolonas": postprocess_yolonas,
|
|
35
|
+
"yunet": postprocess_yunet,
|
|
34
36
|
}
|