@camstack/addon-pipeline 1.2.295 → 1.2.296

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/THIRD_PARTY_MODELS.md +8 -0
  2. package/dist/audio-analyzer/index.js +2 -2
  3. package/dist/audio-analyzer/index.mjs +2 -2
  4. package/dist/{default-detection-model-CAUBVbgK.mjs → default-detection-model-Co578D8C.mjs} +75 -46
  5. package/dist/{default-detection-model-wNEISVA9.js → default-detection-model-D1daTtqT.js} +75 -46
  6. package/dist/detection-pipeline/index.js +1023 -80
  7. package/dist/detection-pipeline/index.mjs +1023 -80
  8. package/dist/{dist-BsVcf5wO.js → dist-8up-f2TX.js} +877 -365
  9. package/dist/{dist-CuxSNLKW.mjs → dist-CCd0Q3nr.mjs} +872 -366
  10. package/dist/motion-wasm/index.js +1 -1
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node-Cqb1QiXk.mjs → node-DgMSXSWP.mjs} +1 -1
  13. package/dist/{node-UFk6I2f6.js → node-lpQgHes9.js} +1 -1
  14. package/dist/pipeline-runner/index.js +789 -203
  15. package/dist/pipeline-runner/index.mjs +789 -203
  16. package/dist/{process-memory-CJV29sPf.js → process-memory-CX_92V_r.js} +1 -1
  17. package/dist/{process-memory-D0Nvs9rr.mjs → process-memory-DFC_O5zE.mjs} +1 -1
  18. package/dist/recorder/index.js +4 -6
  19. package/dist/recorder/index.mjs +4 -6
  20. package/dist/{segment-demux-js-DGwmf5hu.js → segment-demux-js-DzBx6NN2.js} +1 -1
  21. package/dist/{segment-demux-js-DIDWw1iE.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
  22. package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
  23. package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
  24. package/dist/stream-broker/_stub.js +2 -2
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-B7nBFqva.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
  27. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
  28. package/dist/stream-broker/demux-worker-child.js +1 -1
  29. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  30. package/dist/stream-broker/{hostInit-BKlb3qac.mjs → hostInit-BPtppL3W.mjs} +2 -2
  31. package/dist/stream-broker/index.js +4 -4
  32. package/dist/stream-broker/index.mjs +4 -4
  33. package/dist/stream-broker/remoteEntry.js +1 -1
  34. package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
  35. package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
  36. package/package.json +1 -1
  37. package/python/inference_pool.py +422 -64
  38. package/python/test_inference_pool_compile_off_loop.py +414 -0
  39. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CCIyBvRa.mjs +0 -26
  40. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DeD_UFdb.mjs +0 -26
@@ -3,11 +3,11 @@ Object.defineProperties(exports, {
3
3
  [Symbol.toStringTag]: { value: "Module" }
4
4
  });
5
5
  const require_chunk = require("../chunk-emK7D4bc.js");
6
- const require_dist = require("../dist-BsVcf5wO.js");
7
- const require_node = require("../node-UFk6I2f6.js");
8
- const require_default_detection_model = require("../default-detection-model-wNEISVA9.js");
6
+ const require_dist = require("../dist-8up-f2TX.js");
7
+ const require_node = require("../node-lpQgHes9.js");
8
+ const require_default_detection_model = require("../default-detection-model-D1daTtqT.js");
9
9
  const require_lazy_sharp = require("../lazy-sharp-BuwKlQLR.js");
10
- const require_process_memory = require("../process-memory-CJV29sPf.js");
10
+ const require_process_memory = require("../process-memory-CX_92V_r.js");
11
11
  let node_os = require("node:os");
12
12
  node_os = require_chunk.__toESM(node_os);
13
13
  let node_fs = require("node:fs");
@@ -422,6 +422,12 @@ var ClusterModelSource = class {
422
422
  const next = require_dist.pickClusterStepModels(view);
423
423
  const nextSettings = require_dist.pickClusterStepSettings(view);
424
424
  this.lastReadAtMs = this.now();
425
+ const invalid = require_dist.pickInvalidClusterStepModels(view);
426
+ if (invalid.length > 0) this.logger.warn("cluster step model row holds a value that is not a model id — running the catalog default for that step; no re-embed pass will run on it", { meta: {
427
+ owner: CLUSTER_MODEL_OWNER_ADDON_ID,
428
+ invalid,
429
+ inUse: Object.fromEntries(invalid.map((e) => [e.stepId, next[e.stepId]]))
430
+ } });
425
431
  const changed = Object.keys(next).filter((stepId) => next[stepId] !== this.models[stepId]);
426
432
  const settingsChanged = Object.keys(nextSettings).some((stepId) => {
427
433
  const from = this.settings[stepId] ?? {};
@@ -569,6 +575,98 @@ var DeviceOverrideMirror = class {
569
575
  }
570
576
  };
571
577
  //#endregion
578
+ //#region src/detection-pipeline/engine/pool-worker-health.ts
579
+ /**
580
+ * The one table of load bounds (D653, fix round 1).
581
+ *
582
+ * The soft bound decides how soon the operator HEARS about a slow compile; the
583
+ * hard bound decides when a compile is given up for hung. A single 120 s bound
584
+ * that also killed the compile was wrong twice: a slow but finite compile was
585
+ * killed before it could write its cache, so every respawn faced the same cold
586
+ * compile and three of them walked the device to `failed`; and on CoreML the
587
+ * bound was shorter than the cross-process compile-cache lock wait alone.
588
+ */
589
+ var POOL_MODEL_LOAD_BOUNDS = {
590
+ openvino: {
591
+ softMs: 12e4,
592
+ hardMs: 6e5,
593
+ why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
594
+ gilMayBeHeldDuringCompile: false,
595
+ gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
596
+ },
597
+ coreml: {
598
+ softMs: 3e5,
599
+ hardMs: 9e5,
600
+ why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
601
+ gilMayBeHeldDuringCompile: true,
602
+ gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
603
+ },
604
+ onnxruntime: {
605
+ softMs: 12e4,
606
+ hardMs: 6e5,
607
+ why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
608
+ gilMayBeHeldDuringCompile: false,
609
+ gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
610
+ },
611
+ edgetpu: {
612
+ softMs: 12e4,
613
+ hardMs: 6e5,
614
+ why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
615
+ gilMayBeHeldDuringCompile: false,
616
+ gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
617
+ }
618
+ };
619
+ /**
620
+ * The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
621
+ * so the worker's NAMED answer always lands first.
622
+ */
623
+ var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
624
+ function chargesDeviceBudget(reason) {
625
+ return reason !== "compile-hung";
626
+ }
627
+ /**
628
+ * Tracks live inference outcomes on one worker and says when its executor has
629
+ * produced nothing for too long. Consulted on demand — no timer.
630
+ *
631
+ * A SHED (`dropped`) does not count as a result: the Python loop answers sheds
632
+ * itself, so a worker whose executor threads are hung keeps shedding happily.
633
+ */
634
+ var InferStallDetector = class {
635
+ unanswered = 0;
636
+ lastResultAt;
637
+ /**
638
+ * When the current streak of unanswered requests began — the first one after
639
+ * a result. IDLE time is not silence: quiet cameras send nothing, and a
640
+ * worker that was asked nothing has failed nothing. Measuring from the last
641
+ * result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
642
+ * cameras) poison a healthy worker and charge the device.
643
+ */
644
+ streakStartedAt = null;
645
+ constructor(now) {
646
+ this.lastResultAt = now;
647
+ }
648
+ /** A reply that carried a real result (or a real error) from the executor. */
649
+ noteResult(now) {
650
+ this.unanswered = 0;
651
+ this.lastResultAt = now;
652
+ this.streakStartedAt = null;
653
+ }
654
+ /**
655
+ * A live request that ended without a result (deadline expiry or a shed).
656
+ * Returns the stall when the worker has crossed both thresholds.
657
+ */
658
+ noteUnanswered(now) {
659
+ this.unanswered += 1;
660
+ if (this.streakStartedAt === null) this.streakStartedAt = now;
661
+ const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
662
+ if (this.unanswered >= 20 && silentMs >= 3e4) return {
663
+ unanswered: this.unanswered,
664
+ silentMs
665
+ };
666
+ return null;
667
+ }
668
+ };
669
+ //#endregion
572
670
  //#region src/detection-pipeline/engine/shared-inference-pool.ts
573
671
  /**
574
672
  * SharedInferencePool — TypeScript wrapper for inference_pool.py.
@@ -610,6 +708,27 @@ var RAW_FMT_CODE = {
610
708
  gray: 2
611
709
  };
612
710
  /**
711
+ * A load the worker answered with a machine-readable failure class. The
712
+ * provider tells a `compile-timeout` (counted against THAT MODEL on the
713
+ * device) from an ordinary failure by `reason`, never by parsing text.
714
+ */
715
+ var PoolModelLoadError = class extends Error {
716
+ reason;
717
+ /** The model's file stem, as the worker named it. */
718
+ model;
719
+ constructor(message, reason, model) {
720
+ super(message);
721
+ this.name = "PoolModelLoadError";
722
+ this.reason = reason;
723
+ this.model = model;
724
+ }
725
+ };
726
+ /**
727
+ * Request id the worker uses for UNSOLICITED events — a reply to no request
728
+ * (`compile-finished-late`). Never allocated to a request.
729
+ */
730
+ var WORKER_EVENT_REQ_ID = 4294967295;
731
+ /**
613
732
  * Per-inference-request reply timeout (ms). Turns a wedged request (worker
614
733
  * alive but no reply) into a rejection so every caller settles — runtime frame
615
734
  * dispatch drops the frame; the benchmark full-tree run rejects the wedged
@@ -618,6 +737,22 @@ var RAW_FMT_CODE = {
618
737
  * large frame) never trips; override via env for constrained hardware.
619
738
  */
620
739
  var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
740
+ /** Commands that compile a model — the only ones that get the load deadline. */
741
+ var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
742
+ /**
743
+ * Deadline of the liveness probe (`mem_stats`) sent after a command times out.
744
+ * The worker answers `mem_stats` on its event loop, never behind a compile
745
+ * (D653), so a healthy worker replies in milliseconds even mid-compile; ten
746
+ * seconds only has to outlast a GC pause or a burst of inference replies.
747
+ */
748
+ var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
749
+ /**
750
+ * How long a poisoned worker may keep its in-flight LIVE requests before it is
751
+ * killed anyway. A poisoned worker takes no new work, and every live request
752
+ * carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
753
+ * by then; the second of slack keeps the backstop from racing the last reply.
754
+ */
755
+ var POOL_POISON_DRAIN_SLACK_MS = 1e3;
621
756
  /**
622
757
  * Max inference requests outstanding to ONE worker before new ones are SHED.
623
758
  *
@@ -805,6 +940,27 @@ var PoolWorker = class {
805
940
  wireVersion = 1;
806
941
  nextRequestId = 1;
807
942
  ready = false;
943
+ /** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
944
+ * a poisoned worker is recycled, not revived. */
945
+ poisonVerdict = null;
946
+ /** The kill of a poisoned worker has been started. */
947
+ recycling = false;
948
+ /** Kills a poisoned worker whose drain did not finish in time. */
949
+ recycleBackstop = null;
950
+ /** A liveness probe is in flight — one at a time, never a probe storm. */
951
+ probeInFlight = false;
952
+ /** The child has exited (any cause). */
953
+ exited = false;
954
+ /** A compile past its soft bound, still running in the worker (D653). */
955
+ abandonedCompile = null;
956
+ /** Says when live inference has produced nothing for too long (D653). */
957
+ inferStall = new InferStallDetector(Date.now());
958
+ /**
959
+ * A load's HOST deadline expired with no reply since the last reply of any
960
+ * kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
961
+ * is over for this worker until something — anything — comes back.
962
+ */
963
+ loadDeadlineMissed = false;
808
964
  log;
809
965
  opts;
810
966
  constructor(opts) {
@@ -829,6 +985,24 @@ var PoolWorker = class {
829
985
  isReady() {
830
986
  return this.ready;
831
987
  }
988
+ /** Why this worker was recycled, typed for the provider's budget, or `null`. */
989
+ getDeathCause() {
990
+ const v = this.poisonVerdict;
991
+ const message = this.getPoisonDescription();
992
+ if (v === null || message === null) return null;
993
+ return {
994
+ reason: v.reason,
995
+ chargesDeviceBudget: chargesDeviceBudget(v.reason),
996
+ message
997
+ };
998
+ }
999
+ /** Why this live worker was declared unusable, or `null` (D653). */
1000
+ getPoisonDescription() {
1001
+ const v = this.poisonVerdict;
1002
+ if (v === null) return null;
1003
+ const target = v.command.model !== null ? ` of ${v.command.model}` : "";
1004
+ return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
1005
+ }
832
1006
  async initialize(initialModels) {
833
1007
  this.process = (0, node_child_process.spawn)(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
834
1008
  "pipe",
@@ -852,7 +1026,11 @@ var PoolWorker = class {
852
1026
  const spawnedProcess = this.process;
853
1027
  spawnedProcess.on("exit", (code, signal) => {
854
1028
  this.ready = false;
1029
+ this.exited = true;
1030
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1031
+ if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
855
1032
  if (this.process !== spawnedProcess) return;
1033
+ const poisonedBy = this.getPoisonDescription();
856
1034
  this.log.error("Worker process exited", { meta: {
857
1035
  worker: this.opts.workerLabel,
858
1036
  pid: spawnedProcess.pid ?? null,
@@ -860,9 +1038,10 @@ var PoolWorker = class {
860
1038
  device: this.opts.device ?? "default",
861
1039
  code,
862
1040
  signal,
863
- inFlight: this.pending.size
1041
+ inFlight: this.pending.size,
1042
+ ...poisonedBy !== null ? { poisonedBy } : {}
864
1043
  } });
865
- this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
1044
+ this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
866
1045
  });
867
1046
  this.process.stdout.on("data", (chunk) => {
868
1047
  const t0 = Date.now();
@@ -876,6 +1055,7 @@ var PoolWorker = class {
876
1055
  runtime: this.opts.poolRuntime,
877
1056
  concurrency: this.opts.concurrency,
878
1057
  protocolVersion: 2,
1058
+ modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
879
1059
  models: initialModels.map((m) => serializeModelConfig(m))
880
1060
  };
881
1061
  if (this.opts.device) config["device"] = this.opts.device;
@@ -898,6 +1078,7 @@ var PoolWorker = class {
898
1078
  clearTimeout(timeout);
899
1079
  if (result["status"] === "ready") {
900
1080
  this.ready = true;
1081
+ this.inferStall.noteResult(Date.now());
901
1082
  const loadedCount = result["models"];
902
1083
  const startupMs = result["startupMs"];
903
1084
  const workers = result["workers"] ?? 1;
@@ -920,7 +1101,7 @@ var PoolWorker = class {
920
1101
  async infer(modelByte, jpeg, deviceId) {
921
1102
  this.ensureReady();
922
1103
  const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
923
- return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
1104
+ return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
924
1105
  }
925
1106
  async inferRaw(modelByte, raw, width, height, format, deviceId) {
926
1107
  this.ensureReady();
@@ -973,12 +1154,18 @@ var PoolWorker = class {
973
1154
  const payload = Buffer.allocUnsafe(5);
974
1155
  payload[0] = modelByte;
975
1156
  payload.writeUInt32LE(frameId, 1);
976
- return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
1157
+ return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
977
1158
  }
978
1159
  async sendCommand(cmd) {
979
1160
  this.ensureReady();
1161
+ const command = describeCommand(cmd);
980
1162
  const payload = Buffer.from(JSON.stringify(cmd), "utf8");
981
- return await this.dispatch(MSG_COMMAND, payload);
1163
+ const sentAt = Date.now();
1164
+ if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
1165
+ const raw = await this.dispatch(MSG_COMMAND, payload, { command });
1166
+ if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
1167
+ this.inspectCommandReply(raw, command, sentAt);
1168
+ return raw;
982
1169
  }
983
1170
  /** Command whose reply shape is NOT the load/unload/replace/status envelope
984
1171
  * (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
@@ -986,13 +1173,15 @@ var PoolWorker = class {
986
1173
  async sendRawCommand(cmd) {
987
1174
  this.ensureReady();
988
1175
  const payload = Buffer.from(JSON.stringify(cmd), "utf8");
989
- return this.dispatch(MSG_COMMAND, payload);
1176
+ return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
990
1177
  }
991
1178
  async dispose() {
992
1179
  const proc = this.process;
993
1180
  if (!proc) return;
994
1181
  this.process = null;
995
1182
  this.ready = false;
1183
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1184
+ if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
996
1185
  await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
997
1186
  this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
998
1187
  }
@@ -1035,8 +1224,11 @@ var PoolWorker = class {
1035
1224
  }
1036
1225
  /** Live inference gets seconds; commands and model loads keep the long
1037
1226
  * timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
1038
- deadlineFor(msgType) {
1039
- return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
1227
+ deadlineFor(msgType, opts = {}) {
1228
+ if (opts.timeoutMs !== void 0) return opts.timeoutMs;
1229
+ if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
1230
+ if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
1231
+ return POOL_INFER_TIMEOUT_MS;
1040
1232
  }
1041
1233
  /**
1042
1234
  * The request's ABSOLUTE deadline for the wire (D350), or `null` when this
@@ -1053,16 +1245,18 @@ var PoolWorker = class {
1053
1245
  stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
1054
1246
  return stamp;
1055
1247
  }
1056
- dispatch(msgType, payload, deviceId) {
1057
- const shed = this.shedIfSaturated(msgType, deviceId);
1248
+ dispatch(msgType, payload, opts = {}) {
1249
+ const shed = this.shedIfSaturated(msgType, opts.deviceId);
1058
1250
  if (shed) return Promise.resolve(shed);
1059
1251
  const reqId = this.allocRequestId();
1060
1252
  return new Promise((resolve, reject) => {
1061
- const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1253
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
1062
1254
  this.pending.set(reqId, {
1063
1255
  resolve,
1064
1256
  reject,
1065
- timer
1257
+ timer,
1258
+ ...opts.command !== void 0 ? { command: opts.command } : {},
1259
+ ...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
1066
1260
  });
1067
1261
  try {
1068
1262
  const stamp = this.deadlineStamp(msgType);
@@ -1093,8 +1287,9 @@ var PoolWorker = class {
1093
1287
  * - Anything else (commands, model loads) still REJECTS with the error the
1094
1288
  * dashboards grep for — a lost command is a fault, not flow control.
1095
1289
  */
1096
- armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
1097
- const timeoutMs = this.deadlineFor(msgType);
1290
+ armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
1291
+ const { deviceId, command } = opts;
1292
+ const timeoutMs = this.deadlineFor(msgType, opts);
1098
1293
  const timer = setTimeout(() => {
1099
1294
  if (this.pending.delete(reqId)) {
1100
1295
  if (SHEDDABLE_MSG_TYPES.has(msgType)) {
@@ -1118,6 +1313,7 @@ var PoolWorker = class {
1118
1313
  dropped: true,
1119
1314
  shedReason: "deadline-expired"
1120
1315
  });
1316
+ this.noteInferUnanswered();
1121
1317
  return;
1122
1318
  }
1123
1319
  this.timedOutCount++;
@@ -1130,10 +1326,20 @@ var PoolWorker = class {
1130
1326
  device: this.opts.device ?? "default",
1131
1327
  inFlight: this.pending.size,
1132
1328
  reqId,
1133
- timeoutMs
1329
+ timeoutMs,
1330
+ ...command !== void 0 ? {
1331
+ command: command.cmd,
1332
+ modelIndex: command.modelIndex,
1333
+ model: command.model
1334
+ } : {}
1134
1335
  }
1135
1336
  });
1136
- reject(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`));
1337
+ const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
1338
+ if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
1339
+ this.loadDeadlineMissed = true;
1340
+ reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
1341
+ } else reject(new Error(message));
1342
+ if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
1137
1343
  }
1138
1344
  }, timeoutMs);
1139
1345
  timer.unref?.();
@@ -1144,11 +1350,12 @@ var PoolWorker = class {
1144
1350
  if (shed) return Promise.resolve(shed);
1145
1351
  const reqId = this.allocRequestId();
1146
1352
  return new Promise((resolve, reject) => {
1147
- const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1353
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
1148
1354
  this.pending.set(reqId, {
1149
1355
  resolve,
1150
1356
  reject,
1151
- timer
1357
+ timer,
1358
+ ...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
1152
1359
  });
1153
1360
  try {
1154
1361
  if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
@@ -1170,10 +1377,10 @@ var PoolWorker = class {
1170
1377
  }
1171
1378
  allocRequestId() {
1172
1379
  let id = this.nextRequestId;
1173
- this.nextRequestId = id >= 4294967295 ? 1 : id + 1;
1380
+ this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
1174
1381
  while (this.pending.has(id)) {
1175
1382
  id = this.nextRequestId;
1176
- this.nextRequestId = id >= 4294967295 ? 1 : id + 1;
1383
+ this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
1177
1384
  }
1178
1385
  return id;
1179
1386
  }
@@ -1188,6 +1395,8 @@ var PoolWorker = class {
1188
1395
  this.process.stdin.write(payload);
1189
1396
  }
1190
1397
  ensureReady() {
1398
+ const poisonedBy = this.getPoisonDescription();
1399
+ if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
1191
1400
  if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
1192
1401
  }
1193
1402
  /** Time spent in `Buffer.concat` since the last report. */
@@ -1227,23 +1436,273 @@ var PoolWorker = class {
1227
1436
  const reqId = this.receiveBuffer.readUInt32LE(4);
1228
1437
  const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
1229
1438
  this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
1439
+ if (reqId === WORKER_EVENT_REQ_ID) {
1440
+ this.handleWorkerEvent(jsonBytes);
1441
+ continue;
1442
+ }
1230
1443
  const entry = this.pending.get(reqId);
1231
1444
  if (!entry) {
1232
1445
  this.log.warn("Response for unknown request id", { meta: {
1233
1446
  worker: this.opts.workerLabel,
1234
1447
  reqId
1235
1448
  } });
1449
+ this.noteLateReply(jsonBytes);
1236
1450
  continue;
1237
1451
  }
1238
1452
  this.pending.delete(reqId);
1239
1453
  if (entry.timer) clearTimeout(entry.timer);
1240
1454
  try {
1241
1455
  const parsed = JSON.parse(jsonBytes.toString("utf8"));
1456
+ if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
1457
+ else this.inferStall.noteResult(Date.now());
1242
1458
  entry.resolve(parsed);
1243
1459
  } catch (err) {
1244
1460
  entry.reject(err instanceof Error ? err : new Error(String(err)));
1245
1461
  }
1246
1462
  }
1463
+ this.noteAnyReply();
1464
+ if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
1465
+ }
1466
+ /**
1467
+ * Something came back from the worker: its loop was alive just now, so the
1468
+ * missed-deadline verdict is lifted.
1469
+ */
1470
+ noteAnyReply() {
1471
+ this.loadDeadlineMissed = false;
1472
+ }
1473
+ /**
1474
+ * The invariant the load deadline depends on (D653 § 2): loads into one
1475
+ * pool are serialised, so at most ONE load is outstanding per worker. The
1476
+ * worker runs model commands in order and starts a load's soft bound only
1477
+ * when it begins, while the host's deadline for it runs from the SEND. A
1478
+ * load queued behind another could therefore see its host deadline fire
1479
+ * before the worker's named answer. Nothing in the provider issues that; if
1480
+ * anything ever does, this says so.
1481
+ */
1482
+ warnIfLoadQueued(next) {
1483
+ const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
1484
+ if (ahead.length === 0) return;
1485
+ this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
1486
+ worker: this.opts.workerLabel,
1487
+ pid: this.getPid(),
1488
+ runtime: this.opts.poolRuntime,
1489
+ device: this.opts.device ?? "default",
1490
+ model: next.model,
1491
+ queuedBehind: ahead.map((c) => c.model)
1492
+ } });
1493
+ }
1494
+ hasPendingLoad() {
1495
+ for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
1496
+ return false;
1497
+ }
1498
+ /**
1499
+ * A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
1500
+ * or that an earlier one still has (`worker-poisoned`). The worker is not
1501
+ * recycled for it (fix round 1): the compile keeps running so a slow but
1502
+ * finite one writes its cache, the loaded models keep serving, and the worker
1503
+ * starts no new load. Only a compile still running at the HARD bound costs
1504
+ * the process.
1505
+ */
1506
+ inspectCommandReply(raw, command, sentAt) {
1507
+ const reason = raw["reason"];
1508
+ if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
1509
+ if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
1510
+ const model = raw["modelId"];
1511
+ const abandoned = {
1512
+ ...command,
1513
+ model: typeof model === "string" ? model : command.model
1514
+ };
1515
+ const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
1516
+ const reported = raw["compileElapsedMs"];
1517
+ const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
1518
+ const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
1519
+ hardTimer.unref?.();
1520
+ this.abandonedCompile = {
1521
+ command: abandoned,
1522
+ since: Date.now(),
1523
+ hardTimer
1524
+ };
1525
+ this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
1526
+ worker: this.opts.workerLabel,
1527
+ pid: this.getPid(),
1528
+ runtime: this.opts.poolRuntime,
1529
+ device: this.opts.device ?? "default",
1530
+ model: abandoned.model,
1531
+ modelIndex: abandoned.modelIndex,
1532
+ softMs: bounds.softMs,
1533
+ hardMs: bounds.hardMs,
1534
+ loads: "refused until it returns"
1535
+ } });
1536
+ }
1537
+ /** An unsolicited worker event (`compile-finished-late`). */
1538
+ handleWorkerEvent(jsonBytes) {
1539
+ let event;
1540
+ try {
1541
+ event = JSON.parse(jsonBytes.toString("utf8"));
1542
+ } catch {
1543
+ return;
1544
+ }
1545
+ if (event["event"] !== "compile-finished-late") return;
1546
+ const abandoned = this.abandonedCompile;
1547
+ if (abandoned !== null) clearTimeout(abandoned.hardTimer);
1548
+ this.abandonedCompile = null;
1549
+ this.inferStall.noteResult(Date.now());
1550
+ const ok = event["ok"] === true;
1551
+ const meta = {
1552
+ worker: this.opts.workerLabel,
1553
+ pid: this.getPid(),
1554
+ runtime: this.opts.poolRuntime,
1555
+ device: this.opts.device ?? "default",
1556
+ model: typeof event["modelId"] === "string" ? event["modelId"] : null,
1557
+ elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
1558
+ ...typeof event["error"] === "string" ? { error: event["error"] } : {}
1559
+ };
1560
+ if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
1561
+ else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
1562
+ this.opts.onCompileFinishedLate?.({
1563
+ model: meta.model,
1564
+ ok,
1565
+ ...typeof event["error"] === "string" ? { error: event["error"] } : {}
1566
+ });
1567
+ }
1568
+ /**
1569
+ * The GIL grace (D653 round 2): may a silent loop be a compile holding the
1570
+ * GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
1571
+ * on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
1572
+ *
1573
+ * NOT while a compile runs past its soft bound: the worker REPLIED
1574
+ * `compile-timeout`, so its loop is proven alive, and an inference hang
1575
+ * behind that compile — the 2026-09-26 incident exactly — must be seen on
1576
+ * the normal rule, not ~600-900 s later at the hard bound.
1577
+ */
1578
+ gilGraceActive() {
1579
+ if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
1580
+ if (this.loadDeadlineMissed) return false;
1581
+ return this.hasPendingLoad();
1582
+ }
1583
+ /**
1584
+ * The reason a silent worker is recycled with: a freeze that began with a
1585
+ * load missing its deadline is that compile's doing (`compile-hung`, charged
1586
+ * to the model, not the device); anything else is the device's.
1587
+ */
1588
+ silenceReason(fallback) {
1589
+ return this.loadDeadlineMissed ? "compile-hung" : fallback;
1590
+ }
1591
+ /** A live request ended with no result: judge the executor (D653, item 4). */
1592
+ noteInferUnanswered() {
1593
+ const stall = this.inferStall.noteUnanswered(Date.now());
1594
+ if (stall === null || this.gilGraceActive()) return;
1595
+ this.poison(this.silenceReason("infer-unresponsive"), {
1596
+ cmd: "infer",
1597
+ modelIndex: null,
1598
+ model: null
1599
+ }, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
1600
+ }
1601
+ /** A reply whose request was already abandoned: a real result still proves the executor runs. */
1602
+ noteLateReply(jsonBytes) {
1603
+ try {
1604
+ const parsed = JSON.parse(jsonBytes.toString("utf8"));
1605
+ if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
1606
+ } catch {}
1607
+ }
1608
+ /**
1609
+ * Ask a worker whose command just timed out whether its loop still answers.
1610
+ * `mem_stats` is served on the loop, never behind a compile, so a worker
1611
+ * that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
1612
+ * slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
1613
+ * `mem_stats` lost while the process looked alive.
1614
+ */
1615
+ probeLiveness(timedOut) {
1616
+ if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
1617
+ this.probeInFlight = true;
1618
+ this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
1619
+ command: {
1620
+ cmd: "mem_stats",
1621
+ modelIndex: null,
1622
+ model: null
1623
+ },
1624
+ timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
1625
+ probe: true
1626
+ }).then(() => {
1627
+ this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
1628
+ worker: this.opts.workerLabel,
1629
+ pid: this.getPid(),
1630
+ runtime: this.opts.poolRuntime,
1631
+ device: this.opts.device ?? "default",
1632
+ command: timedOut.cmd,
1633
+ modelIndex: timedOut.modelIndex,
1634
+ model: timedOut.model
1635
+ } });
1636
+ }, (err) => {
1637
+ if (this.gilGraceActive()) {
1638
+ this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
1639
+ worker: this.opts.workerLabel,
1640
+ pid: this.getPid(),
1641
+ runtime: this.opts.poolRuntime,
1642
+ device: this.opts.device ?? "default",
1643
+ command: timedOut.cmd,
1644
+ model: timedOut.model
1645
+ } });
1646
+ return;
1647
+ }
1648
+ this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
1649
+ }).finally(() => {
1650
+ this.probeInFlight = false;
1651
+ });
1652
+ }
1653
+ /**
1654
+ * Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
1655
+ * and recycle it once it has drained.
1656
+ *
1657
+ * `ready = false` is what hands the pool back to the provider: its next
1658
+ * dispatch finds the factory not ready and condemns it under the per-device
1659
+ * restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
1660
+ * excludes) — the same path a crashed worker takes, except that a
1661
+ * `compile-hung` death is not charged to the device (see PoolPoisonReason).
1662
+ * An unresponsive worker is killed at once (it will answer nothing it
1663
+ * holds); a compile-hung one keeps serving its loaded models until its
1664
+ * in-flight requests are answered, bounded by the live deadline.
1665
+ */
1666
+ poison(reason, command, detail) {
1667
+ if (this.poisonVerdict !== null || this.exited || this.process === null) return;
1668
+ this.poisonVerdict = {
1669
+ reason,
1670
+ command,
1671
+ detail,
1672
+ pid: this.getPid()
1673
+ };
1674
+ this.ready = false;
1675
+ const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
1676
+ this.log.error("pool worker POISONED — recycling it", { meta: {
1677
+ worker: this.opts.workerLabel,
1678
+ pid: this.getPid(),
1679
+ runtime: this.opts.poolRuntime,
1680
+ device: this.opts.device ?? "default",
1681
+ reason,
1682
+ command: command.cmd,
1683
+ modelIndex: command.modelIndex,
1684
+ model: command.model,
1685
+ detail,
1686
+ inFlight: this.pending.size,
1687
+ inFlightCommands
1688
+ } });
1689
+ if (reason !== "compile-hung" || this.pending.size === 0) {
1690
+ this.recycle();
1691
+ return;
1692
+ }
1693
+ this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
1694
+ this.recycleBackstop.unref?.();
1695
+ }
1696
+ /**
1697
+ * Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
1698
+ * reported and rejects whatever it still held — a deliberate dispose would
1699
+ * silence both, and this is not one.
1700
+ */
1701
+ recycle() {
1702
+ if (this.recycling || this.exited || this.process === null) return;
1703
+ this.recycling = true;
1704
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1705
+ terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
1247
1706
  }
1248
1707
  rejectAll(err) {
1249
1708
  const entries = [...this.pending.values()];
@@ -1290,8 +1749,10 @@ var SharedInferencePool = class {
1290
1749
  this.tuning = options.tuning ?? null;
1291
1750
  this.numWorkers = Math.max(1, options.numWorkers ?? 1);
1292
1751
  this.device = options.device;
1752
+ this.onCompileFinishedLate = options.onCompileFinishedLate;
1293
1753
  }
1294
1754
  device;
1755
+ onCompileFinishedLate;
1295
1756
  /** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
1296
1757
  /** Summed backlog and shed count across the pool's workers. A rising
1297
1758
  * `inFlight` with a rising `shed` is a worker falling behind; a rising
@@ -1330,7 +1791,8 @@ var SharedInferencePool = class {
1330
1791
  tuning: this.tuning,
1331
1792
  logger: this.log,
1332
1793
  workerLabel: `w${i}`,
1333
- ...this.device ? { device: this.device } : {}
1794
+ ...this.device ? { device: this.device } : {},
1795
+ ...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
1334
1796
  }));
1335
1797
  const t0 = performance.now();
1336
1798
  const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
@@ -1425,7 +1887,7 @@ var SharedInferencePool = class {
1425
1887
  index,
1426
1888
  config: serializeModelConfig(config)
1427
1889
  })));
1428
- for (const resp of responses) if (resp.status !== "ok") throw new Error(`Failed to load model at index ${index}: ${resp.error ?? "unknown"}`);
1890
+ for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
1429
1891
  if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
1430
1892
  return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
1431
1893
  }
@@ -1461,7 +1923,7 @@ var SharedInferencePool = class {
1461
1923
  index,
1462
1924
  config: serializeModelConfig(config)
1463
1925
  })));
1464
- for (const resp of responses) if (resp.status !== "ok") throw new Error(`Failed to replace model at index ${index}: ${resp.error ?? "unknown"}`);
1926
+ for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
1465
1927
  return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
1466
1928
  }
1467
1929
  /**
@@ -1511,9 +1973,35 @@ var SharedInferencePool = class {
1511
1973
  allocateIndex() {
1512
1974
  return this.nextFreeIndex++;
1513
1975
  }
1976
+ /**
1977
+ * Give back an index whose load FAILED, so the retry reuses it. Only the most
1978
+ * recent allocation can be returned. Loads into one pool are serialised by
1979
+ * the provider, so a failed load is normally the latest one; anything else
1980
+ * is left allocated rather than risk handing out a live slot twice.
1981
+ */
1982
+ releaseIndex(index) {
1983
+ if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
1984
+ }
1514
1985
  isReady() {
1515
1986
  return this.workers.length > 0 && this.workers.every((w) => w.isReady());
1516
1987
  }
1988
+ /**
1989
+ * Why a worker of this pool was declared unusable while alive (D653), or
1990
+ * `null`. The provider charges the restart budget with THIS — or, for a
1991
+ * `compile-hung` death, does not charge the device at all — so the
1992
+ * `inference device FAILED` line names the cause instead of "pool worker is
1993
+ * not ready".
1994
+ */
1995
+ getDeathCause() {
1996
+ for (const w of this.workers) {
1997
+ const cause = w.getDeathCause();
1998
+ if (cause !== null) return {
1999
+ ...cause,
2000
+ message: `worker ${cause.message}`
2001
+ };
2002
+ }
2003
+ return null;
2004
+ }
1517
2005
  async dispose() {
1518
2006
  await Promise.all(this.workers.map((w) => w.dispose()));
1519
2007
  this.workers.length = 0;
@@ -1578,6 +2066,23 @@ var SharedInferencePool = class {
1578
2066
  return found;
1579
2067
  }
1580
2068
  };
2069
+ /** A failed command's reply as one message: its reason class first, when it has one. */
2070
+ function describeCommandFailure(resp) {
2071
+ const error = resp.error ?? "unknown";
2072
+ return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
2073
+ }
2074
+ /** Name a command for the lines that must say which one hung (D653). */
2075
+ function describeCommand(cmd) {
2076
+ const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
2077
+ const index = cmd["index"];
2078
+ const config = cmd["config"];
2079
+ const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
2080
+ return {
2081
+ cmd: name,
2082
+ modelIndex: typeof index === "number" ? index : null,
2083
+ model: typeof modelPath === "string" && modelPath.length > 0 ? node_path.basename(modelPath, node_path.extname(modelPath)) : null
2084
+ };
2085
+ }
1581
2086
  function serializeModelConfig(config) {
1582
2087
  const result = {
1583
2088
  path: config.path,
@@ -1865,6 +2370,28 @@ function poolDecodeSettingsEqual(a, b) {
1865
2370
  }
1866
2371
  //#endregion
1867
2372
  //#region src/detection-pipeline/engine/pipeline-model-manager.ts
2373
+ /**
2374
+ * A (step, model) load the pool rejected — carrying the pool's failure class
2375
+ * (`compile-timeout`, …) so the provider can count a timeout against THAT
2376
+ * model without parsing a message (D653).
2377
+ */
2378
+ var StepVariantLoadError = class extends Error {
2379
+ stepId;
2380
+ modelId;
2381
+ poolIndex;
2382
+ reason;
2383
+ /** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
2384
+ poolModel;
2385
+ constructor(stepId, modelId, poolIndex, cause) {
2386
+ super(cause instanceof Error ? cause.message : String(cause));
2387
+ this.name = "StepVariantLoadError";
2388
+ this.stepId = stepId;
2389
+ this.modelId = modelId;
2390
+ this.poolIndex = poolIndex;
2391
+ this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
2392
+ this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
2393
+ }
2394
+ };
1868
2395
  var PipelineModelManager = class {
1869
2396
  pool;
1870
2397
  source;
@@ -1876,11 +2403,20 @@ var PipelineModelManager = class {
1876
2403
  lruClock = 0;
1877
2404
  log;
1878
2405
  maxModelsPerStep;
2406
+ deviceKey;
2407
+ /**
2408
+ * Loads issued and not yet answered, keyed `stepId::modelId`. A second
2409
+ * caller for the same pair JOINS the pending load instead of issuing its own
2410
+ * (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
2411
+ * compile, each allocating a pool index of its own.
2412
+ */
2413
+ pendingLoads = /* @__PURE__ */ new Map();
1879
2414
  constructor(pool, source, logger, options) {
1880
2415
  this.pool = pool;
1881
2416
  this.source = source;
1882
2417
  this.log = logger;
1883
2418
  this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
2419
+ this.deviceKey = options?.deviceKey ?? null;
1884
2420
  }
1885
2421
  /**
1886
2422
  * Apply a new pipeline configuration — driven by the runtime config
@@ -2030,6 +2566,23 @@ var PipelineModelManager = class {
2030
2566
  await this.reconcileDecode(existing, settings);
2031
2567
  return existing;
2032
2568
  }
2569
+ const key = `${stepId}::${modelId}`;
2570
+ const pending = this.pendingLoads.get(key);
2571
+ if (pending) {
2572
+ const joined = await pending;
2573
+ await this.reconcileDecode(joined, settings);
2574
+ return joined;
2575
+ }
2576
+ const load = this.issueLoad(perStep, stepId, modelId, settings);
2577
+ this.pendingLoads.set(key, load);
2578
+ try {
2579
+ return await load;
2580
+ } finally {
2581
+ this.pendingLoads.delete(key);
2582
+ }
2583
+ }
2584
+ /** Evict if at capacity, then send ONE load command and record the result. */
2585
+ async issueLoad(perStep, stepId, modelId, settings) {
2033
2586
  while (perStep.size >= this.maxModelsPerStep) {
2034
2587
  const evicted = this.pickEvictionTarget(stepId);
2035
2588
  if (!evicted) break;
@@ -2041,16 +2594,31 @@ var PipelineModelManager = class {
2041
2594
  cap: this.maxModelsPerStep
2042
2595
  } });
2043
2596
  }
2044
- const index = this.pool.allocateIndex();
2045
2597
  const config = this.source.buildConfig(stepId, modelId, settings);
2046
2598
  const decode = poolDecodeSettingsOf(config);
2599
+ const index = this.pool.allocateIndex();
2047
2600
  this.log.info("Loading step variant", { meta: {
2048
2601
  step: stepId,
2049
2602
  modelId,
2050
2603
  poolIndex: index,
2604
+ deviceKey: this.deviceKey,
2051
2605
  ...decode
2052
2606
  } });
2053
- const { loadMs } = await this.pool.loadModel(index, config);
2607
+ let loadMs;
2608
+ try {
2609
+ ({loadMs} = await this.pool.loadModel(index, config));
2610
+ } catch (err) {
2611
+ this.pool.releaseIndex(index);
2612
+ this.log.error("Step variant load failed", { meta: {
2613
+ step: stepId,
2614
+ modelId,
2615
+ poolIndex: index,
2616
+ deviceKey: this.deviceKey,
2617
+ reason: err instanceof PoolModelLoadError ? err.reason : null,
2618
+ error: err instanceof Error ? err.message : String(err)
2619
+ } });
2620
+ throw new StepVariantLoadError(stepId, modelId, index, err);
2621
+ }
2054
2622
  this.log.info("Step variant loaded", { meta: {
2055
2623
  step: stepId,
2056
2624
  modelId,
@@ -2420,6 +2988,15 @@ var EngineFactory = class {
2420
2988
  isReady() {
2421
2989
  return this.pool?.isReady() ?? false;
2422
2990
  }
2991
+ /**
2992
+ * Why this factory's pool stopped being usable while its process was alive —
2993
+ * a compile that never returned, or a worker that answered nothing (D653) —
2994
+ * or `null`. A pool that merely crashed says nothing here; its `Worker
2995
+ * process exited` line already names the signal.
2996
+ */
2997
+ getDeathCause() {
2998
+ return this.pool?.getDeathCause() ?? null;
2999
+ }
2423
3000
  /** Native pid of the underlying Python pool, if any. */
2424
3001
  getPoolPid() {
2425
3002
  return this.pool?.getPid() ?? null;
@@ -2546,12 +3123,13 @@ var EngineFactory = class {
2546
3123
  concurrency,
2547
3124
  tuning: resolvedTuning,
2548
3125
  numWorkers,
2549
- ...this.opts.engine.device ? { device: this.opts.engine.device } : {}
3126
+ ...this.opts.engine.device ? { device: this.opts.engine.device } : {},
3127
+ ...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
2550
3128
  });
2551
3129
  this.poolManager = new PipelineModelManager(this.pool, {
2552
3130
  buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
2553
3131
  resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
2554
- }, this.log.child("model-mgr"));
3132
+ }, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
2555
3133
  await this.pool.initialize([]);
2556
3134
  await this.poolManager.applyConfig(steps);
2557
3135
  }
@@ -2667,6 +3245,175 @@ function buildPoolModelConfigForStep(inputs) {
2667
3245
  };
2668
3246
  }
2669
3247
  //#endregion
3248
+ //#region src/detection-pipeline/engine/model-load-governor.ts
3249
+ /**
3250
+ * What the dispatch path may ask of a pool's model loads, and what it must say
3251
+ * when it may not (D653, fix round 1).
3252
+ *
3253
+ * Three pieces of state, consulted ON DEMAND by the dispatch that needs a
3254
+ * model — there is no timer anywhere in here:
3255
+ *
3256
+ * 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
3257
+ * failed load, so without it every frame of every camera re-issued the
3258
+ * failing load and wrote two ERRORs. After a failure the model is not
3259
+ * re-issued on that pool for an interval that doubles per consecutive
3260
+ * failure up to a cap; a success clears it.
3261
+ * 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
3262
+ * bound is not a crash of the device. The SAME model timing out
3263
+ * {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
3264
+ * since a slow compile now runs on and writes its cache, only a compile
3265
+ * that never finishes can do — refuses THAT MODEL there, by name. The
3266
+ * device keeps serving every other model.
3267
+ * 3. Which camera has already been TOLD about the current state of a model,
3268
+ * so each camera gets one line per state change, never one per frame.
3269
+ */
3270
+ /** The first back-off after a failed load. */
3271
+ var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
3272
+ /** The back-off doubles per consecutive failure up to this. */
3273
+ var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
3274
+ /** Timeouts older than this no longer count toward a refusal. */
3275
+ var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
3276
+ var ModelLoadGovernor = class {
3277
+ /** Keyed by the pool OBJECT, so a respawned pool starts clean. */
3278
+ backoff = /* @__PURE__ */ new WeakMap();
3279
+ timeouts = /* @__PURE__ */ new Map();
3280
+ reported = /* @__PURE__ */ new Map();
3281
+ generation = 0;
3282
+ /** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
3283
+ check(pool, deviceKey, models, now = Date.now()) {
3284
+ for (const model of models) {
3285
+ const t = this.timeouts.get(timeoutKey(deviceKey, model));
3286
+ if (t?.refused === true) return {
3287
+ kind: "refused",
3288
+ model,
3289
+ timeouts: t.at.length,
3290
+ error: t.lastError
3291
+ };
3292
+ }
3293
+ const perPool = this.backoff.get(pool);
3294
+ if (perPool === void 0) return { kind: "go" };
3295
+ for (const model of models) {
3296
+ const b = perPool.get(model);
3297
+ if (b !== void 0 && now < b.retryAtMs) return {
3298
+ kind: "backoff",
3299
+ model,
3300
+ retryInMs: b.retryAtMs - now,
3301
+ failures: b.failures,
3302
+ error: b.error,
3303
+ generation: b.generation
3304
+ };
3305
+ }
3306
+ return { kind: "go" };
3307
+ }
3308
+ /** One load of `model` on `pool` failed. */
3309
+ recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
3310
+ let perPool = this.backoff.get(pool);
3311
+ if (perPool === void 0) {
3312
+ perPool = /* @__PURE__ */ new Map();
3313
+ this.backoff.set(pool, perPool);
3314
+ }
3315
+ const previousEntry = perPool.get(model);
3316
+ const failures = (previousEntry?.failures ?? 0) + 1;
3317
+ const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
3318
+ this.generation += 1;
3319
+ const generation = this.generation;
3320
+ perPool.set(model, {
3321
+ failures,
3322
+ retryAtMs: now + retryInMs,
3323
+ error,
3324
+ generation,
3325
+ poolModel: poolModel ?? previousEntry?.poolModel ?? null,
3326
+ reason
3327
+ });
3328
+ if (!compileTimeout) return {
3329
+ generation,
3330
+ retryInMs,
3331
+ refusedNow: false,
3332
+ compileTimeouts: 0
3333
+ };
3334
+ const key = timeoutKey(deviceKey, model);
3335
+ const previous = this.timeouts.get(key);
3336
+ const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
3337
+ const refused = at.length >= 2;
3338
+ this.timeouts.set(key, {
3339
+ at,
3340
+ refused,
3341
+ lastError: error
3342
+ });
3343
+ return {
3344
+ generation,
3345
+ retryInMs,
3346
+ refusedNow: refused && previous?.refused !== true,
3347
+ compileTimeouts: at.length
3348
+ };
3349
+ }
3350
+ /**
3351
+ * A compile that outlived its bound came back on `pool` (D653 round 3).
3352
+ *
3353
+ * `ok`: THAT model's cache is written and the worker loads again, so its
3354
+ * back-off is lifted — and only its: another model's failure on the same
3355
+ * pool is not the compile that just finished. Not ok: it is one more
3356
+ * failure of that model, and its back-off grows. Refusals by name are never
3357
+ * lifted here; a model refused for timing out twice stays refused until the
3358
+ * operator re-arms. Returns the governor keys it touched.
3359
+ */
3360
+ settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
3361
+ const perPool = this.backoff.get(pool);
3362
+ if (perPool === void 0) return [];
3363
+ const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
3364
+ for (const key of refusedMeanwhile) perPool.delete(key);
3365
+ const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
3366
+ for (const key of own) if (ok) perPool.delete(key);
3367
+ else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
3368
+ return [...refusedMeanwhile, ...own];
3369
+ }
3370
+ /** `models` loaded on `pool`: forget their failures there. */
3371
+ recordSuccess(pool, deviceKey, models) {
3372
+ const perPool = this.backoff.get(pool);
3373
+ for (const model of models) {
3374
+ perPool?.delete(model);
3375
+ this.timeouts.delete(timeoutKey(deviceKey, model));
3376
+ }
3377
+ }
3378
+ /**
3379
+ * Has `camera` been told about this `state` of `model` on `deviceKey` yet?
3380
+ * Returns `true` exactly once per state change.
3381
+ */
3382
+ shouldReport(camera, deviceKey, model, state) {
3383
+ const key = `${camera ?? "-"}|${deviceKey}|${model}`;
3384
+ if (this.reported.get(key) === state) return false;
3385
+ this.reported.set(key, state);
3386
+ return true;
3387
+ }
3388
+ /** Every model refused on some device. */
3389
+ refused() {
3390
+ const out = [];
3391
+ for (const [key, t] of this.timeouts) {
3392
+ if (!t.refused) continue;
3393
+ const [deviceKey = "", model = ""] = key.split("\0");
3394
+ out.push({
3395
+ deviceKey,
3396
+ model,
3397
+ timeouts: t.at.length,
3398
+ lastError: t.lastError
3399
+ });
3400
+ }
3401
+ return out;
3402
+ }
3403
+ /** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
3404
+ rearm(deviceKey) {
3405
+ let cleared = 0;
3406
+ for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
3407
+ this.timeouts.delete(key);
3408
+ cleared += 1;
3409
+ }
3410
+ return cleared;
3411
+ }
3412
+ };
3413
+ function timeoutKey(deviceKey, model) {
3414
+ return `${deviceKey}\u0000${model}`;
3415
+ }
3416
+ //#endregion
2670
3417
  //#region src/detection-pipeline/engine/idle-pool-reaper.ts
2671
3418
  var IdlePoolReaper = class {
2672
3419
  lastUsed = /* @__PURE__ */ new Map();
@@ -6563,6 +7310,34 @@ function parseWavToAudioChunk(filePath) {
6563
7310
  function enginesEqual(a, b) {
6564
7311
  return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
6565
7312
  }
7313
+ /** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
7314
+ var ModelLoadRefusedError = class extends Error {
7315
+ model;
7316
+ state;
7317
+ constructor(message, model, state) {
7318
+ super(message);
7319
+ this.name = "ModelLoadRefusedError";
7320
+ this.model = model;
7321
+ this.state = state;
7322
+ }
7323
+ };
7324
+ /** `step/model` — how the governor and the log lines name a model. */
7325
+ function stepModelKey(step) {
7326
+ return `${step.addonId}/${step.modelId}`;
7327
+ }
7328
+ /** The reporting state a refusal verdict stands for. */
7329
+ function stateOf(verdict) {
7330
+ return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
7331
+ }
7332
+ /**
7333
+ * The reason a condemned pool is charged against its device's restart budget.
7334
+ * A pool recycled because it HUNG (D653) names the hang — `inference device
7335
+ * FAILED … reason: pool worker is not ready` could not tell a GPU compile that
7336
+ * never returned from a crash.
7337
+ */
7338
+ function deathReasonOf(factory) {
7339
+ return factory.getDeathCause()?.message ?? "pool worker is not ready";
7340
+ }
6566
7341
  /** Build a `RuntimeEnv` from the running process + probed hardware. */
6567
7342
  function runtimeEnvFromProcess(hardware) {
6568
7343
  return {
@@ -8040,7 +8815,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8040
8815
  addonId: s.addonId,
8041
8816
  modelId: s.modelId
8042
8817
  });
8043
- await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format);
8818
+ await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
8819
+ ...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
8820
+ deviceKey: deviceKeyOf(dispatchEngine)
8821
+ });
8044
8822
  emit(`All models loaded`);
8045
8823
  }
8046
8824
  emit("Running inference...");
@@ -8337,53 +9115,193 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8337
9115
  throw new Error(message);
8338
9116
  }
8339
9117
  /**
8340
- * Single-flight gate around `ensureModelsForSteps`. Concurrent
8341
- * callers (every camera that fires motion in the same window calls
8342
- * `runPipeline` → `ensureModelsForSteps`) share the same in-flight
8343
- * promise instead of each racing into `loadAdditional`. Without this
8344
- * the pool gets the SAME model loaded N times under N distinct
8345
- * pool indices because every caller saw `isLoadedWithModel === false`
8346
- * before any of them committed via `stepToLoaded.set`.
8347
- */
8348
- modelLoadInFlight = null;
9118
+ * In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
9119
+ * needing the same models on the same pool share ONE load and its outcome;
9120
+ * a dispatch needing a different set is never handed another set's failure
9121
+ * — it queues behind the pool's current load ({@link modelLoadTail}) and
9122
+ * then decides for itself. Keyed by pool, not node-wide: one GPU compile
9123
+ * that never returned held EVERY load on the node — the NPU's and the CPU's
9124
+ * too — behind it.
9125
+ */
9126
+ modelLoadInFlight = /* @__PURE__ */ new Map();
9127
+ /** The last load issued on each pool — loads into one pool run one at a time. */
9128
+ modelLoadTail = /* @__PURE__ */ new Map();
9129
+ /** Stands in for "the node-default pool" before it exists. */
9130
+ nullPoolKey = {};
9131
+ /** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
9132
+ loadGovernor = new ModelLoadGovernor();
8349
9133
  /** Ensure all models needed by steps are downloaded and loaded in the engine pool.
8350
9134
  * `factory`/`format` default to the node's engine; a per-device dispatch passes
8351
- * that device's factory + format (Phase 2 multi-device). */
8352
- async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx") {
8353
- while (this.modelLoadInFlight) await this.modelLoadInFlight;
8354
- const allEnabled = flattenSteps(steps);
9135
+ * that device's factory + format (Phase 2 multi-device). `context` names the
9136
+ * camera and the accelerator on the line a rejected load writes. */
9137
+ async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
9138
+ const needed = this.stepsNeedingLoad(steps, factory);
9139
+ if (needed.length === 0) return;
9140
+ const pool = factory ?? this.nullPoolKey;
9141
+ const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
9142
+ const models = needed.map(stepModelKey);
9143
+ const refusal = this.loadRefusal(pool, deviceKey, models);
9144
+ if (refusal !== null) {
9145
+ this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
9146
+ throw refusal;
9147
+ }
9148
+ const setKey = [...models].sort().join(",");
9149
+ let perPool = this.modelLoadInFlight.get(pool);
9150
+ if (perPool === void 0) {
9151
+ perPool = /* @__PURE__ */ new Map();
9152
+ this.modelLoadInFlight.set(pool, perPool);
9153
+ }
9154
+ const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
9155
+ try {
9156
+ await flight.work;
9157
+ } catch (err) {
9158
+ if (err instanceof ModelLoadRefusedError) {
9159
+ this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
9160
+ throw err;
9161
+ }
9162
+ const failed = err instanceof StepVariantLoadError ? `${err.stepId}/${err.modelId}` : models[0] ?? "";
9163
+ this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, require_dist.errMsg(err));
9164
+ throw err;
9165
+ }
9166
+ }
9167
+ /**
9168
+ * Why `models` may not be loaded on `pool` right now — a back-off after a
9169
+ * failure, or a refusal by name after repeated compile timeouts — or `null`.
9170
+ */
9171
+ loadRefusal(pool, deviceKey, models) {
9172
+ const verdict = this.loadGovernor.check(pool, deviceKey, models);
9173
+ if (verdict.kind === "go") return null;
9174
+ return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
9175
+ }
9176
+ /**
9177
+ * A compile that outlived its soft bound came back on `factory`'s pool
9178
+ * (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
9179
+ * again, so that model's back-off is lifted at once rather than waited out;
9180
+ * no other model's is. A FAILURE is one more failure of that model, and its
9181
+ * back-off grows.
9182
+ */
9183
+ noteLateCompile(factory, deviceKey, event) {
9184
+ const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
9185
+ if (event.ok) {
9186
+ this.log.info("model loadable again after late compile", { meta: {
9187
+ deviceKey,
9188
+ model: event.model,
9189
+ backoffLifted: models
9190
+ } });
9191
+ return;
9192
+ }
9193
+ this.log.warn("late compile failed — the model stays backed off", { meta: {
9194
+ deviceKey,
9195
+ model: event.model,
9196
+ backoffExtended: models,
9197
+ error: event.error ?? null
9198
+ } });
9199
+ }
9200
+ /** The enabled steps the factory's pool does not reflect yet. */
9201
+ stepsNeedingLoad(steps, factory) {
8355
9202
  const needed = [];
8356
- for (const step of allEnabled) {
9203
+ for (const step of flattenSteps(steps)) {
8357
9204
  if (!step.enabled) continue;
8358
9205
  if (factory !== null && !factory.needsPoolUpdate(step)) continue;
8359
9206
  needed.push(step);
8360
9207
  }
8361
- if (needed.length === 0) return;
8362
- const work = (async () => {
8363
- for (const step of needed) {
8364
- const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
8365
- if (modelEntry && !(0, _camstack_system_addon_utils.isModelDownloaded)(this.modelsDir, modelEntry, format)) {
8366
- if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
8367
- if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
8368
- this.log.info("Downloading model for step", { meta: {
8369
- modelId: step.modelId,
8370
- format,
8371
- step: step.addonId
8372
- } });
8373
- await this.downloadWithRetry(modelEntry, format, 3);
8374
- }
9208
+ return needed;
9209
+ }
9210
+ /**
9211
+ * Issue ONE load for a model set on a pool, queued behind the pool's
9212
+ * previous load, and record its outcome with the governor. The set is
9213
+ * re-derived once the queue reaches it: the load ahead may have brought
9214
+ * some of it in.
9215
+ */
9216
+ issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
9217
+ const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
9218
+ const flight = {
9219
+ work: Promise.resolve(),
9220
+ generation: null
9221
+ };
9222
+ flight.work = (async () => {
9223
+ await previous;
9224
+ const needed = this.stepsNeedingLoad(steps, factory);
9225
+ if (needed.length === 0) return;
9226
+ const models = needed.map(stepModelKey);
9227
+ const refusal = this.loadRefusal(pool, deviceKey, models);
9228
+ if (refusal !== null) throw refusal;
9229
+ try {
9230
+ await this.downloadNeededModels(needed, format);
9231
+ this.log.info("Loading additional models for benchmark", { meta: {
9232
+ count: needed.length,
9233
+ models,
9234
+ deviceKey
9235
+ } });
9236
+ await factory.loadAdditional(needed);
9237
+ } catch (err) {
9238
+ flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
9239
+ throw err;
8375
9240
  }
8376
- this.log.info("Loading additional models for benchmark", { meta: {
8377
- count: needed.length,
8378
- models: needed.map((s) => `${s.addonId}/${s.modelId}`)
8379
- } });
8380
- await factory.loadAdditional(needed);
9241
+ this.loadGovernor.recordSuccess(pool, deviceKey, models);
8381
9242
  })();
8382
- this.modelLoadInFlight = work;
8383
- try {
8384
- await work;
8385
- } finally {
8386
- if (this.modelLoadInFlight === work) this.modelLoadInFlight = null;
9243
+ perPool.set(setKey, flight);
9244
+ const tail = flight.work.catch(() => void 0);
9245
+ this.modelLoadTail.set(pool, tail);
9246
+ tail.then(() => {
9247
+ if (perPool.get(setKey) === flight) perPool.delete(setKey);
9248
+ if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
9249
+ });
9250
+ return flight;
9251
+ }
9252
+ /** Charge a failed load to the model that failed; say so once if it is now refused. */
9253
+ recordModelLoadFailure(pool, deviceKey, models, err) {
9254
+ const failedModels = err instanceof StepVariantLoadError ? [`${err.stepId}/${err.modelId}`] : models;
9255
+ const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
9256
+ const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
9257
+ let generation = 0;
9258
+ for (const model of failedModels) {
9259
+ const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, require_dist.errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
9260
+ generation = outcome.generation;
9261
+ if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
9262
+ deviceKey,
9263
+ model,
9264
+ compileTimeouts: outcome.compileTimeouts,
9265
+ error: require_dist.errMsg(err),
9266
+ device: "keeps serving every other model",
9267
+ rearm: "pipelineExecutor.rearmInferenceDevice"
9268
+ } });
9269
+ }
9270
+ return generation;
9271
+ }
9272
+ /**
9273
+ * One WARN per camera per state change of the model that cost it its frame.
9274
+ * The ERROR for the failed load itself is written once, where the load
9275
+ * failed (`Step variant load failed`); this line answers "which cameras
9276
+ * did it cost?", tagged so it can be counted per camera.
9277
+ */
9278
+ reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
9279
+ if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
9280
+ this.log.warn("model load for dispatch failed", {
9281
+ ...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
9282
+ meta: {
9283
+ deviceKey,
9284
+ models,
9285
+ model,
9286
+ state,
9287
+ error
9288
+ }
9289
+ });
9290
+ }
9291
+ /** Download any model of `needed` missing in `format` (the device's). */
9292
+ async downloadNeededModels(needed, format) {
9293
+ for (const step of needed) {
9294
+ const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
9295
+ if (modelEntry && !(0, _camstack_system_addon_utils.isModelDownloaded)(this.modelsDir, modelEntry, format)) {
9296
+ if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
9297
+ if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
9298
+ this.log.info("Downloading model for step", { meta: {
9299
+ modelId: step.modelId,
9300
+ format,
9301
+ step: step.addonId
9302
+ } });
9303
+ await this.downloadWithRetry(modelEntry, format, 3);
9304
+ }
8387
9305
  }
8388
9306
  }
8389
9307
  /** Download a model with retry + exponential backoff */
@@ -8511,7 +9429,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8511
9429
  await this.ensureEngineFactory();
8512
9430
  const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
8513
9431
  if (!factory) throw new Error("runPipelineBatch: factory not initialised");
8514
- if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat);
9432
+ if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
8515
9433
  const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
8516
9434
  this.log.info("runPipelineBatch path decision", { meta: {
8517
9435
  phase: "batch",
@@ -8766,7 +9684,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8766
9684
  if (this.engineFactory.isReady()) return;
8767
9685
  const dead = this.engineFactory;
8768
9686
  this.engineFactory = null;
8769
- this.noteDeviceDeath(defaultDeviceKey, "pool worker is not ready");
9687
+ this.noteFactoryDeath(defaultDeviceKey, dead);
8770
9688
  await dead.dispose().catch(() => void 0);
8771
9689
  this.refuseIfDeviceUnusable(defaultDeviceKey);
8772
9690
  }
@@ -8784,7 +9702,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8784
9702
  logger: this.log.child("engine"),
8785
9703
  pythonPath: this.executorOptions.pythonPath ?? "",
8786
9704
  provisioning: this.executorOptions.provisioning,
8787
- resolveCustomModel: this.customModelResolver
9705
+ resolveCustomModel: this.customModelResolver,
9706
+ onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
8788
9707
  });
8789
9708
  try {
8790
9709
  await factory.initialize([]);
@@ -8904,15 +9823,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8904
9823
  * dispose — which rejects whatever was still in flight on it with a reason
8905
9824
  * rather than letting those requests sit until their own deadlines.
8906
9825
  */
8907
- async condemnDeviceFactory(deviceKey, factory, reason) {
9826
+ async condemnDeviceFactory(deviceKey, factory) {
8908
9827
  if (this.factoriesByDevice.get(deviceKey) === factory) {
8909
9828
  this.factoriesByDevice.delete(deviceKey);
8910
9829
  this.deviceReaper.cancel(deviceKey);
8911
- this.noteDeviceDeath(deviceKey, reason);
9830
+ this.noteFactoryDeath(deviceKey, factory);
8912
9831
  }
8913
9832
  await factory.dispose().catch(() => void 0);
8914
9833
  }
8915
9834
  /**
9835
+ * Charge one dead pool to its device — unless it died of a compile that was
9836
+ * still running at its hard bound (D653, fix round 1). That death is not a
9837
+ * crash of the DEVICE: the model that hung was already counted when its
9838
+ * load answered `compile-timeout`, and the same model timing out again on
9839
+ * the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
9840
+ * the respawns. A crash stays a crash.
9841
+ */
9842
+ noteFactoryDeath(deviceKey, factory) {
9843
+ const cause = factory.getDeathCause();
9844
+ if (cause !== null && !cause.chargesDeviceBudget) {
9845
+ this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
9846
+ deviceKey,
9847
+ reason: cause.reason,
9848
+ cause: cause.message
9849
+ } });
9850
+ return;
9851
+ }
9852
+ this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
9853
+ }
9854
+ /**
8916
9855
  * Every inference device this node currently refuses, and why.
8917
9856
  *
8918
9857
  * The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
@@ -8947,7 +9886,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8947
9886
  state: "backoff",
8948
9887
  since: Date.now(),
8949
9888
  deaths: 0,
8950
- lastError: "pool worker is not ready"
9889
+ lastError: deathReasonOf(factory)
8951
9890
  });
8952
9891
  }
8953
9892
  return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
@@ -8964,10 +9903,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8964
9903
  * no-op, reported as such.
8965
9904
  */
8966
9905
  async rearmInferenceDevice(input) {
8967
- const rearmed = this.deviceLiveness.rearm(input.deviceKey);
9906
+ const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
9907
+ const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
9908
+ const rearmed = rearmedDevice || rearmedModels > 0;
8968
9909
  this.log.info("inference device re-armed by operator", { meta: {
8969
9910
  deviceKey: input.deviceKey,
8970
- rearmed
9911
+ rearmed,
9912
+ rearmedModels
8971
9913
  } });
8972
9914
  return { rearmed };
8973
9915
  }
@@ -8984,7 +9926,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8984
9926
  this.deviceReaper.touch(deviceKey);
8985
9927
  return existing;
8986
9928
  }
8987
- await this.condemnDeviceFactory(deviceKey, existing, "pool worker is not ready");
9929
+ await this.condemnDeviceFactory(deviceKey, existing);
8988
9930
  this.refuseIfDeviceUnusable(deviceKey);
8989
9931
  }
8990
9932
  const inflight = this.deviceFactoryInflight.get(deviceKey);
@@ -8997,7 +9939,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8997
9939
  logger: this.log.child(`engine:${deviceKey}`),
8998
9940
  pythonPath: this.executorOptions.pythonPath ?? "",
8999
9941
  provisioning: this.executorOptions.provisioning,
9000
- resolveCustomModel: this.customModelResolver
9942
+ resolveCustomModel: this.customModelResolver,
9943
+ onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
9001
9944
  });
9002
9945
  try {
9003
9946
  await factory.initialize([]);