@camstack/addon-pipeline 1.2.295 → 1.2.296

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/THIRD_PARTY_MODELS.md +8 -0
  2. package/dist/audio-analyzer/index.js +2 -2
  3. package/dist/audio-analyzer/index.mjs +2 -2
  4. package/dist/{default-detection-model-CAUBVbgK.mjs → default-detection-model-Co578D8C.mjs} +75 -46
  5. package/dist/{default-detection-model-wNEISVA9.js → default-detection-model-D1daTtqT.js} +75 -46
  6. package/dist/detection-pipeline/index.js +1023 -80
  7. package/dist/detection-pipeline/index.mjs +1023 -80
  8. package/dist/{dist-BsVcf5wO.js → dist-8up-f2TX.js} +877 -365
  9. package/dist/{dist-CuxSNLKW.mjs → dist-CCd0Q3nr.mjs} +872 -366
  10. package/dist/motion-wasm/index.js +1 -1
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node-Cqb1QiXk.mjs → node-DgMSXSWP.mjs} +1 -1
  13. package/dist/{node-UFk6I2f6.js → node-lpQgHes9.js} +1 -1
  14. package/dist/pipeline-runner/index.js +789 -203
  15. package/dist/pipeline-runner/index.mjs +789 -203
  16. package/dist/{process-memory-CJV29sPf.js → process-memory-CX_92V_r.js} +1 -1
  17. package/dist/{process-memory-D0Nvs9rr.mjs → process-memory-DFC_O5zE.mjs} +1 -1
  18. package/dist/recorder/index.js +4 -6
  19. package/dist/recorder/index.mjs +4 -6
  20. package/dist/{segment-demux-js-DGwmf5hu.js → segment-demux-js-DzBx6NN2.js} +1 -1
  21. package/dist/{segment-demux-js-DIDWw1iE.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
  22. package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
  23. package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
  24. package/dist/stream-broker/_stub.js +2 -2
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-B7nBFqva.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
  27. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
  28. package/dist/stream-broker/demux-worker-child.js +1 -1
  29. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  30. package/dist/stream-broker/{hostInit-BKlb3qac.mjs → hostInit-BPtppL3W.mjs} +2 -2
  31. package/dist/stream-broker/index.js +4 -4
  32. package/dist/stream-broker/index.mjs +4 -4
  33. package/dist/stream-broker/remoteEntry.js +1 -1
  34. package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
  35. package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
  36. package/package.json +1 -1
  37. package/python/inference_pool.py +422 -64
  38. package/python/test_inference_pool_compile_off_loop.py +414 -0
  39. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CCIyBvRa.mjs +0 -26
  40. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DeD_UFdb.mjs +0 -26
@@ -1,9 +1,9 @@
1
1
  import { n as __require } from "../chunk-DnnnRqeS.mjs";
2
- import { At as pickClusterStepSettings, Bt as resolveClusterStepModelId, Dn as array, Dt as overlayClusterStepSettings, H as YAMNET_TO_MACRO, In as string, Kt as runtimeDevices$1, Ln as union, Nn as object, Nt as pipelineExecutorCapability, Rn as EventCategory, Sn as parseJsonUnknown, Tn as sleep, Wt as resolvePoolMemoryPolicy, Zt as supportedRuntimes$1, _n as hydrateSchema, en as errMsg, et as defaultDeviceFor$1, gt as inferModelProvider, it as detectionPipelineCapability, j as PoolMemoryWatchdog, kt as pickClusterStepModels, lt as evaluateZoneRules, m as DEFAULT_CLUSTER_STEP_SETTINGS, p as DEFAULT_CLUSTER_STEP_MODELS, pn as createEvent, sn as BaseAddon, st as enumerateInferenceDevices, t as APPLE_SA_TO_MACRO, vt as loadContributionCapability, xn as nodePin, y as DEVICE_BACKEND_TO_FORMAT } from "../dist-CuxSNLKW.mjs";
3
- import { s as readProcessCost } from "../node-Cqb1QiXk.mjs";
4
- import { a as localFrameRegistry, c as ALL_STEPS, d as getStepDefinition, f as resolveModelForFormat, l as getDefaultModelForFormat, m as landmarkPrecisionVerdict, n as resolveDefaultDetectionModel, r as macroGateVerdict, s as ALL_PIPELINE_STEPS, u as getStep } from "../default-detection-model-CAUBVbgK.mjs";
2
+ import { At as pickClusterStepSettings, Cn as parseJsonUnknown, Dt as overlayClusterStepSettings, En as sleep, Gt as resolvePoolMemoryPolicy, H as YAMNET_TO_MACRO, Ln as string, Mt as pickInvalidClusterStepModels, On as array, Pn as object, Pt as pipelineExecutorCapability, Qt as supportedRuntimes$1, Rn as union, Sn as nodePin, Vt as resolveClusterStepModelId, cn as BaseAddon, et as defaultDeviceFor$1, gt as inferModelProvider, it as detectionPipelineCapability, j as PoolMemoryWatchdog, kt as pickClusterStepModels, lt as evaluateZoneRules, m as DEFAULT_CLUSTER_STEP_SETTINGS, mn as createEvent, p as DEFAULT_CLUSTER_STEP_MODELS, qt as runtimeDevices$1, st as enumerateInferenceDevices, t as APPLE_SA_TO_MACRO, tn as errMsg, vn as hydrateSchema, vt as loadContributionCapability, y as DEVICE_BACKEND_TO_FORMAT, zn as EventCategory } from "../dist-CCd0Q3nr.mjs";
3
+ import { s as readProcessCost } from "../node-DgMSXSWP.mjs";
4
+ import { a as localFrameRegistry, c as ALL_STEPS, d as getStepDefinition, f as resolveModelForFormat, l as getDefaultModelForFormat, m as landmarkPrecisionVerdict, n as resolveDefaultDetectionModel, r as macroGateVerdict, s as ALL_PIPELINE_STEPS, u as getStep } from "../default-detection-model-Co578D8C.mjs";
5
5
  import { t as getSharp } from "../lazy-sharp-DXsqwpph.mjs";
6
- import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-D0Nvs9rr.mjs";
6
+ import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-DFC_O5zE.mjs";
7
7
  import * as os from "node:os";
8
8
  import * as fs from "node:fs";
9
9
  import * as path$1 from "node:path";
@@ -415,6 +415,12 @@ var ClusterModelSource = class {
415
415
  const next = pickClusterStepModels(view);
416
416
  const nextSettings = pickClusterStepSettings(view);
417
417
  this.lastReadAtMs = this.now();
418
+ const invalid = pickInvalidClusterStepModels(view);
419
+ if (invalid.length > 0) this.logger.warn("cluster step model row holds a value that is not a model id — running the catalog default for that step; no re-embed pass will run on it", { meta: {
420
+ owner: CLUSTER_MODEL_OWNER_ADDON_ID,
421
+ invalid,
422
+ inUse: Object.fromEntries(invalid.map((e) => [e.stepId, next[e.stepId]]))
423
+ } });
418
424
  const changed = Object.keys(next).filter((stepId) => next[stepId] !== this.models[stepId]);
419
425
  const settingsChanged = Object.keys(nextSettings).some((stepId) => {
420
426
  const from = this.settings[stepId] ?? {};
@@ -562,6 +568,98 @@ var DeviceOverrideMirror = class {
562
568
  }
563
569
  };
564
570
  //#endregion
571
+ //#region src/detection-pipeline/engine/pool-worker-health.ts
572
+ /**
573
+ * The one table of load bounds (D653, fix round 1).
574
+ *
575
+ * The soft bound decides how soon the operator HEARS about a slow compile; the
576
+ * hard bound decides when a compile is given up for hung. A single 120 s bound
577
+ * that also killed the compile was wrong twice: a slow but finite compile was
578
+ * killed before it could write its cache, so every respawn faced the same cold
579
+ * compile and three of them walked the device to `failed`; and on CoreML the
580
+ * bound was shorter than the cross-process compile-cache lock wait alone.
581
+ */
582
+ var POOL_MODEL_LOAD_BOUNDS = {
583
+ openvino: {
584
+ softMs: 12e4,
585
+ hardMs: 6e5,
586
+ why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
587
+ gilMayBeHeldDuringCompile: false,
588
+ gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
589
+ },
590
+ coreml: {
591
+ softMs: 3e5,
592
+ hardMs: 9e5,
593
+ why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
594
+ gilMayBeHeldDuringCompile: true,
595
+ gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
596
+ },
597
+ onnxruntime: {
598
+ softMs: 12e4,
599
+ hardMs: 6e5,
600
+ why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
601
+ gilMayBeHeldDuringCompile: false,
602
+ gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
603
+ },
604
+ edgetpu: {
605
+ softMs: 12e4,
606
+ hardMs: 6e5,
607
+ why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
608
+ gilMayBeHeldDuringCompile: false,
609
+ gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
610
+ }
611
+ };
612
+ /**
613
+ * The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
614
+ * so the worker's NAMED answer always lands first.
615
+ */
616
+ var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
617
+ function chargesDeviceBudget(reason) {
618
+ return reason !== "compile-hung";
619
+ }
620
+ /**
621
+ * Tracks live inference outcomes on one worker and says when its executor has
622
+ * produced nothing for too long. Consulted on demand — no timer.
623
+ *
624
+ * A SHED (`dropped`) does not count as a result: the Python loop answers sheds
625
+ * itself, so a worker whose executor threads are hung keeps shedding happily.
626
+ */
627
+ var InferStallDetector = class {
628
+ unanswered = 0;
629
+ lastResultAt;
630
+ /**
631
+ * When the current streak of unanswered requests began — the first one after
632
+ * a result. IDLE time is not silence: quiet cameras send nothing, and a
633
+ * worker that was asked nothing has failed nothing. Measuring from the last
634
+ * result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
635
+ * cameras) poison a healthy worker and charge the device.
636
+ */
637
+ streakStartedAt = null;
638
+ constructor(now) {
639
+ this.lastResultAt = now;
640
+ }
641
+ /** A reply that carried a real result (or a real error) from the executor. */
642
+ noteResult(now) {
643
+ this.unanswered = 0;
644
+ this.lastResultAt = now;
645
+ this.streakStartedAt = null;
646
+ }
647
+ /**
648
+ * A live request that ended without a result (deadline expiry or a shed).
649
+ * Returns the stall when the worker has crossed both thresholds.
650
+ */
651
+ noteUnanswered(now) {
652
+ this.unanswered += 1;
653
+ if (this.streakStartedAt === null) this.streakStartedAt = now;
654
+ const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
655
+ if (this.unanswered >= 20 && silentMs >= 3e4) return {
656
+ unanswered: this.unanswered,
657
+ silentMs
658
+ };
659
+ return null;
660
+ }
661
+ };
662
+ //#endregion
565
663
  //#region src/detection-pipeline/engine/shared-inference-pool.ts
566
664
  /**
567
665
  * SharedInferencePool — TypeScript wrapper for inference_pool.py.
@@ -603,6 +701,27 @@ var RAW_FMT_CODE = {
603
701
  gray: 2
604
702
  };
605
703
  /**
704
+ * A load the worker answered with a machine-readable failure class. The
705
+ * provider tells a `compile-timeout` (counted against THAT MODEL on the
706
+ * device) from an ordinary failure by `reason`, never by parsing text.
707
+ */
708
+ var PoolModelLoadError = class extends Error {
709
+ reason;
710
+ /** The model's file stem, as the worker named it. */
711
+ model;
712
+ constructor(message, reason, model) {
713
+ super(message);
714
+ this.name = "PoolModelLoadError";
715
+ this.reason = reason;
716
+ this.model = model;
717
+ }
718
+ };
719
+ /**
720
+ * Request id the worker uses for UNSOLICITED events — a reply to no request
721
+ * (`compile-finished-late`). Never allocated to a request.
722
+ */
723
+ var WORKER_EVENT_REQ_ID = 4294967295;
724
+ /**
606
725
  * Per-inference-request reply timeout (ms). Turns a wedged request (worker
607
726
  * alive but no reply) into a rejection so every caller settles — runtime frame
608
727
  * dispatch drops the frame; the benchmark full-tree run rejects the wedged
@@ -611,6 +730,22 @@ var RAW_FMT_CODE = {
611
730
  * large frame) never trips; override via env for constrained hardware.
612
731
  */
613
732
  var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
733
+ /** Commands that compile a model — the only ones that get the load deadline. */
734
+ var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
735
+ /**
736
+ * Deadline of the liveness probe (`mem_stats`) sent after a command times out.
737
+ * The worker answers `mem_stats` on its event loop, never behind a compile
738
+ * (D653), so a healthy worker replies in milliseconds even mid-compile; ten
739
+ * seconds only has to outlast a GC pause or a burst of inference replies.
740
+ */
741
+ var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
742
+ /**
743
+ * How long a poisoned worker may keep its in-flight LIVE requests before it is
744
+ * killed anyway. A poisoned worker takes no new work, and every live request
745
+ * carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
746
+ * by then; the second of slack keeps the backstop from racing the last reply.
747
+ */
748
+ var POOL_POISON_DRAIN_SLACK_MS = 1e3;
614
749
  /**
615
750
  * Max inference requests outstanding to ONE worker before new ones are SHED.
616
751
  *
@@ -798,6 +933,27 @@ var PoolWorker = class {
798
933
  wireVersion = 1;
799
934
  nextRequestId = 1;
800
935
  ready = false;
936
+ /** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
937
+ * a poisoned worker is recycled, not revived. */
938
+ poisonVerdict = null;
939
+ /** The kill of a poisoned worker has been started. */
940
+ recycling = false;
941
+ /** Kills a poisoned worker whose drain did not finish in time. */
942
+ recycleBackstop = null;
943
+ /** A liveness probe is in flight — one at a time, never a probe storm. */
944
+ probeInFlight = false;
945
+ /** The child has exited (any cause). */
946
+ exited = false;
947
+ /** A compile past its soft bound, still running in the worker (D653). */
948
+ abandonedCompile = null;
949
+ /** Says when live inference has produced nothing for too long (D653). */
950
+ inferStall = new InferStallDetector(Date.now());
951
+ /**
952
+ * A load's HOST deadline expired with no reply since the last reply of any
953
+ * kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
954
+ * is over for this worker until something — anything — comes back.
955
+ */
956
+ loadDeadlineMissed = false;
801
957
  log;
802
958
  opts;
803
959
  constructor(opts) {
@@ -822,6 +978,24 @@ var PoolWorker = class {
822
978
  isReady() {
823
979
  return this.ready;
824
980
  }
981
+ /** Why this worker was recycled, typed for the provider's budget, or `null`. */
982
+ getDeathCause() {
983
+ const v = this.poisonVerdict;
984
+ const message = this.getPoisonDescription();
985
+ if (v === null || message === null) return null;
986
+ return {
987
+ reason: v.reason,
988
+ chargesDeviceBudget: chargesDeviceBudget(v.reason),
989
+ message
990
+ };
991
+ }
992
+ /** Why this live worker was declared unusable, or `null` (D653). */
993
+ getPoisonDescription() {
994
+ const v = this.poisonVerdict;
995
+ if (v === null) return null;
996
+ const target = v.command.model !== null ? ` of ${v.command.model}` : "";
997
+ return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
998
+ }
825
999
  async initialize(initialModels) {
826
1000
  this.process = spawn(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
827
1001
  "pipe",
@@ -845,7 +1019,11 @@ var PoolWorker = class {
845
1019
  const spawnedProcess = this.process;
846
1020
  spawnedProcess.on("exit", (code, signal) => {
847
1021
  this.ready = false;
1022
+ this.exited = true;
1023
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1024
+ if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
848
1025
  if (this.process !== spawnedProcess) return;
1026
+ const poisonedBy = this.getPoisonDescription();
849
1027
  this.log.error("Worker process exited", { meta: {
850
1028
  worker: this.opts.workerLabel,
851
1029
  pid: spawnedProcess.pid ?? null,
@@ -853,9 +1031,10 @@ var PoolWorker = class {
853
1031
  device: this.opts.device ?? "default",
854
1032
  code,
855
1033
  signal,
856
- inFlight: this.pending.size
1034
+ inFlight: this.pending.size,
1035
+ ...poisonedBy !== null ? { poisonedBy } : {}
857
1036
  } });
858
- this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
1037
+ this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
859
1038
  });
860
1039
  this.process.stdout.on("data", (chunk) => {
861
1040
  const t0 = Date.now();
@@ -869,6 +1048,7 @@ var PoolWorker = class {
869
1048
  runtime: this.opts.poolRuntime,
870
1049
  concurrency: this.opts.concurrency,
871
1050
  protocolVersion: 2,
1051
+ modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
872
1052
  models: initialModels.map((m) => serializeModelConfig(m))
873
1053
  };
874
1054
  if (this.opts.device) config["device"] = this.opts.device;
@@ -891,6 +1071,7 @@ var PoolWorker = class {
891
1071
  clearTimeout(timeout);
892
1072
  if (result["status"] === "ready") {
893
1073
  this.ready = true;
1074
+ this.inferStall.noteResult(Date.now());
894
1075
  const loadedCount = result["models"];
895
1076
  const startupMs = result["startupMs"];
896
1077
  const workers = result["workers"] ?? 1;
@@ -913,7 +1094,7 @@ var PoolWorker = class {
913
1094
  async infer(modelByte, jpeg, deviceId) {
914
1095
  this.ensureReady();
915
1096
  const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
916
- return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
1097
+ return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
917
1098
  }
918
1099
  async inferRaw(modelByte, raw, width, height, format, deviceId) {
919
1100
  this.ensureReady();
@@ -966,12 +1147,18 @@ var PoolWorker = class {
966
1147
  const payload = Buffer.allocUnsafe(5);
967
1148
  payload[0] = modelByte;
968
1149
  payload.writeUInt32LE(frameId, 1);
969
- return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
1150
+ return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
970
1151
  }
971
1152
  async sendCommand(cmd) {
972
1153
  this.ensureReady();
1154
+ const command = describeCommand(cmd);
973
1155
  const payload = Buffer.from(JSON.stringify(cmd), "utf8");
974
- return await this.dispatch(MSG_COMMAND, payload);
1156
+ const sentAt = Date.now();
1157
+ if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
1158
+ const raw = await this.dispatch(MSG_COMMAND, payload, { command });
1159
+ if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
1160
+ this.inspectCommandReply(raw, command, sentAt);
1161
+ return raw;
975
1162
  }
976
1163
  /** Command whose reply shape is NOT the load/unload/replace/status envelope
977
1164
  * (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
@@ -979,13 +1166,15 @@ var PoolWorker = class {
979
1166
  async sendRawCommand(cmd) {
980
1167
  this.ensureReady();
981
1168
  const payload = Buffer.from(JSON.stringify(cmd), "utf8");
982
- return this.dispatch(MSG_COMMAND, payload);
1169
+ return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
983
1170
  }
984
1171
  async dispose() {
985
1172
  const proc = this.process;
986
1173
  if (!proc) return;
987
1174
  this.process = null;
988
1175
  this.ready = false;
1176
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1177
+ if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
989
1178
  await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
990
1179
  this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
991
1180
  }
@@ -1028,8 +1217,11 @@ var PoolWorker = class {
1028
1217
  }
1029
1218
  /** Live inference gets seconds; commands and model loads keep the long
1030
1219
  * timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
1031
- deadlineFor(msgType) {
1032
- return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
1220
+ deadlineFor(msgType, opts = {}) {
1221
+ if (opts.timeoutMs !== void 0) return opts.timeoutMs;
1222
+ if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
1223
+ if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
1224
+ return POOL_INFER_TIMEOUT_MS;
1033
1225
  }
1034
1226
  /**
1035
1227
  * The request's ABSOLUTE deadline for the wire (D350), or `null` when this
@@ -1046,16 +1238,18 @@ var PoolWorker = class {
1046
1238
  stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
1047
1239
  return stamp;
1048
1240
  }
1049
- dispatch(msgType, payload, deviceId) {
1050
- const shed = this.shedIfSaturated(msgType, deviceId);
1241
+ dispatch(msgType, payload, opts = {}) {
1242
+ const shed = this.shedIfSaturated(msgType, opts.deviceId);
1051
1243
  if (shed) return Promise.resolve(shed);
1052
1244
  const reqId = this.allocRequestId();
1053
1245
  return new Promise((resolve, reject) => {
1054
- const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1246
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
1055
1247
  this.pending.set(reqId, {
1056
1248
  resolve,
1057
1249
  reject,
1058
- timer
1250
+ timer,
1251
+ ...opts.command !== void 0 ? { command: opts.command } : {},
1252
+ ...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
1059
1253
  });
1060
1254
  try {
1061
1255
  const stamp = this.deadlineStamp(msgType);
@@ -1086,8 +1280,9 @@ var PoolWorker = class {
1086
1280
  * - Anything else (commands, model loads) still REJECTS with the error the
1087
1281
  * dashboards grep for — a lost command is a fault, not flow control.
1088
1282
  */
1089
- armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
1090
- const timeoutMs = this.deadlineFor(msgType);
1283
+ armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
1284
+ const { deviceId, command } = opts;
1285
+ const timeoutMs = this.deadlineFor(msgType, opts);
1091
1286
  const timer = setTimeout(() => {
1092
1287
  if (this.pending.delete(reqId)) {
1093
1288
  if (SHEDDABLE_MSG_TYPES.has(msgType)) {
@@ -1111,6 +1306,7 @@ var PoolWorker = class {
1111
1306
  dropped: true,
1112
1307
  shedReason: "deadline-expired"
1113
1308
  });
1309
+ this.noteInferUnanswered();
1114
1310
  return;
1115
1311
  }
1116
1312
  this.timedOutCount++;
@@ -1123,10 +1319,20 @@ var PoolWorker = class {
1123
1319
  device: this.opts.device ?? "default",
1124
1320
  inFlight: this.pending.size,
1125
1321
  reqId,
1126
- timeoutMs
1322
+ timeoutMs,
1323
+ ...command !== void 0 ? {
1324
+ command: command.cmd,
1325
+ modelIndex: command.modelIndex,
1326
+ model: command.model
1327
+ } : {}
1127
1328
  }
1128
1329
  });
1129
- reject(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`));
1330
+ const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
1331
+ if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
1332
+ this.loadDeadlineMissed = true;
1333
+ reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
1334
+ } else reject(new Error(message));
1335
+ if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
1130
1336
  }
1131
1337
  }, timeoutMs);
1132
1338
  timer.unref?.();
@@ -1137,11 +1343,12 @@ var PoolWorker = class {
1137
1343
  if (shed) return Promise.resolve(shed);
1138
1344
  const reqId = this.allocRequestId();
1139
1345
  return new Promise((resolve, reject) => {
1140
- const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1346
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
1141
1347
  this.pending.set(reqId, {
1142
1348
  resolve,
1143
1349
  reject,
1144
- timer
1350
+ timer,
1351
+ ...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
1145
1352
  });
1146
1353
  try {
1147
1354
  if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
@@ -1163,10 +1370,10 @@ var PoolWorker = class {
1163
1370
  }
1164
1371
  allocRequestId() {
1165
1372
  let id = this.nextRequestId;
1166
- this.nextRequestId = id >= 4294967295 ? 1 : id + 1;
1373
+ this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
1167
1374
  while (this.pending.has(id)) {
1168
1375
  id = this.nextRequestId;
1169
- this.nextRequestId = id >= 4294967295 ? 1 : id + 1;
1376
+ this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
1170
1377
  }
1171
1378
  return id;
1172
1379
  }
@@ -1181,6 +1388,8 @@ var PoolWorker = class {
1181
1388
  this.process.stdin.write(payload);
1182
1389
  }
1183
1390
  ensureReady() {
1391
+ const poisonedBy = this.getPoisonDescription();
1392
+ if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
1184
1393
  if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
1185
1394
  }
1186
1395
  /** Time spent in `Buffer.concat` since the last report. */
@@ -1220,23 +1429,273 @@ var PoolWorker = class {
1220
1429
  const reqId = this.receiveBuffer.readUInt32LE(4);
1221
1430
  const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
1222
1431
  this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
1432
+ if (reqId === WORKER_EVENT_REQ_ID) {
1433
+ this.handleWorkerEvent(jsonBytes);
1434
+ continue;
1435
+ }
1223
1436
  const entry = this.pending.get(reqId);
1224
1437
  if (!entry) {
1225
1438
  this.log.warn("Response for unknown request id", { meta: {
1226
1439
  worker: this.opts.workerLabel,
1227
1440
  reqId
1228
1441
  } });
1442
+ this.noteLateReply(jsonBytes);
1229
1443
  continue;
1230
1444
  }
1231
1445
  this.pending.delete(reqId);
1232
1446
  if (entry.timer) clearTimeout(entry.timer);
1233
1447
  try {
1234
1448
  const parsed = JSON.parse(jsonBytes.toString("utf8"));
1449
+ if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
1450
+ else this.inferStall.noteResult(Date.now());
1235
1451
  entry.resolve(parsed);
1236
1452
  } catch (err) {
1237
1453
  entry.reject(err instanceof Error ? err : new Error(String(err)));
1238
1454
  }
1239
1455
  }
1456
+ this.noteAnyReply();
1457
+ if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
1458
+ }
1459
+ /**
1460
+ * Something came back from the worker: its loop was alive just now, so the
1461
+ * missed-deadline verdict is lifted.
1462
+ */
1463
+ noteAnyReply() {
1464
+ this.loadDeadlineMissed = false;
1465
+ }
1466
+ /**
1467
+ * The invariant the load deadline depends on (D653 § 2): loads into one
1468
+ * pool are serialised, so at most ONE load is outstanding per worker. The
1469
+ * worker runs model commands in order and starts a load's soft bound only
1470
+ * when it begins, while the host's deadline for it runs from the SEND. A
1471
+ * load queued behind another could therefore see its host deadline fire
1472
+ * before the worker's named answer. Nothing in the provider issues that; if
1473
+ * anything ever does, this says so.
1474
+ */
1475
+ warnIfLoadQueued(next) {
1476
+ const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
1477
+ if (ahead.length === 0) return;
1478
+ this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
1479
+ worker: this.opts.workerLabel,
1480
+ pid: this.getPid(),
1481
+ runtime: this.opts.poolRuntime,
1482
+ device: this.opts.device ?? "default",
1483
+ model: next.model,
1484
+ queuedBehind: ahead.map((c) => c.model)
1485
+ } });
1486
+ }
1487
+ hasPendingLoad() {
1488
+ for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
1489
+ return false;
1490
+ }
1491
+ /**
1492
+ * A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
1493
+ * or that an earlier one still has (`worker-poisoned`). The worker is not
1494
+ * recycled for it (fix round 1): the compile keeps running so a slow but
1495
+ * finite one writes its cache, the loaded models keep serving, and the worker
1496
+ * starts no new load. Only a compile still running at the HARD bound costs
1497
+ * the process.
1498
+ */
1499
+ inspectCommandReply(raw, command, sentAt) {
1500
+ const reason = raw["reason"];
1501
+ if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
1502
+ if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
1503
+ const model = raw["modelId"];
1504
+ const abandoned = {
1505
+ ...command,
1506
+ model: typeof model === "string" ? model : command.model
1507
+ };
1508
+ const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
1509
+ const reported = raw["compileElapsedMs"];
1510
+ const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
1511
+ const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
1512
+ hardTimer.unref?.();
1513
+ this.abandonedCompile = {
1514
+ command: abandoned,
1515
+ since: Date.now(),
1516
+ hardTimer
1517
+ };
1518
+ this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
1519
+ worker: this.opts.workerLabel,
1520
+ pid: this.getPid(),
1521
+ runtime: this.opts.poolRuntime,
1522
+ device: this.opts.device ?? "default",
1523
+ model: abandoned.model,
1524
+ modelIndex: abandoned.modelIndex,
1525
+ softMs: bounds.softMs,
1526
+ hardMs: bounds.hardMs,
1527
+ loads: "refused until it returns"
1528
+ } });
1529
+ }
1530
+ /** An unsolicited worker event (`compile-finished-late`). */
1531
+ handleWorkerEvent(jsonBytes) {
1532
+ let event;
1533
+ try {
1534
+ event = JSON.parse(jsonBytes.toString("utf8"));
1535
+ } catch {
1536
+ return;
1537
+ }
1538
+ if (event["event"] !== "compile-finished-late") return;
1539
+ const abandoned = this.abandonedCompile;
1540
+ if (abandoned !== null) clearTimeout(abandoned.hardTimer);
1541
+ this.abandonedCompile = null;
1542
+ this.inferStall.noteResult(Date.now());
1543
+ const ok = event["ok"] === true;
1544
+ const meta = {
1545
+ worker: this.opts.workerLabel,
1546
+ pid: this.getPid(),
1547
+ runtime: this.opts.poolRuntime,
1548
+ device: this.opts.device ?? "default",
1549
+ model: typeof event["modelId"] === "string" ? event["modelId"] : null,
1550
+ elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
1551
+ ...typeof event["error"] === "string" ? { error: event["error"] } : {}
1552
+ };
1553
+ if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
1554
+ else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
1555
+ this.opts.onCompileFinishedLate?.({
1556
+ model: meta.model,
1557
+ ok,
1558
+ ...typeof event["error"] === "string" ? { error: event["error"] } : {}
1559
+ });
1560
+ }
1561
+ /**
1562
+ * The GIL grace (D653 round 2): may a silent loop be a compile holding the
1563
+ * GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
1564
+ * on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
1565
+ *
1566
+ * NOT while a compile runs past its soft bound: the worker REPLIED
1567
+ * `compile-timeout`, so its loop is proven alive, and an inference hang
1568
+ * behind that compile — the 2026-09-26 incident exactly — must be seen on
1569
+ * the normal rule, not ~600-900 s later at the hard bound.
1570
+ */
1571
+ gilGraceActive() {
1572
+ if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
1573
+ if (this.loadDeadlineMissed) return false;
1574
+ return this.hasPendingLoad();
1575
+ }
1576
+ /**
1577
+ * The reason a silent worker is recycled with: a freeze that began with a
1578
+ * load missing its deadline is that compile's doing (`compile-hung`, charged
1579
+ * to the model, not the device); anything else is the device's.
1580
+ */
1581
+ silenceReason(fallback) {
1582
+ return this.loadDeadlineMissed ? "compile-hung" : fallback;
1583
+ }
1584
+ /** A live request ended with no result: judge the executor (D653, item 4). */
1585
+ noteInferUnanswered() {
1586
+ const stall = this.inferStall.noteUnanswered(Date.now());
1587
+ if (stall === null || this.gilGraceActive()) return;
1588
+ this.poison(this.silenceReason("infer-unresponsive"), {
1589
+ cmd: "infer",
1590
+ modelIndex: null,
1591
+ model: null
1592
+ }, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
1593
+ }
1594
+ /** A reply whose request was already abandoned: a real result still proves the executor runs. */
1595
+ noteLateReply(jsonBytes) {
1596
+ try {
1597
+ const parsed = JSON.parse(jsonBytes.toString("utf8"));
1598
+ if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
1599
+ } catch {}
1600
+ }
1601
+ /**
1602
+ * Ask a worker whose command just timed out whether its loop still answers.
1603
+ * `mem_stats` is served on the loop, never behind a compile, so a worker
1604
+ * that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
1605
+ * slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
1606
+ * `mem_stats` lost while the process looked alive.
1607
+ */
1608
+ probeLiveness(timedOut) {
1609
+ if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
1610
+ this.probeInFlight = true;
1611
+ this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
1612
+ command: {
1613
+ cmd: "mem_stats",
1614
+ modelIndex: null,
1615
+ model: null
1616
+ },
1617
+ timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
1618
+ probe: true
1619
+ }).then(() => {
1620
+ this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
1621
+ worker: this.opts.workerLabel,
1622
+ pid: this.getPid(),
1623
+ runtime: this.opts.poolRuntime,
1624
+ device: this.opts.device ?? "default",
1625
+ command: timedOut.cmd,
1626
+ modelIndex: timedOut.modelIndex,
1627
+ model: timedOut.model
1628
+ } });
1629
+ }, (err) => {
1630
+ if (this.gilGraceActive()) {
1631
+ this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
1632
+ worker: this.opts.workerLabel,
1633
+ pid: this.getPid(),
1634
+ runtime: this.opts.poolRuntime,
1635
+ device: this.opts.device ?? "default",
1636
+ command: timedOut.cmd,
1637
+ model: timedOut.model
1638
+ } });
1639
+ return;
1640
+ }
1641
+ this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
1642
+ }).finally(() => {
1643
+ this.probeInFlight = false;
1644
+ });
1645
+ }
1646
+ /**
1647
+ * Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
1648
+ * and recycle it once it has drained.
1649
+ *
1650
+ * `ready = false` is what hands the pool back to the provider: its next
1651
+ * dispatch finds the factory not ready and condemns it under the per-device
1652
+ * restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
1653
+ * excludes) — the same path a crashed worker takes, except that a
1654
+ * `compile-hung` death is not charged to the device (see PoolPoisonReason).
1655
+ * An unresponsive worker is killed at once (it will answer nothing it
1656
+ * holds); a compile-hung one keeps serving its loaded models until its
1657
+ * in-flight requests are answered, bounded by the live deadline.
1658
+ */
1659
+ poison(reason, command, detail) {
1660
+ if (this.poisonVerdict !== null || this.exited || this.process === null) return;
1661
+ this.poisonVerdict = {
1662
+ reason,
1663
+ command,
1664
+ detail,
1665
+ pid: this.getPid()
1666
+ };
1667
+ this.ready = false;
1668
+ const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
1669
+ this.log.error("pool worker POISONED — recycling it", { meta: {
1670
+ worker: this.opts.workerLabel,
1671
+ pid: this.getPid(),
1672
+ runtime: this.opts.poolRuntime,
1673
+ device: this.opts.device ?? "default",
1674
+ reason,
1675
+ command: command.cmd,
1676
+ modelIndex: command.modelIndex,
1677
+ model: command.model,
1678
+ detail,
1679
+ inFlight: this.pending.size,
1680
+ inFlightCommands
1681
+ } });
1682
+ if (reason !== "compile-hung" || this.pending.size === 0) {
1683
+ this.recycle();
1684
+ return;
1685
+ }
1686
+ this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
1687
+ this.recycleBackstop.unref?.();
1688
+ }
1689
+ /**
1690
+ * Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
1691
+ * reported and rejects whatever it still held — a deliberate dispose would
1692
+ * silence both, and this is not one.
1693
+ */
1694
+ recycle() {
1695
+ if (this.recycling || this.exited || this.process === null) return;
1696
+ this.recycling = true;
1697
+ if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
1698
+ terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
1240
1699
  }
1241
1700
  rejectAll(err) {
1242
1701
  const entries = [...this.pending.values()];
@@ -1283,8 +1742,10 @@ var SharedInferencePool = class {
1283
1742
  this.tuning = options.tuning ?? null;
1284
1743
  this.numWorkers = Math.max(1, options.numWorkers ?? 1);
1285
1744
  this.device = options.device;
1745
+ this.onCompileFinishedLate = options.onCompileFinishedLate;
1286
1746
  }
1287
1747
  device;
1748
+ onCompileFinishedLate;
1288
1749
  /** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
1289
1750
  /** Summed backlog and shed count across the pool's workers. A rising
1290
1751
  * `inFlight` with a rising `shed` is a worker falling behind; a rising
@@ -1323,7 +1784,8 @@ var SharedInferencePool = class {
1323
1784
  tuning: this.tuning,
1324
1785
  logger: this.log,
1325
1786
  workerLabel: `w${i}`,
1326
- ...this.device ? { device: this.device } : {}
1787
+ ...this.device ? { device: this.device } : {},
1788
+ ...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
1327
1789
  }));
1328
1790
  const t0 = performance.now();
1329
1791
  const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
@@ -1418,7 +1880,7 @@ var SharedInferencePool = class {
1418
1880
  index,
1419
1881
  config: serializeModelConfig(config)
1420
1882
  })));
1421
- for (const resp of responses) if (resp.status !== "ok") throw new Error(`Failed to load model at index ${index}: ${resp.error ?? "unknown"}`);
1883
+ for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
1422
1884
  if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
1423
1885
  return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
1424
1886
  }
@@ -1454,7 +1916,7 @@ var SharedInferencePool = class {
1454
1916
  index,
1455
1917
  config: serializeModelConfig(config)
1456
1918
  })));
1457
- for (const resp of responses) if (resp.status !== "ok") throw new Error(`Failed to replace model at index ${index}: ${resp.error ?? "unknown"}`);
1919
+ for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
1458
1920
  return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
1459
1921
  }
1460
1922
  /**
@@ -1504,9 +1966,35 @@ var SharedInferencePool = class {
1504
1966
  allocateIndex() {
1505
1967
  return this.nextFreeIndex++;
1506
1968
  }
1969
+ /**
1970
+ * Give back an index whose load FAILED, so the retry reuses it. Only the most
1971
+ * recent allocation can be returned. Loads into one pool are serialised by
1972
+ * the provider, so a failed load is normally the latest one; anything else
1973
+ * is left allocated rather than risk handing out a live slot twice.
1974
+ */
1975
+ releaseIndex(index) {
1976
+ if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
1977
+ }
1507
1978
  isReady() {
1508
1979
  return this.workers.length > 0 && this.workers.every((w) => w.isReady());
1509
1980
  }
1981
+ /**
1982
+ * Why a worker of this pool was declared unusable while alive (D653), or
1983
+ * `null`. The provider charges the restart budget with THIS — or, for a
1984
+ * `compile-hung` death, does not charge the device at all — so the
1985
+ * `inference device FAILED` line names the cause instead of "pool worker is
1986
+ * not ready".
1987
+ */
1988
+ getDeathCause() {
1989
+ for (const w of this.workers) {
1990
+ const cause = w.getDeathCause();
1991
+ if (cause !== null) return {
1992
+ ...cause,
1993
+ message: `worker ${cause.message}`
1994
+ };
1995
+ }
1996
+ return null;
1997
+ }
1510
1998
  async dispose() {
1511
1999
  await Promise.all(this.workers.map((w) => w.dispose()));
1512
2000
  this.workers.length = 0;
@@ -1571,6 +2059,23 @@ var SharedInferencePool = class {
1571
2059
  return found;
1572
2060
  }
1573
2061
  };
2062
+ /** A failed command's reply as one message: its reason class first, when it has one. */
2063
+ function describeCommandFailure(resp) {
2064
+ const error = resp.error ?? "unknown";
2065
+ return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
2066
+ }
2067
+ /** Name a command for the lines that must say which one hung (D653). */
2068
+ function describeCommand(cmd) {
2069
+ const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
2070
+ const index = cmd["index"];
2071
+ const config = cmd["config"];
2072
+ const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
2073
+ return {
2074
+ cmd: name,
2075
+ modelIndex: typeof index === "number" ? index : null,
2076
+ model: typeof modelPath === "string" && modelPath.length > 0 ? path$1.basename(modelPath, path$1.extname(modelPath)) : null
2077
+ };
2078
+ }
1574
2079
  function serializeModelConfig(config) {
1575
2080
  const result = {
1576
2081
  path: config.path,
@@ -1858,6 +2363,28 @@ function poolDecodeSettingsEqual(a, b) {
1858
2363
  }
1859
2364
  //#endregion
1860
2365
  //#region src/detection-pipeline/engine/pipeline-model-manager.ts
2366
+ /**
2367
+ * A (step, model) load the pool rejected — carrying the pool's failure class
2368
+ * (`compile-timeout`, …) so the provider can count a timeout against THAT
2369
+ * model without parsing a message (D653).
2370
+ */
2371
+ var StepVariantLoadError = class extends Error {
2372
+ stepId;
2373
+ modelId;
2374
+ poolIndex;
2375
+ reason;
2376
+ /** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
2377
+ poolModel;
2378
+ constructor(stepId, modelId, poolIndex, cause) {
2379
+ super(cause instanceof Error ? cause.message : String(cause));
2380
+ this.name = "StepVariantLoadError";
2381
+ this.stepId = stepId;
2382
+ this.modelId = modelId;
2383
+ this.poolIndex = poolIndex;
2384
+ this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
2385
+ this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
2386
+ }
2387
+ };
1861
2388
  var PipelineModelManager = class {
1862
2389
  pool;
1863
2390
  source;
@@ -1869,11 +2396,20 @@ var PipelineModelManager = class {
1869
2396
  lruClock = 0;
1870
2397
  log;
1871
2398
  maxModelsPerStep;
2399
+ deviceKey;
2400
+ /**
2401
+ * Loads issued and not yet answered, keyed `stepId::modelId`. A second
2402
+ * caller for the same pair JOINS the pending load instead of issuing its own
2403
+ * (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
2404
+ * compile, each allocating a pool index of its own.
2405
+ */
2406
+ pendingLoads = /* @__PURE__ */ new Map();
1872
2407
  constructor(pool, source, logger, options) {
1873
2408
  this.pool = pool;
1874
2409
  this.source = source;
1875
2410
  this.log = logger;
1876
2411
  this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
2412
+ this.deviceKey = options?.deviceKey ?? null;
1877
2413
  }
1878
2414
  /**
1879
2415
  * Apply a new pipeline configuration — driven by the runtime config
@@ -2023,6 +2559,23 @@ var PipelineModelManager = class {
2023
2559
  await this.reconcileDecode(existing, settings);
2024
2560
  return existing;
2025
2561
  }
2562
+ const key = `${stepId}::${modelId}`;
2563
+ const pending = this.pendingLoads.get(key);
2564
+ if (pending) {
2565
+ const joined = await pending;
2566
+ await this.reconcileDecode(joined, settings);
2567
+ return joined;
2568
+ }
2569
+ const load = this.issueLoad(perStep, stepId, modelId, settings);
2570
+ this.pendingLoads.set(key, load);
2571
+ try {
2572
+ return await load;
2573
+ } finally {
2574
+ this.pendingLoads.delete(key);
2575
+ }
2576
+ }
2577
+ /** Evict if at capacity, then send ONE load command and record the result. */
2578
+ async issueLoad(perStep, stepId, modelId, settings) {
2026
2579
  while (perStep.size >= this.maxModelsPerStep) {
2027
2580
  const evicted = this.pickEvictionTarget(stepId);
2028
2581
  if (!evicted) break;
@@ -2034,16 +2587,31 @@ var PipelineModelManager = class {
2034
2587
  cap: this.maxModelsPerStep
2035
2588
  } });
2036
2589
  }
2037
- const index = this.pool.allocateIndex();
2038
2590
  const config = this.source.buildConfig(stepId, modelId, settings);
2039
2591
  const decode = poolDecodeSettingsOf(config);
2592
+ const index = this.pool.allocateIndex();
2040
2593
  this.log.info("Loading step variant", { meta: {
2041
2594
  step: stepId,
2042
2595
  modelId,
2043
2596
  poolIndex: index,
2597
+ deviceKey: this.deviceKey,
2044
2598
  ...decode
2045
2599
  } });
2046
- const { loadMs } = await this.pool.loadModel(index, config);
2600
+ let loadMs;
2601
+ try {
2602
+ ({loadMs} = await this.pool.loadModel(index, config));
2603
+ } catch (err) {
2604
+ this.pool.releaseIndex(index);
2605
+ this.log.error("Step variant load failed", { meta: {
2606
+ step: stepId,
2607
+ modelId,
2608
+ poolIndex: index,
2609
+ deviceKey: this.deviceKey,
2610
+ reason: err instanceof PoolModelLoadError ? err.reason : null,
2611
+ error: err instanceof Error ? err.message : String(err)
2612
+ } });
2613
+ throw new StepVariantLoadError(stepId, modelId, index, err);
2614
+ }
2047
2615
  this.log.info("Step variant loaded", { meta: {
2048
2616
  step: stepId,
2049
2617
  modelId,
@@ -2413,6 +2981,15 @@ var EngineFactory = class {
2413
2981
  isReady() {
2414
2982
  return this.pool?.isReady() ?? false;
2415
2983
  }
2984
+ /**
2985
+ * Why this factory's pool stopped being usable while its process was alive —
2986
+ * a compile that never returned, or a worker that answered nothing (D653) —
2987
+ * or `null`. A pool that merely crashed says nothing here; its `Worker
2988
+ * process exited` line already names the signal.
2989
+ */
2990
+ getDeathCause() {
2991
+ return this.pool?.getDeathCause() ?? null;
2992
+ }
2416
2993
  /** Native pid of the underlying Python pool, if any. */
2417
2994
  getPoolPid() {
2418
2995
  return this.pool?.getPid() ?? null;
@@ -2539,12 +3116,13 @@ var EngineFactory = class {
2539
3116
  concurrency,
2540
3117
  tuning: resolvedTuning,
2541
3118
  numWorkers,
2542
- ...this.opts.engine.device ? { device: this.opts.engine.device } : {}
3119
+ ...this.opts.engine.device ? { device: this.opts.engine.device } : {},
3120
+ ...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
2543
3121
  });
2544
3122
  this.poolManager = new PipelineModelManager(this.pool, {
2545
3123
  buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
2546
3124
  resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
2547
- }, this.log.child("model-mgr"));
3125
+ }, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
2548
3126
  await this.pool.initialize([]);
2549
3127
  await this.poolManager.applyConfig(steps);
2550
3128
  }
@@ -2660,6 +3238,175 @@ function buildPoolModelConfigForStep(inputs) {
2660
3238
  };
2661
3239
  }
2662
3240
  //#endregion
3241
+ //#region src/detection-pipeline/engine/model-load-governor.ts
3242
+ /**
3243
+ * What the dispatch path may ask of a pool's model loads, and what it must say
3244
+ * when it may not (D653, fix round 1).
3245
+ *
3246
+ * Three pieces of state, consulted ON DEMAND by the dispatch that needs a
3247
+ * model — there is no timer anywhere in here:
3248
+ *
3249
+ * 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
3250
+ * failed load, so without it every frame of every camera re-issued the
3251
+ * failing load and wrote two ERRORs. After a failure the model is not
3252
+ * re-issued on that pool for an interval that doubles per consecutive
3253
+ * failure up to a cap; a success clears it.
3254
+ * 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
3255
+ * bound is not a crash of the device. The SAME model timing out
3256
+ * {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
3257
+ * since a slow compile now runs on and writes its cache, only a compile
3258
+ * that never finishes can do — refuses THAT MODEL there, by name. The
3259
+ * device keeps serving every other model.
3260
+ * 3. Which camera has already been TOLD about the current state of a model,
3261
+ * so each camera gets one line per state change, never one per frame.
3262
+ */
3263
+ /** The first back-off after a failed load. */
3264
+ var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
3265
+ /** The back-off doubles per consecutive failure up to this. */
3266
+ var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
3267
+ /** Timeouts older than this no longer count toward a refusal. */
3268
+ var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
3269
+ var ModelLoadGovernor = class {
3270
+ /** Keyed by the pool OBJECT, so a respawned pool starts clean. */
3271
+ backoff = /* @__PURE__ */ new WeakMap();
3272
+ timeouts = /* @__PURE__ */ new Map();
3273
+ reported = /* @__PURE__ */ new Map();
3274
+ generation = 0;
3275
+ /** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
3276
+ check(pool, deviceKey, models, now = Date.now()) {
3277
+ for (const model of models) {
3278
+ const t = this.timeouts.get(timeoutKey(deviceKey, model));
3279
+ if (t?.refused === true) return {
3280
+ kind: "refused",
3281
+ model,
3282
+ timeouts: t.at.length,
3283
+ error: t.lastError
3284
+ };
3285
+ }
3286
+ const perPool = this.backoff.get(pool);
3287
+ if (perPool === void 0) return { kind: "go" };
3288
+ for (const model of models) {
3289
+ const b = perPool.get(model);
3290
+ if (b !== void 0 && now < b.retryAtMs) return {
3291
+ kind: "backoff",
3292
+ model,
3293
+ retryInMs: b.retryAtMs - now,
3294
+ failures: b.failures,
3295
+ error: b.error,
3296
+ generation: b.generation
3297
+ };
3298
+ }
3299
+ return { kind: "go" };
3300
+ }
3301
+ /** One load of `model` on `pool` failed. */
3302
+ recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
3303
+ let perPool = this.backoff.get(pool);
3304
+ if (perPool === void 0) {
3305
+ perPool = /* @__PURE__ */ new Map();
3306
+ this.backoff.set(pool, perPool);
3307
+ }
3308
+ const previousEntry = perPool.get(model);
3309
+ const failures = (previousEntry?.failures ?? 0) + 1;
3310
+ const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
3311
+ this.generation += 1;
3312
+ const generation = this.generation;
3313
+ perPool.set(model, {
3314
+ failures,
3315
+ retryAtMs: now + retryInMs,
3316
+ error,
3317
+ generation,
3318
+ poolModel: poolModel ?? previousEntry?.poolModel ?? null,
3319
+ reason
3320
+ });
3321
+ if (!compileTimeout) return {
3322
+ generation,
3323
+ retryInMs,
3324
+ refusedNow: false,
3325
+ compileTimeouts: 0
3326
+ };
3327
+ const key = timeoutKey(deviceKey, model);
3328
+ const previous = this.timeouts.get(key);
3329
+ const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
3330
+ const refused = at.length >= 2;
3331
+ this.timeouts.set(key, {
3332
+ at,
3333
+ refused,
3334
+ lastError: error
3335
+ });
3336
+ return {
3337
+ generation,
3338
+ retryInMs,
3339
+ refusedNow: refused && previous?.refused !== true,
3340
+ compileTimeouts: at.length
3341
+ };
3342
+ }
3343
+ /**
3344
+ * A compile that outlived its bound came back on `pool` (D653 round 3).
3345
+ *
3346
+ * `ok`: THAT model's cache is written and the worker loads again, so its
3347
+ * back-off is lifted — and only its: another model's failure on the same
3348
+ * pool is not the compile that just finished. Not ok: it is one more
3349
+ * failure of that model, and its back-off grows. Refusals by name are never
3350
+ * lifted here; a model refused for timing out twice stays refused until the
3351
+ * operator re-arms. Returns the governor keys it touched.
3352
+ */
3353
+ settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
3354
+ const perPool = this.backoff.get(pool);
3355
+ if (perPool === void 0) return [];
3356
+ const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
3357
+ for (const key of refusedMeanwhile) perPool.delete(key);
3358
+ const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
3359
+ for (const key of own) if (ok) perPool.delete(key);
3360
+ else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
3361
+ return [...refusedMeanwhile, ...own];
3362
+ }
3363
+ /** `models` loaded on `pool`: forget their failures there. */
3364
+ recordSuccess(pool, deviceKey, models) {
3365
+ const perPool = this.backoff.get(pool);
3366
+ for (const model of models) {
3367
+ perPool?.delete(model);
3368
+ this.timeouts.delete(timeoutKey(deviceKey, model));
3369
+ }
3370
+ }
3371
+ /**
3372
+ * Has `camera` been told about this `state` of `model` on `deviceKey` yet?
3373
+ * Returns `true` exactly once per state change.
3374
+ */
3375
+ shouldReport(camera, deviceKey, model, state) {
3376
+ const key = `${camera ?? "-"}|${deviceKey}|${model}`;
3377
+ if (this.reported.get(key) === state) return false;
3378
+ this.reported.set(key, state);
3379
+ return true;
3380
+ }
3381
+ /** Every model refused on some device. */
3382
+ refused() {
3383
+ const out = [];
3384
+ for (const [key, t] of this.timeouts) {
3385
+ if (!t.refused) continue;
3386
+ const [deviceKey = "", model = ""] = key.split("\0");
3387
+ out.push({
3388
+ deviceKey,
3389
+ model,
3390
+ timeouts: t.at.length,
3391
+ lastError: t.lastError
3392
+ });
3393
+ }
3394
+ return out;
3395
+ }
3396
+ /** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
3397
+ rearm(deviceKey) {
3398
+ let cleared = 0;
3399
+ for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
3400
+ this.timeouts.delete(key);
3401
+ cleared += 1;
3402
+ }
3403
+ return cleared;
3404
+ }
3405
+ };
3406
+ function timeoutKey(deviceKey, model) {
3407
+ return `${deviceKey}\u0000${model}`;
3408
+ }
3409
+ //#endregion
2663
3410
  //#region src/detection-pipeline/engine/idle-pool-reaper.ts
2664
3411
  var IdlePoolReaper = class {
2665
3412
  lastUsed = /* @__PURE__ */ new Map();
@@ -6556,6 +7303,34 @@ function parseWavToAudioChunk(filePath) {
6556
7303
  function enginesEqual(a, b) {
6557
7304
  return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
6558
7305
  }
7306
+ /** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
7307
+ var ModelLoadRefusedError = class extends Error {
7308
+ model;
7309
+ state;
7310
+ constructor(message, model, state) {
7311
+ super(message);
7312
+ this.name = "ModelLoadRefusedError";
7313
+ this.model = model;
7314
+ this.state = state;
7315
+ }
7316
+ };
7317
+ /** `step/model` — how the governor and the log lines name a model. */
7318
+ function stepModelKey(step) {
7319
+ return `${step.addonId}/${step.modelId}`;
7320
+ }
7321
+ /** The reporting state a refusal verdict stands for. */
7322
+ function stateOf(verdict) {
7323
+ return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
7324
+ }
7325
+ /**
7326
+ * The reason a condemned pool is charged against its device's restart budget.
7327
+ * A pool recycled because it HUNG (D653) names the hang — `inference device
7328
+ * FAILED … reason: pool worker is not ready` could not tell a GPU compile that
7329
+ * never returned from a crash.
7330
+ */
7331
+ function deathReasonOf(factory) {
7332
+ return factory.getDeathCause()?.message ?? "pool worker is not ready";
7333
+ }
6559
7334
  /** Build a `RuntimeEnv` from the running process + probed hardware. */
6560
7335
  function runtimeEnvFromProcess(hardware) {
6561
7336
  return {
@@ -8033,7 +8808,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8033
8808
  addonId: s.addonId,
8034
8809
  modelId: s.modelId
8035
8810
  });
8036
- await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format);
8811
+ await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
8812
+ ...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
8813
+ deviceKey: deviceKeyOf(dispatchEngine)
8814
+ });
8037
8815
  emit(`All models loaded`);
8038
8816
  }
8039
8817
  emit("Running inference...");
@@ -8330,53 +9108,193 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8330
9108
  throw new Error(message);
8331
9109
  }
8332
9110
  /**
8333
- * Single-flight gate around `ensureModelsForSteps`. Concurrent
8334
- * callers (every camera that fires motion in the same window calls
8335
- * `runPipeline` → `ensureModelsForSteps`) share the same in-flight
8336
- * promise instead of each racing into `loadAdditional`. Without this
8337
- * the pool gets the SAME model loaded N times under N distinct
8338
- * pool indices because every caller saw `isLoadedWithModel === false`
8339
- * before any of them committed via `stepToLoaded.set`.
8340
- */
8341
- modelLoadInFlight = null;
9111
+ * In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
9112
+ * needing the same models on the same pool share ONE load and its outcome;
9113
+ * a dispatch needing a different set is never handed another set's failure
9114
+ * — it queues behind the pool's current load ({@link modelLoadTail}) and
9115
+ * then decides for itself. Keyed by pool, not node-wide: one GPU compile
9116
+ * that never returned held EVERY load on the node — the NPU's and the CPU's
9117
+ * too — behind it.
9118
+ */
9119
+ modelLoadInFlight = /* @__PURE__ */ new Map();
9120
+ /** The last load issued on each pool — loads into one pool run one at a time. */
9121
+ modelLoadTail = /* @__PURE__ */ new Map();
9122
+ /** Stands in for "the node-default pool" before it exists. */
9123
+ nullPoolKey = {};
9124
+ /** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
9125
+ loadGovernor = new ModelLoadGovernor();
8342
9126
  /** Ensure all models needed by steps are downloaded and loaded in the engine pool.
8343
9127
  * `factory`/`format` default to the node's engine; a per-device dispatch passes
8344
- * that device's factory + format (Phase 2 multi-device). */
8345
- async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx") {
8346
- while (this.modelLoadInFlight) await this.modelLoadInFlight;
8347
- const allEnabled = flattenSteps(steps);
9128
+ * that device's factory + format (Phase 2 multi-device). `context` names the
9129
+ * camera and the accelerator on the line a rejected load writes. */
9130
+ async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
9131
+ const needed = this.stepsNeedingLoad(steps, factory);
9132
+ if (needed.length === 0) return;
9133
+ const pool = factory ?? this.nullPoolKey;
9134
+ const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
9135
+ const models = needed.map(stepModelKey);
9136
+ const refusal = this.loadRefusal(pool, deviceKey, models);
9137
+ if (refusal !== null) {
9138
+ this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
9139
+ throw refusal;
9140
+ }
9141
+ const setKey = [...models].sort().join(",");
9142
+ let perPool = this.modelLoadInFlight.get(pool);
9143
+ if (perPool === void 0) {
9144
+ perPool = /* @__PURE__ */ new Map();
9145
+ this.modelLoadInFlight.set(pool, perPool);
9146
+ }
9147
+ const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
9148
+ try {
9149
+ await flight.work;
9150
+ } catch (err) {
9151
+ if (err instanceof ModelLoadRefusedError) {
9152
+ this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
9153
+ throw err;
9154
+ }
9155
+ const failed = err instanceof StepVariantLoadError ? `${err.stepId}/${err.modelId}` : models[0] ?? "";
9156
+ this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, errMsg(err));
9157
+ throw err;
9158
+ }
9159
+ }
9160
+ /**
9161
+ * Why `models` may not be loaded on `pool` right now — a back-off after a
9162
+ * failure, or a refusal by name after repeated compile timeouts — or `null`.
9163
+ */
9164
+ loadRefusal(pool, deviceKey, models) {
9165
+ const verdict = this.loadGovernor.check(pool, deviceKey, models);
9166
+ if (verdict.kind === "go") return null;
9167
+ return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
9168
+ }
9169
+ /**
9170
+ * A compile that outlived its soft bound came back on `factory`'s pool
9171
+ * (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
9172
+ * again, so that model's back-off is lifted at once rather than waited out;
9173
+ * no other model's is. A FAILURE is one more failure of that model, and its
9174
+ * back-off grows.
9175
+ */
9176
+ noteLateCompile(factory, deviceKey, event) {
9177
+ const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
9178
+ if (event.ok) {
9179
+ this.log.info("model loadable again after late compile", { meta: {
9180
+ deviceKey,
9181
+ model: event.model,
9182
+ backoffLifted: models
9183
+ } });
9184
+ return;
9185
+ }
9186
+ this.log.warn("late compile failed — the model stays backed off", { meta: {
9187
+ deviceKey,
9188
+ model: event.model,
9189
+ backoffExtended: models,
9190
+ error: event.error ?? null
9191
+ } });
9192
+ }
9193
+ /** The enabled steps the factory's pool does not reflect yet. */
9194
+ stepsNeedingLoad(steps, factory) {
8348
9195
  const needed = [];
8349
- for (const step of allEnabled) {
9196
+ for (const step of flattenSteps(steps)) {
8350
9197
  if (!step.enabled) continue;
8351
9198
  if (factory !== null && !factory.needsPoolUpdate(step)) continue;
8352
9199
  needed.push(step);
8353
9200
  }
8354
- if (needed.length === 0) return;
8355
- const work = (async () => {
8356
- for (const step of needed) {
8357
- const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
8358
- if (modelEntry && !isModelDownloaded(this.modelsDir, modelEntry, format)) {
8359
- if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
8360
- if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
8361
- this.log.info("Downloading model for step", { meta: {
8362
- modelId: step.modelId,
8363
- format,
8364
- step: step.addonId
8365
- } });
8366
- await this.downloadWithRetry(modelEntry, format, 3);
8367
- }
9201
+ return needed;
9202
+ }
9203
+ /**
9204
+ * Issue ONE load for a model set on a pool, queued behind the pool's
9205
+ * previous load, and record its outcome with the governor. The set is
9206
+ * re-derived once the queue reaches it: the load ahead may have brought
9207
+ * some of it in.
9208
+ */
9209
+ issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
9210
+ const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
9211
+ const flight = {
9212
+ work: Promise.resolve(),
9213
+ generation: null
9214
+ };
9215
+ flight.work = (async () => {
9216
+ await previous;
9217
+ const needed = this.stepsNeedingLoad(steps, factory);
9218
+ if (needed.length === 0) return;
9219
+ const models = needed.map(stepModelKey);
9220
+ const refusal = this.loadRefusal(pool, deviceKey, models);
9221
+ if (refusal !== null) throw refusal;
9222
+ try {
9223
+ await this.downloadNeededModels(needed, format);
9224
+ this.log.info("Loading additional models for benchmark", { meta: {
9225
+ count: needed.length,
9226
+ models,
9227
+ deviceKey
9228
+ } });
9229
+ await factory.loadAdditional(needed);
9230
+ } catch (err) {
9231
+ flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
9232
+ throw err;
8368
9233
  }
8369
- this.log.info("Loading additional models for benchmark", { meta: {
8370
- count: needed.length,
8371
- models: needed.map((s) => `${s.addonId}/${s.modelId}`)
8372
- } });
8373
- await factory.loadAdditional(needed);
9234
+ this.loadGovernor.recordSuccess(pool, deviceKey, models);
8374
9235
  })();
8375
- this.modelLoadInFlight = work;
8376
- try {
8377
- await work;
8378
- } finally {
8379
- if (this.modelLoadInFlight === work) this.modelLoadInFlight = null;
9236
+ perPool.set(setKey, flight);
9237
+ const tail = flight.work.catch(() => void 0);
9238
+ this.modelLoadTail.set(pool, tail);
9239
+ tail.then(() => {
9240
+ if (perPool.get(setKey) === flight) perPool.delete(setKey);
9241
+ if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
9242
+ });
9243
+ return flight;
9244
+ }
9245
+ /** Charge a failed load to the model that failed; say so once if it is now refused. */
9246
+ recordModelLoadFailure(pool, deviceKey, models, err) {
9247
+ const failedModels = err instanceof StepVariantLoadError ? [`${err.stepId}/${err.modelId}`] : models;
9248
+ const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
9249
+ const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
9250
+ let generation = 0;
9251
+ for (const model of failedModels) {
9252
+ const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
9253
+ generation = outcome.generation;
9254
+ if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
9255
+ deviceKey,
9256
+ model,
9257
+ compileTimeouts: outcome.compileTimeouts,
9258
+ error: errMsg(err),
9259
+ device: "keeps serving every other model",
9260
+ rearm: "pipelineExecutor.rearmInferenceDevice"
9261
+ } });
9262
+ }
9263
+ return generation;
9264
+ }
9265
+ /**
9266
+ * One WARN per camera per state change of the model that cost it its frame.
9267
+ * The ERROR for the failed load itself is written once, where the load
9268
+ * failed (`Step variant load failed`); this line answers "which cameras
9269
+ * did it cost?", tagged so it can be counted per camera.
9270
+ */
9271
+ reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
9272
+ if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
9273
+ this.log.warn("model load for dispatch failed", {
9274
+ ...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
9275
+ meta: {
9276
+ deviceKey,
9277
+ models,
9278
+ model,
9279
+ state,
9280
+ error
9281
+ }
9282
+ });
9283
+ }
9284
+ /** Download any model of `needed` missing in `format` (the device's). */
9285
+ async downloadNeededModels(needed, format) {
9286
+ for (const step of needed) {
9287
+ const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
9288
+ if (modelEntry && !isModelDownloaded(this.modelsDir, modelEntry, format)) {
9289
+ if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
9290
+ if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
9291
+ this.log.info("Downloading model for step", { meta: {
9292
+ modelId: step.modelId,
9293
+ format,
9294
+ step: step.addonId
9295
+ } });
9296
+ await this.downloadWithRetry(modelEntry, format, 3);
9297
+ }
8380
9298
  }
8381
9299
  }
8382
9300
  /** Download a model with retry + exponential backoff */
@@ -8504,7 +9422,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8504
9422
  await this.ensureEngineFactory();
8505
9423
  const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
8506
9424
  if (!factory) throw new Error("runPipelineBatch: factory not initialised");
8507
- if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat);
9425
+ if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
8508
9426
  const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
8509
9427
  this.log.info("runPipelineBatch path decision", { meta: {
8510
9428
  phase: "batch",
@@ -8759,7 +9677,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8759
9677
  if (this.engineFactory.isReady()) return;
8760
9678
  const dead = this.engineFactory;
8761
9679
  this.engineFactory = null;
8762
- this.noteDeviceDeath(defaultDeviceKey, "pool worker is not ready");
9680
+ this.noteFactoryDeath(defaultDeviceKey, dead);
8763
9681
  await dead.dispose().catch(() => void 0);
8764
9682
  this.refuseIfDeviceUnusable(defaultDeviceKey);
8765
9683
  }
@@ -8777,7 +9695,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8777
9695
  logger: this.log.child("engine"),
8778
9696
  pythonPath: this.executorOptions.pythonPath ?? "",
8779
9697
  provisioning: this.executorOptions.provisioning,
8780
- resolveCustomModel: this.customModelResolver
9698
+ resolveCustomModel: this.customModelResolver,
9699
+ onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
8781
9700
  });
8782
9701
  try {
8783
9702
  await factory.initialize([]);
@@ -8897,15 +9816,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8897
9816
  * dispose — which rejects whatever was still in flight on it with a reason
8898
9817
  * rather than letting those requests sit until their own deadlines.
8899
9818
  */
8900
- async condemnDeviceFactory(deviceKey, factory, reason) {
9819
+ async condemnDeviceFactory(deviceKey, factory) {
8901
9820
  if (this.factoriesByDevice.get(deviceKey) === factory) {
8902
9821
  this.factoriesByDevice.delete(deviceKey);
8903
9822
  this.deviceReaper.cancel(deviceKey);
8904
- this.noteDeviceDeath(deviceKey, reason);
9823
+ this.noteFactoryDeath(deviceKey, factory);
8905
9824
  }
8906
9825
  await factory.dispose().catch(() => void 0);
8907
9826
  }
8908
9827
  /**
9828
+ * Charge one dead pool to its device — unless it died of a compile that was
9829
+ * still running at its hard bound (D653, fix round 1). That death is not a
9830
+ * crash of the DEVICE: the model that hung was already counted when its
9831
+ * load answered `compile-timeout`, and the same model timing out again on
9832
+ * the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
9833
+ * the respawns. A crash stays a crash.
9834
+ */
9835
+ noteFactoryDeath(deviceKey, factory) {
9836
+ const cause = factory.getDeathCause();
9837
+ if (cause !== null && !cause.chargesDeviceBudget) {
9838
+ this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
9839
+ deviceKey,
9840
+ reason: cause.reason,
9841
+ cause: cause.message
9842
+ } });
9843
+ return;
9844
+ }
9845
+ this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
9846
+ }
9847
+ /**
8909
9848
  * Every inference device this node currently refuses, and why.
8910
9849
  *
8911
9850
  * The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
@@ -8940,7 +9879,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8940
9879
  state: "backoff",
8941
9880
  since: Date.now(),
8942
9881
  deaths: 0,
8943
- lastError: "pool worker is not ready"
9882
+ lastError: deathReasonOf(factory)
8944
9883
  });
8945
9884
  }
8946
9885
  return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
@@ -8957,10 +9896,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8957
9896
  * no-op, reported as such.
8958
9897
  */
8959
9898
  async rearmInferenceDevice(input) {
8960
- const rearmed = this.deviceLiveness.rearm(input.deviceKey);
9899
+ const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
9900
+ const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
9901
+ const rearmed = rearmedDevice || rearmedModels > 0;
8961
9902
  this.log.info("inference device re-armed by operator", { meta: {
8962
9903
  deviceKey: input.deviceKey,
8963
- rearmed
9904
+ rearmed,
9905
+ rearmedModels
8964
9906
  } });
8965
9907
  return { rearmed };
8966
9908
  }
@@ -8977,7 +9919,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8977
9919
  this.deviceReaper.touch(deviceKey);
8978
9920
  return existing;
8979
9921
  }
8980
- await this.condemnDeviceFactory(deviceKey, existing, "pool worker is not ready");
9922
+ await this.condemnDeviceFactory(deviceKey, existing);
8981
9923
  this.refuseIfDeviceUnusable(deviceKey);
8982
9924
  }
8983
9925
  const inflight = this.deviceFactoryInflight.get(deviceKey);
@@ -8990,7 +9932,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8990
9932
  logger: this.log.child(`engine:${deviceKey}`),
8991
9933
  pythonPath: this.executorOptions.pythonPath ?? "",
8992
9934
  provisioning: this.executorOptions.provisioning,
8993
- resolveCustomModel: this.customModelResolver
9935
+ resolveCustomModel: this.customModelResolver,
9936
+ onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
8994
9937
  });
8995
9938
  try {
8996
9939
  await factory.initialize([]);