@camstack/addon-pipeline 1.2.177 → 1.2.179

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/audio-analyzer/index.js +2 -2
  2. package/dist/audio-analyzer/index.mjs +2 -2
  3. package/dist/detection-pipeline/index.js +130 -22
  4. package/dist/detection-pipeline/index.mjs +129 -21
  5. package/dist/{dist-DxA44rNd.js → dist-Bbu27T_N.js} +42 -0
  6. package/dist/{dist-D7aPrnxq.mjs → dist-pMHigtQu.mjs} +42 -0
  7. package/dist/{lazy-sharp-B9gAzlWk.js → lazy-sharp-V_rih5vD.js} +1 -1
  8. package/dist/{local-frame-registry-CZGAexUl.mjs → local-frame-registry-C7uCSrQG.mjs} +1 -1
  9. package/dist/{local-frame-registry-8Hzf6pPx.js → local-frame-registry-DK-Z1Ev1.js} +1 -1
  10. package/dist/motion-wasm/index.js +2 -2
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node--gufH5K3.mjs → node-CmPPt-uJ.mjs} +1 -1
  13. package/dist/{node-D8a2tmPR.js → node-DMTe6WuM.js} +1 -1
  14. package/dist/pipeline-runner/index.js +5 -5
  15. package/dist/pipeline-runner/index.mjs +4 -4
  16. package/dist/{process-memory-D2ik9mux.mjs → process-memory-BSR1vUsp.mjs} +1 -1
  17. package/dist/{process-memory-C6gjUbUI.js → process-memory-DFA42X6A.js} +1 -1
  18. package/dist/recorder/index.js +2 -2
  19. package/dist/recorder/index.mjs +2 -2
  20. package/dist/{segment-demux-js-CUrDbmvm.mjs → segment-demux-js-D-yE6_2O.mjs} +1 -1
  21. package/dist/{segment-demux-js-CRPLwKUP.js → segment-demux-js-jj19HZoq.js} +1 -1
  22. package/dist/session-decode/decode-worker-child.js +2 -2
  23. package/dist/session-decode/decode-worker-child.mjs +1 -1
  24. package/dist/stream-broker/_stub.js +1 -1
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-DVfYOF91.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-CPhBUByO.mjs} +3 -3
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-BpHWJmKV.mjs +26 -0
  27. package/dist/stream-broker/demux-worker-child.js +1 -1
  28. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  29. package/dist/stream-broker/{hostInit-BaMXeegw.mjs → hostInit-9unPdu_p.mjs} +3 -3
  30. package/dist/stream-broker/index.js +3 -3
  31. package/dist/stream-broker/index.mjs +3 -3
  32. package/dist/stream-broker/remoteEntry.js +1 -1
  33. package/dist/{worker-protocol-BZQMaDfm.mjs → worker-protocol-BqiZM7J3.mjs} +1 -1
  34. package/dist/{worker-protocol-DX7eACGC.js → worker-protocol-D_44Incp.js} +1 -1
  35. package/package.json +1 -1
  36. package/python/__pycache__/inference_pool.cpython-314.pyc +0 -0
  37. package/python/__pycache__/test_inference_pool_deadline.cpython-314.pyc +0 -0
  38. package/python/inference_pool.py +225 -28
  39. package/python/test_inference_pool_deadline.py +253 -0
  40. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-B_HZNoVq.mjs +0 -26
@@ -2,8 +2,8 @@ Object.defineProperties(exports, {
2
2
  __esModule: { value: true },
3
3
  [Symbol.toStringTag]: { value: "Module" }
4
4
  });
5
- const require_dist = require("../dist-DxA44rNd.js");
6
- const require_process_memory = require("../process-memory-C6gjUbUI.js");
5
+ const require_dist = require("../dist-Bbu27T_N.js");
6
+ const require_process_memory = require("../process-memory-DFA42X6A.js");
7
7
  let node_fs = require("node:fs");
8
8
  node_fs = require_dist.__toESM(node_fs);
9
9
  let node_path = require("node:path");
@@ -1,6 +1,6 @@
1
1
  import { n as __require } from "../chunk-DnnnRqeS.mjs";
2
- import { A as audioAnalysisCapability, It as hydrateSchema, S as PoolMemoryWatchdog, _t as resolvePoolMemoryPolicy, j as audioAnalyzerCapability, jt as BaseAddon, n as AUDIO_BACKEND_CHOICES, nt as mapAudioLabelToMacro, s as DEFAULT_AUDIO_ANALYZER_CONFIG, v as HF_BASE_URL, wt as errMsg } from "../dist-D7aPrnxq.mjs";
3
- import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-D2ik9mux.mjs";
2
+ import { A as audioAnalysisCapability, It as hydrateSchema, S as PoolMemoryWatchdog, _t as resolvePoolMemoryPolicy, j as audioAnalyzerCapability, jt as BaseAddon, n as AUDIO_BACKEND_CHOICES, nt as mapAudioLabelToMacro, s as DEFAULT_AUDIO_ANALYZER_CONFIG, v as HF_BASE_URL, wt as errMsg } from "../dist-pMHigtQu.mjs";
3
+ import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-BSR1vUsp.mjs";
4
4
  import * as fs from "node:fs";
5
5
  import * as path$1 from "node:path";
6
6
  import { downloadFile } from "@camstack/system/addon-utils";
@@ -2,12 +2,12 @@ Object.defineProperties(exports, {
2
2
  __esModule: { value: true },
3
3
  [Symbol.toStringTag]: { value: "Module" }
4
4
  });
5
- const require_dist = require("../dist-DxA44rNd.js");
6
- const require_node = require("../node-D8a2tmPR.js");
7
- const require_local_frame_registry = require("../local-frame-registry-8Hzf6pPx.js");
8
- const require_lazy_sharp = require("../lazy-sharp-B9gAzlWk.js");
5
+ const require_dist = require("../dist-Bbu27T_N.js");
6
+ const require_node = require("../node-DMTe6WuM.js");
7
+ const require_local_frame_registry = require("../local-frame-registry-DK-Z1Ev1.js");
8
+ const require_lazy_sharp = require("../lazy-sharp-V_rih5vD.js");
9
9
  const require_event_loop_stall_monitor = require("../event-loop-stall-monitor-DaUBcb13.js");
10
- const require_process_memory = require("../process-memory-C6gjUbUI.js");
10
+ const require_process_memory = require("../process-memory-DFA42X6A.js");
11
11
  let node_fs = require("node:fs");
12
12
  node_fs = require_dist.__toESM(node_fs);
13
13
  let node_path = require("node:path");
@@ -679,6 +679,8 @@ var MSG_INFER_CACHED = 5;
679
679
  */
680
680
  var MSG_INFER_BATCH = 3;
681
681
  var PREFIX_LEN = 9;
682
+ /** Bytes of the absolute-deadline prefix a v2 sheddable payload carries. */
683
+ var DEADLINE_PREFIX_LEN = 8;
682
684
  /**
683
685
  * Wire-level enum for the raw-frame fast path. Values are append-only:
684
686
  * the Python pool reads the byte directly off the IPC frame; reordering
@@ -867,6 +869,20 @@ var PoolWorker = class {
867
869
  * Surfaced so "the camera is quiet" and "the camera is saturated" can be
868
870
  * told apart without reading the logs. */
869
871
  shedCount = 0;
872
+ /**
873
+ * Live requests whose deadline fired IN FLIGHT and were answered dropped by
874
+ * the local watchdog (the worker never replied in time). With a v2 worker
875
+ * this is the rare backstop — the worker sheds expired work itself and
876
+ * replies fast; against a v1 worker it is the only shed there is.
877
+ */
878
+ deadlineShedCount = 0;
879
+ /**
880
+ * Negotiated wire version for THIS worker: `min(POOL_PROTOCOL_VERSION,
881
+ * what the ready handshake reported)`. A worker that reports nothing is a
882
+ * v1 worker and keeps receiving the byte-identical old layout — the deploy
883
+ * skew case that must always work (D350).
884
+ */
885
+ wireVersion = 1;
870
886
  nextRequestId = 1;
871
887
  ready = false;
872
888
  log;
@@ -936,6 +952,7 @@ var PoolWorker = class {
936
952
  const config = {
937
953
  runtime: this.opts.poolRuntime,
938
954
  concurrency: this.opts.concurrency,
955
+ protocolVersion: 2,
939
956
  models: initialModels.map((m) => serializeModelConfig(m))
940
957
  };
941
958
  if (this.opts.device) config["device"] = this.opts.device;
@@ -960,10 +977,13 @@ var PoolWorker = class {
960
977
  this.ready = true;
961
978
  const loadedCount = result["models"];
962
979
  const startupMs = result["startupMs"];
980
+ const workers = result["workers"] ?? 1;
981
+ const reported = result["protocolVersion"];
982
+ this.wireVersion = Math.min(2, typeof reported === "number" && Number.isFinite(reported) ? reported : 1);
963
983
  resolve({
964
984
  startupMs,
965
985
  loadedCount,
966
- workers: result["workers"] ?? 1
986
+ workers
967
987
  });
968
988
  } else reject(/* @__PURE__ */ new Error(`Unexpected pool status: ${JSON.stringify(result)}`));
969
989
  },
@@ -1095,19 +1115,35 @@ var PoolWorker = class {
1095
1115
  deadlineFor(msgType) {
1096
1116
  return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
1097
1117
  }
1118
+ /**
1119
+ * The request's ABSOLUTE deadline for the wire (D350), or `null` when this
1120
+ * request must not carry one: commands / model loads / cacheFrame at any
1121
+ * version, and EVERYTHING against a v1 worker — the old layout has no room
1122
+ * for the prefix, and a worker at a different version is the normal state
1123
+ * during a rolling deploy. The stamp is the same instant the local watchdog
1124
+ * arms, so "the TS side has abandoned this" and "the worker refuses to run
1125
+ * it" are one moment, not two clocks drifting apart.
1126
+ */
1127
+ deadlineStamp(msgType) {
1128
+ if (this.wireVersion < 2 || !SHEDDABLE_MSG_TYPES.has(msgType)) return null;
1129
+ const stamp = Buffer.allocUnsafe(DEADLINE_PREFIX_LEN);
1130
+ stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
1131
+ return stamp;
1132
+ }
1098
1133
  dispatch(msgType, payload, deviceId) {
1099
1134
  const shed = this.shedIfSaturated(msgType, deviceId);
1100
1135
  if (shed) return Promise.resolve(shed);
1101
1136
  const reqId = this.allocRequestId();
1102
1137
  return new Promise((resolve, reject) => {
1103
- const timer = this.armRequestTimeout(reqId, reject, this.deadlineFor(msgType), deviceId);
1138
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1104
1139
  this.pending.set(reqId, {
1105
1140
  resolve,
1106
1141
  reject,
1107
1142
  timer
1108
1143
  });
1109
1144
  try {
1110
- this.writeFrame(reqId, msgType, payload);
1145
+ const stamp = this.deadlineStamp(msgType);
1146
+ this.writeFrame(reqId, msgType, stamp ? Buffer.concat([stamp, payload]) : payload);
1111
1147
  } catch (err) {
1112
1148
  clearTimeout(timer);
1113
1149
  this.pending.delete(reqId);
@@ -1116,13 +1152,51 @@ var PoolWorker = class {
1116
1152
  });
1117
1153
  }
1118
1154
  /**
1119
- * Arm a watchdog that rejects a still-pending inference request after
1120
- * `POOL_INFER_TIMEOUT_MS`. `unref` so it never keeps the event loop alive.
1121
- * The normal response + `rejectAll` paths clear it via `PendingRequest.timer`.
1122
- */
1123
- armRequestTimeout(reqId, reject, timeoutMs, deviceId) {
1155
+ * Arm a watchdog for a still-pending request. `unref` so it never keeps the
1156
+ * event loop alive. The normal response + `rejectAll` paths clear it via
1157
+ * `PendingRequest.timer`.
1158
+ *
1159
+ * The two request classes settle DIFFERENTLY on expiry (D350):
1160
+ *
1161
+ * - A SHEDDABLE (live inference) request RESOLVES as a shed —
1162
+ * `{dropped: true, shedReason: 'deadline-expired'}` — because at this
1163
+ * point the answer is worthless whether or not it eventually arrives, and
1164
+ * a shed is flow control the executor already handles. It fires only when
1165
+ * the worker never replied at all: a v2 worker sheds expired work itself
1166
+ * (fast reply, this timer is cleared), so this is the wedged-worker
1167
+ * backstop and the v1-worker compatibility path. The 2026-09-04 storm
1168
+ * logged this state 16 146 times at ERROR; the shed is logged sampled at
1169
+ * WARN and counted (`getDeadlineExpiredShedCount`) instead.
1170
+ * - Anything else (commands, model loads) still REJECTS with the error the
1171
+ * dashboards grep for — a lost command is a fault, not flow control.
1172
+ */
1173
+ armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
1174
+ const timeoutMs = this.deadlineFor(msgType);
1124
1175
  const timer = setTimeout(() => {
1125
1176
  if (this.pending.delete(reqId)) {
1177
+ if (SHEDDABLE_MSG_TYPES.has(msgType)) {
1178
+ this.deadlineShedCount++;
1179
+ const total = this.deadlineShedCount;
1180
+ if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.warn("live inference deadline expired in flight — answered dropped", {
1181
+ ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1182
+ meta: {
1183
+ worker: this.opts.workerLabel,
1184
+ pid: this.getPid(),
1185
+ runtime: this.opts.poolRuntime,
1186
+ device: this.opts.device ?? "default",
1187
+ inFlight: this.pending.size,
1188
+ reqId,
1189
+ timeoutMs,
1190
+ deadlineShedTotal: total,
1191
+ sampledEvery: SHED_LOG_SAMPLE_EVERY
1192
+ }
1193
+ });
1194
+ resolve({
1195
+ dropped: true,
1196
+ shedReason: "deadline-expired"
1197
+ });
1198
+ return;
1199
+ }
1126
1200
  this.log.error("inference request timed out", {
1127
1201
  ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1128
1202
  meta: {
@@ -1146,7 +1220,7 @@ var PoolWorker = class {
1146
1220
  if (shed) return Promise.resolve(shed);
1147
1221
  const reqId = this.allocRequestId();
1148
1222
  return new Promise((resolve, reject) => {
1149
- const timer = this.armRequestTimeout(reqId, reject, this.deadlineFor(msgType), deviceId);
1223
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1150
1224
  this.pending.set(reqId, {
1151
1225
  resolve,
1152
1226
  reject,
@@ -1154,11 +1228,14 @@ var PoolWorker = class {
1154
1228
  });
1155
1229
  try {
1156
1230
  if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
1231
+ const stamp = this.deadlineStamp(msgType);
1232
+ const wireLen = payloadLen + (stamp ? DEADLINE_PREFIX_LEN : 0);
1157
1233
  const prefix = Buffer.allocUnsafe(PREFIX_LEN);
1158
- prefix.writeUInt32LE(5 + payloadLen, 0);
1234
+ prefix.writeUInt32LE(5 + wireLen, 0);
1159
1235
  prefix.writeUInt32LE(reqId, 4);
1160
1236
  prefix[8] = msgType;
1161
1237
  this.process.stdin.write(prefix);
1238
+ if (stamp) this.process.stdin.write(stamp);
1162
1239
  for (const part of parts) this.process.stdin.write(part);
1163
1240
  } catch (err) {
1164
1241
  clearTimeout(timer);
@@ -1261,12 +1338,22 @@ var SharedInferencePool = class {
1261
1338
  nextFreeIndex = 0;
1262
1339
  nextFrameId = 1;
1263
1340
  /**
1264
- * Cumulative count of frames the Python pool SHED under overload
1265
- * (`"dropped": true` responses from the per-model in-flight bound in
1266
- * inference_pool.py). Without this the shed response is
1267
- * indistinguishable from a genuine "no detections" result.
1341
+ * Cumulative count of dropped (shed) responses on the single-frame and
1342
+ * batch inference paths: the Python per-model bound, the Python
1343
+ * deadline-expired shed, and the TS in-flight watchdog's own
1344
+ * deadline-expired resolution all land here. Without this a shed response
1345
+ * is indistinguishable from a genuine "no detections" result.
1268
1346
  */
1269
1347
  droppedResponseCount = 0;
1348
+ /**
1349
+ * The `shedReason: 'deadline-expired'` subset of {@link
1350
+ * droppedResponseCount} — work that was ALREADY DEAD when it would have
1351
+ * run (D350). This is the number that was zero-by-construction while the
1352
+ * 2026-09-04 storm burned the accelerator on abandoned requests for 4h15m;
1353
+ * a rising value here under load is the pool refusing dead work, which is
1354
+ * the fix working, not a fault.
1355
+ */
1356
+ deadlineExpiredShedCount = 0;
1270
1357
  log;
1271
1358
  concurrency;
1272
1359
  tuning;
@@ -1352,7 +1439,12 @@ var SharedInferencePool = class {
1352
1439
  }
1353
1440
  async inferBatch(modelIndex, items, frameId = 0) {
1354
1441
  if (items.length > 255) throw new Error(`SharedInferencePool.inferBatch: max 255 items per call, got ${items.length}`);
1355
- return this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
1442
+ const results = await this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
1443
+ for (const item of results) if (item["dropped"] === true) {
1444
+ this.droppedResponseCount++;
1445
+ if (item["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
1446
+ }
1447
+ return results;
1356
1448
  }
1357
1449
  async inferCached(modelIndex, frameId, deviceId) {
1358
1450
  const w = this.pickWorker();
@@ -1366,6 +1458,16 @@ var SharedInferencePool = class {
1366
1458
  getDroppedResponseCount() {
1367
1459
  return this.droppedResponseCount;
1368
1460
  }
1461
+ /**
1462
+ * The `deadline-expired` subset of {@link getDroppedResponseCount}:
1463
+ * requests refused (by the Python worker, or by the local in-flight
1464
+ * watchdog as its backstop) because their absolute deadline had already
1465
+ * passed — dead work NOT executed (D350). Monotonic for the pool's
1466
+ * lifetime; surfaced on the memory watchdog's `pool memory` line.
1467
+ */
1468
+ getDeadlineExpiredShedCount() {
1469
+ return this.deadlineExpiredShedCount;
1470
+ }
1369
1471
  getHandle(modelIndex) {
1370
1472
  return new PoolHandle(this, modelIndex);
1371
1473
  }
@@ -1504,10 +1606,13 @@ var SharedInferencePool = class {
1504
1606
  trackDroppedResponse(result, modelIndex) {
1505
1607
  if (result["dropped"] === true) {
1506
1608
  this.droppedResponseCount++;
1609
+ if (result["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
1507
1610
  const total = this.droppedResponseCount;
1508
1611
  if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.debug("Python pool shed frame under overload", { meta: {
1509
1612
  modelIndex,
1510
1613
  droppedTotal: total,
1614
+ deadlineExpiredTotal: this.deadlineExpiredShedCount,
1615
+ ...typeof result["shedReason"] === "string" ? { shedReason: result["shedReason"] } : {},
1511
1616
  sampledEvery: SHED_LOG_SAMPLE_EVERY,
1512
1617
  runtime: this.poolRuntime,
1513
1618
  device: this.device ?? "default"
@@ -2396,11 +2501,13 @@ var EngineFactory = class {
2396
2501
  if (!this.pool) return {
2397
2502
  inFlight: 0,
2398
2503
  shed: 0,
2399
- dropped: 0
2504
+ dropped: 0,
2505
+ deadlineExpired: 0
2400
2506
  };
2401
2507
  return {
2402
2508
  ...this.pool.getBacklog(),
2403
- dropped: this.pool.getDroppedResponseCount()
2509
+ dropped: this.pool.getDroppedResponseCount(),
2510
+ deadlineExpired: this.pool.getDeadlineExpiredShedCount()
2404
2511
  };
2405
2512
  }
2406
2513
  /** Python-side per-worker memory diagnostics (see SharedInferencePool). */
@@ -2887,6 +2994,7 @@ async function samplePool(pool, log, reportedDead) {
2887
2994
  inFlight: backlog.inFlight,
2888
2995
  shed: backlog.shed,
2889
2996
  dropped: backlog.dropped,
2997
+ deadlineExpired: backlog.deadlineExpired,
2890
2998
  modelsLoaded: pool.factory.listLoaded().length,
2891
2999
  memStats
2892
3000
  }
@@ -1,10 +1,10 @@
1
1
  import { n as __require } from "../chunk-DnnnRqeS.mjs";
2
- import { Bt as nodePin, Ft as createEvent, It as hydrateSchema, Kt as array, O as YAMNET_TO_MACRO, Q as inferModelProvider, Qt as object, S as PoolMemoryWatchdog, St as supportedRuntimes$1, Vt as parseJsonUnknown, W as detectionPipelineCapability, Wt as sleep, Y as evaluateZoneRules, _t as resolvePoolMemoryPolicy, at as overlayClusterStepSettings, c as DEFAULT_CLUSTER_STEP_MODELS, ct as pickClusterStepSettings, dt as pipelineExecutorCapability, en as string, et as loadContributionCapability, ht as resolveClusterStepModelId, jt as BaseAddon, l as DEFAULT_CLUSTER_STEP_SETTINGS, m as DEVICE_BACKEND_TO_FORMAT, nn as EventCategory, q as enumerateInferenceDevices, st as pickClusterStepModels, t as APPLE_SA_TO_MACRO, tn as union, wt as errMsg, yt as runtimeDevices$1, z as defaultDeviceFor$1 } from "../dist-D7aPrnxq.mjs";
3
- import { a as readProcessCost } from "../node--gufH5K3.mjs";
4
- import { a as getDefaultModelForFormat, c as getStepDefinition, i as ALL_STEPS, l as resolveModelForFormat, o as getDefaultModelForFormatFromDef, r as ALL_PIPELINE_STEPS, s as getStep, t as localFrameRegistry, u as landmarkPrecisionVerdict } from "../local-frame-registry-CZGAexUl.mjs";
2
+ import { Bt as nodePin, Ft as createEvent, It as hydrateSchema, Kt as array, O as YAMNET_TO_MACRO, Q as inferModelProvider, Qt as object, S as PoolMemoryWatchdog, St as supportedRuntimes$1, Vt as parseJsonUnknown, W as detectionPipelineCapability, Wt as sleep, Y as evaluateZoneRules, _t as resolvePoolMemoryPolicy, at as overlayClusterStepSettings, c as DEFAULT_CLUSTER_STEP_MODELS, ct as pickClusterStepSettings, dt as pipelineExecutorCapability, en as string, et as loadContributionCapability, ht as resolveClusterStepModelId, jt as BaseAddon, l as DEFAULT_CLUSTER_STEP_SETTINGS, m as DEVICE_BACKEND_TO_FORMAT, nn as EventCategory, q as enumerateInferenceDevices, st as pickClusterStepModels, t as APPLE_SA_TO_MACRO, tn as union, wt as errMsg, yt as runtimeDevices$1, z as defaultDeviceFor$1 } from "../dist-pMHigtQu.mjs";
3
+ import { a as readProcessCost } from "../node-CmPPt-uJ.mjs";
4
+ import { a as getDefaultModelForFormat, c as getStepDefinition, i as ALL_STEPS, l as resolveModelForFormat, o as getDefaultModelForFormatFromDef, r as ALL_PIPELINE_STEPS, s as getStep, t as localFrameRegistry, u as landmarkPrecisionVerdict } from "../local-frame-registry-C7uCSrQG.mjs";
5
5
  import { t as getSharp } from "../lazy-sharp-DXsqwpph.mjs";
6
6
  import { n as startEventLoopStallMonitor } from "../event-loop-stall-monitor-DXLNBMkY.mjs";
7
- import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-D2ik9mux.mjs";
7
+ import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-BSR1vUsp.mjs";
8
8
  import * as fs from "node:fs";
9
9
  import * as path$1 from "node:path";
10
10
  import { spawn } from "node:child_process";
@@ -672,6 +672,8 @@ var MSG_INFER_CACHED = 5;
672
672
  */
673
673
  var MSG_INFER_BATCH = 3;
674
674
  var PREFIX_LEN = 9;
675
+ /** Bytes of the absolute-deadline prefix a v2 sheddable payload carries. */
676
+ var DEADLINE_PREFIX_LEN = 8;
675
677
  /**
676
678
  * Wire-level enum for the raw-frame fast path. Values are append-only:
677
679
  * the Python pool reads the byte directly off the IPC frame; reordering
@@ -860,6 +862,20 @@ var PoolWorker = class {
860
862
  * Surfaced so "the camera is quiet" and "the camera is saturated" can be
861
863
  * told apart without reading the logs. */
862
864
  shedCount = 0;
865
+ /**
866
+ * Live requests whose deadline fired IN FLIGHT and were answered dropped by
867
+ * the local watchdog (the worker never replied in time). With a v2 worker
868
+ * this is the rare backstop — the worker sheds expired work itself and
869
+ * replies fast; against a v1 worker it is the only shed there is.
870
+ */
871
+ deadlineShedCount = 0;
872
+ /**
873
+ * Negotiated wire version for THIS worker: `min(POOL_PROTOCOL_VERSION,
874
+ * what the ready handshake reported)`. A worker that reports nothing is a
875
+ * v1 worker and keeps receiving the byte-identical old layout — the deploy
876
+ * skew case that must always work (D350).
877
+ */
878
+ wireVersion = 1;
863
879
  nextRequestId = 1;
864
880
  ready = false;
865
881
  log;
@@ -929,6 +945,7 @@ var PoolWorker = class {
929
945
  const config = {
930
946
  runtime: this.opts.poolRuntime,
931
947
  concurrency: this.opts.concurrency,
948
+ protocolVersion: 2,
932
949
  models: initialModels.map((m) => serializeModelConfig(m))
933
950
  };
934
951
  if (this.opts.device) config["device"] = this.opts.device;
@@ -953,10 +970,13 @@ var PoolWorker = class {
953
970
  this.ready = true;
954
971
  const loadedCount = result["models"];
955
972
  const startupMs = result["startupMs"];
973
+ const workers = result["workers"] ?? 1;
974
+ const reported = result["protocolVersion"];
975
+ this.wireVersion = Math.min(2, typeof reported === "number" && Number.isFinite(reported) ? reported : 1);
956
976
  resolve({
957
977
  startupMs,
958
978
  loadedCount,
959
- workers: result["workers"] ?? 1
979
+ workers
960
980
  });
961
981
  } else reject(/* @__PURE__ */ new Error(`Unexpected pool status: ${JSON.stringify(result)}`));
962
982
  },
@@ -1088,19 +1108,35 @@ var PoolWorker = class {
1088
1108
  deadlineFor(msgType) {
1089
1109
  return SHEDDABLE_MSG_TYPES.has(msgType) ? POOL_LIVE_INFER_TIMEOUT_MS : POOL_INFER_TIMEOUT_MS;
1090
1110
  }
1111
+ /**
1112
+ * The request's ABSOLUTE deadline for the wire (D350), or `null` when this
1113
+ * request must not carry one: commands / model loads / cacheFrame at any
1114
+ * version, and EVERYTHING against a v1 worker — the old layout has no room
1115
+ * for the prefix, and a worker at a different version is the normal state
1116
+ * during a rolling deploy. The stamp is the same instant the local watchdog
1117
+ * arms, so "the TS side has abandoned this" and "the worker refuses to run
1118
+ * it" are one moment, not two clocks drifting apart.
1119
+ */
1120
+ deadlineStamp(msgType) {
1121
+ if (this.wireVersion < 2 || !SHEDDABLE_MSG_TYPES.has(msgType)) return null;
1122
+ const stamp = Buffer.allocUnsafe(DEADLINE_PREFIX_LEN);
1123
+ stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
1124
+ return stamp;
1125
+ }
1091
1126
  dispatch(msgType, payload, deviceId) {
1092
1127
  const shed = this.shedIfSaturated(msgType, deviceId);
1093
1128
  if (shed) return Promise.resolve(shed);
1094
1129
  const reqId = this.allocRequestId();
1095
1130
  return new Promise((resolve, reject) => {
1096
- const timer = this.armRequestTimeout(reqId, reject, this.deadlineFor(msgType), deviceId);
1131
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1097
1132
  this.pending.set(reqId, {
1098
1133
  resolve,
1099
1134
  reject,
1100
1135
  timer
1101
1136
  });
1102
1137
  try {
1103
- this.writeFrame(reqId, msgType, payload);
1138
+ const stamp = this.deadlineStamp(msgType);
1139
+ this.writeFrame(reqId, msgType, stamp ? Buffer.concat([stamp, payload]) : payload);
1104
1140
  } catch (err) {
1105
1141
  clearTimeout(timer);
1106
1142
  this.pending.delete(reqId);
@@ -1109,13 +1145,51 @@ var PoolWorker = class {
1109
1145
  });
1110
1146
  }
1111
1147
  /**
1112
- * Arm a watchdog that rejects a still-pending inference request after
1113
- * `POOL_INFER_TIMEOUT_MS`. `unref` so it never keeps the event loop alive.
1114
- * The normal response + `rejectAll` paths clear it via `PendingRequest.timer`.
1115
- */
1116
- armRequestTimeout(reqId, reject, timeoutMs, deviceId) {
1148
+ * Arm a watchdog for a still-pending request. `unref` so it never keeps the
1149
+ * event loop alive. The normal response + `rejectAll` paths clear it via
1150
+ * `PendingRequest.timer`.
1151
+ *
1152
+ * The two request classes settle DIFFERENTLY on expiry (D350):
1153
+ *
1154
+ * - A SHEDDABLE (live inference) request RESOLVES as a shed —
1155
+ * `{dropped: true, shedReason: 'deadline-expired'}` — because at this
1156
+ * point the answer is worthless whether or not it eventually arrives, and
1157
+ * a shed is flow control the executor already handles. It fires only when
1158
+ * the worker never replied at all: a v2 worker sheds expired work itself
1159
+ * (fast reply, this timer is cleared), so this is the wedged-worker
1160
+ * backstop and the v1-worker compatibility path. The 2026-09-04 storm
1161
+ * logged this state 16 146 times at ERROR; the shed is logged sampled at
1162
+ * WARN and counted (`getDeadlineExpiredShedCount`) instead.
1163
+ * - Anything else (commands, model loads) still REJECTS with the error the
1164
+ * dashboards grep for — a lost command is a fault, not flow control.
1165
+ */
1166
+ armRequestTimeout(reqId, msgType, resolve, reject, deviceId) {
1167
+ const timeoutMs = this.deadlineFor(msgType);
1117
1168
  const timer = setTimeout(() => {
1118
1169
  if (this.pending.delete(reqId)) {
1170
+ if (SHEDDABLE_MSG_TYPES.has(msgType)) {
1171
+ this.deadlineShedCount++;
1172
+ const total = this.deadlineShedCount;
1173
+ if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.warn("live inference deadline expired in flight — answered dropped", {
1174
+ ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1175
+ meta: {
1176
+ worker: this.opts.workerLabel,
1177
+ pid: this.getPid(),
1178
+ runtime: this.opts.poolRuntime,
1179
+ device: this.opts.device ?? "default",
1180
+ inFlight: this.pending.size,
1181
+ reqId,
1182
+ timeoutMs,
1183
+ deadlineShedTotal: total,
1184
+ sampledEvery: SHED_LOG_SAMPLE_EVERY
1185
+ }
1186
+ });
1187
+ resolve({
1188
+ dropped: true,
1189
+ shedReason: "deadline-expired"
1190
+ });
1191
+ return;
1192
+ }
1119
1193
  this.log.error("inference request timed out", {
1120
1194
  ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1121
1195
  meta: {
@@ -1139,7 +1213,7 @@ var PoolWorker = class {
1139
1213
  if (shed) return Promise.resolve(shed);
1140
1214
  const reqId = this.allocRequestId();
1141
1215
  return new Promise((resolve, reject) => {
1142
- const timer = this.armRequestTimeout(reqId, reject, this.deadlineFor(msgType), deviceId);
1216
+ const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
1143
1217
  this.pending.set(reqId, {
1144
1218
  resolve,
1145
1219
  reject,
@@ -1147,11 +1221,14 @@ var PoolWorker = class {
1147
1221
  });
1148
1222
  try {
1149
1223
  if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
1224
+ const stamp = this.deadlineStamp(msgType);
1225
+ const wireLen = payloadLen + (stamp ? DEADLINE_PREFIX_LEN : 0);
1150
1226
  const prefix = Buffer.allocUnsafe(PREFIX_LEN);
1151
- prefix.writeUInt32LE(5 + payloadLen, 0);
1227
+ prefix.writeUInt32LE(5 + wireLen, 0);
1152
1228
  prefix.writeUInt32LE(reqId, 4);
1153
1229
  prefix[8] = msgType;
1154
1230
  this.process.stdin.write(prefix);
1231
+ if (stamp) this.process.stdin.write(stamp);
1155
1232
  for (const part of parts) this.process.stdin.write(part);
1156
1233
  } catch (err) {
1157
1234
  clearTimeout(timer);
@@ -1254,12 +1331,22 @@ var SharedInferencePool = class {
1254
1331
  nextFreeIndex = 0;
1255
1332
  nextFrameId = 1;
1256
1333
  /**
1257
- * Cumulative count of frames the Python pool SHED under overload
1258
- * (`"dropped": true` responses from the per-model in-flight bound in
1259
- * inference_pool.py). Without this the shed response is
1260
- * indistinguishable from a genuine "no detections" result.
1334
+ * Cumulative count of dropped (shed) responses on the single-frame and
1335
+ * batch inference paths: the Python per-model bound, the Python
1336
+ * deadline-expired shed, and the TS in-flight watchdog's own
1337
+ * deadline-expired resolution all land here. Without this a shed response
1338
+ * is indistinguishable from a genuine "no detections" result.
1261
1339
  */
1262
1340
  droppedResponseCount = 0;
1341
+ /**
1342
+ * The `shedReason: 'deadline-expired'` subset of {@link
1343
+ * droppedResponseCount} — work that was ALREADY DEAD when it would have
1344
+ * run (D350). This is the number that was zero-by-construction while the
1345
+ * 2026-09-04 storm burned the accelerator on abandoned requests for 4h15m;
1346
+ * a rising value here under load is the pool refusing dead work, which is
1347
+ * the fix working, not a fault.
1348
+ */
1349
+ deadlineExpiredShedCount = 0;
1263
1350
  log;
1264
1351
  concurrency;
1265
1352
  tuning;
@@ -1345,7 +1432,12 @@ var SharedInferencePool = class {
1345
1432
  }
1346
1433
  async inferBatch(modelIndex, items, frameId = 0) {
1347
1434
  if (items.length > 255) throw new Error(`SharedInferencePool.inferBatch: max 255 items per call, got ${items.length}`);
1348
- return this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
1435
+ const results = await this.pickWorker().inferBatch(this.encodeModelByte(modelIndex), items, frameId);
1436
+ for (const item of results) if (item["dropped"] === true) {
1437
+ this.droppedResponseCount++;
1438
+ if (item["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
1439
+ }
1440
+ return results;
1349
1441
  }
1350
1442
  async inferCached(modelIndex, frameId, deviceId) {
1351
1443
  const w = this.pickWorker();
@@ -1359,6 +1451,16 @@ var SharedInferencePool = class {
1359
1451
  getDroppedResponseCount() {
1360
1452
  return this.droppedResponseCount;
1361
1453
  }
1454
+ /**
1455
+ * The `deadline-expired` subset of {@link getDroppedResponseCount}:
1456
+ * requests refused (by the Python worker, or by the local in-flight
1457
+ * watchdog as its backstop) because their absolute deadline had already
1458
+ * passed — dead work NOT executed (D350). Monotonic for the pool's
1459
+ * lifetime; surfaced on the memory watchdog's `pool memory` line.
1460
+ */
1461
+ getDeadlineExpiredShedCount() {
1462
+ return this.deadlineExpiredShedCount;
1463
+ }
1362
1464
  getHandle(modelIndex) {
1363
1465
  return new PoolHandle(this, modelIndex);
1364
1466
  }
@@ -1497,10 +1599,13 @@ var SharedInferencePool = class {
1497
1599
  trackDroppedResponse(result, modelIndex) {
1498
1600
  if (result["dropped"] === true) {
1499
1601
  this.droppedResponseCount++;
1602
+ if (result["shedReason"] === "deadline-expired") this.deadlineExpiredShedCount++;
1500
1603
  const total = this.droppedResponseCount;
1501
1604
  if (total === 1 || total % SHED_LOG_SAMPLE_EVERY === 0) this.log.debug("Python pool shed frame under overload", { meta: {
1502
1605
  modelIndex,
1503
1606
  droppedTotal: total,
1607
+ deadlineExpiredTotal: this.deadlineExpiredShedCount,
1608
+ ...typeof result["shedReason"] === "string" ? { shedReason: result["shedReason"] } : {},
1504
1609
  sampledEvery: SHED_LOG_SAMPLE_EVERY,
1505
1610
  runtime: this.poolRuntime,
1506
1611
  device: this.device ?? "default"
@@ -2389,11 +2494,13 @@ var EngineFactory = class {
2389
2494
  if (!this.pool) return {
2390
2495
  inFlight: 0,
2391
2496
  shed: 0,
2392
- dropped: 0
2497
+ dropped: 0,
2498
+ deadlineExpired: 0
2393
2499
  };
2394
2500
  return {
2395
2501
  ...this.pool.getBacklog(),
2396
- dropped: this.pool.getDroppedResponseCount()
2502
+ dropped: this.pool.getDroppedResponseCount(),
2503
+ deadlineExpired: this.pool.getDeadlineExpiredShedCount()
2397
2504
  };
2398
2505
  }
2399
2506
  /** Python-side per-worker memory diagnostics (see SharedInferencePool). */
@@ -2880,6 +2987,7 @@ async function samplePool(pool, log, reportedDead) {
2880
2987
  inFlight: backlog.inFlight,
2881
2988
  shed: backlog.shed,
2882
2989
  dropped: backlog.dropped,
2990
+ deadlineExpired: backlog.deadlineExpired,
2883
2991
  modelsLoaded: pool.factory.listLoaded().length,
2884
2992
  memStats
2885
2993
  }
@@ -14577,6 +14577,24 @@ method(_void(), _void(), { kind: "mutation" }), method(_void(), _void(), { kind:
14577
14577
  kind: "mutation",
14578
14578
  auth: "admin"
14579
14579
  });
14580
+ /**
14581
+ * Device Manager capability — hub-side singleton that unifies device persistence,
14582
+ * live registry access, and all management operations into a single tRPC surface.
14583
+ *
14584
+ * Replaces:
14585
+ * - `device-persistence` capability (persistence methods absorbed here)
14586
+ * - `device-management.router.ts` (deleted in Phase 2)
14587
+ * - `device-ops.router.ts` (compat layer — deleted; device-provider ops absorbed here)
14588
+ *
14589
+ * All device provider addons (rtsp, onvif, frigate, …) are hub-local: they may
14590
+ * fork into separate processes but never run on remote cluster agents. Therefore:
14591
+ * - No nodeId routing needed — this is a pure hub singleton.
14592
+ * - The hub's DeviceRegistry is the single source of truth for all live devices.
14593
+ * - No shadow registry or cross-node aggregation required.
14594
+ *
14595
+ * Forked workers register devices back to the hub via `ctx.devices`
14596
+ * (DeviceManagerApi → ctx.api.deviceManager.registerDevice), same as today.
14597
+ */
14580
14598
  /** One child-placement directive on a container's `childLayout`. Structurally
14581
14599
  * identical to `ChildLayoutEntry` in `device-management.ts` — the cap wire
14582
14600
  * shape for the same field. The child is identified by its re-sync-stable
@@ -20806,6 +20824,9 @@ var RetrainStatusSchema = _enum([
20806
20824
  *
20807
20825
  * `debug` does NOT pin; it is attention, not durability.
20808
20826
  *
20827
+ * `debugNote` is the operator's own words about WHAT should be checked on this
20828
+ * track, and it lives and dies with `debug` — see {@link MAX_TRACK_DEBUG_NOTE_LEN}.
20829
+ *
20809
20830
  * `favourited` IS a BOOLEAN pin (durability bit), not a retrain lifecycle.
20810
20831
  * A favourited track is skipped by retention the same way `staging` is, but
20811
20832
  * it does not enter `none|staging|trained` and has no staging budget.
@@ -20816,6 +20837,21 @@ var TrackFlagFields = {
20816
20837
  markForTrain: boolean().optional(),
20817
20838
  /** Operator marked this track for diagnostic attention. */
20818
20839
  debug: boolean().optional(),
20840
+ /**
20841
+ * What the operator wants CHECKED on this track, in their own words.
20842
+ *
20843
+ * The flag alone says "look at this" and nothing about what for; a debug set
20844
+ * reviewed hours later is a list of tracks with no question attached. Bound
20845
+ * to `debug` on the WRITE side — the body clears it when `debug` goes off,
20846
+ * and refuses it on a track whose debug is off — so a note can never outlive
20847
+ * the attention it belongs to.
20848
+ *
20849
+ * `''` is a real value ("marked, nothing to add"); the ABSENT key means
20850
+ * "leave whatever is stored alone", which is what lets a surface turn debug
20851
+ * on without a note (a cancelled prompt) and what lets it edit the note
20852
+ * without touching the flag.
20853
+ */
20854
+ debugNote: string().max(500).optional(),
20819
20855
  /** Operator favourited this track. Pins it against pruning. */
20820
20856
  favourited: boolean().optional()
20821
20857
  };
@@ -20842,6 +20878,12 @@ var TrackFlagsSchema = object({
20842
20878
  trackId: string(),
20843
20879
  markForTrain: boolean(),
20844
20880
  debug: boolean(),
20881
+ /** The note as it STANDS after the write — never absent here (a track with no
20882
+ * note reports `''`), for the same reason the booleans are required: a
20883
+ * surface that has just written must be able to render the note it now owns
20884
+ * without a re-fetch, and an absent key would read as "unchanged" on a
20885
+ * screen that has no previous value to keep. */
20886
+ debugNote: string(),
20845
20887
  favourited: boolean(),
20846
20888
  /** The lifecycle state the boolean was derived from. Required here (unlike on
20847
20889
  * a track row) because this shape is only ever produced by the write body,