@camstack/addon-pipeline 1.2.180 → 1.2.182

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -869,6 +869,8 @@ var PoolWorker = class {
869
869
  * Surfaced so "the camera is quiet" and "the camera is saturated" can be
870
870
  * told apart without reading the logs. */
871
871
  shedCount = 0;
872
+ /** Requests that hit their client-side deadline (`inference request timed out`). */
873
+ timedOutCount = 0;
872
874
  /**
873
875
  * Live requests whose deadline fired IN FLIGHT and were answered dropped by
874
876
  * the local watchdog (the worker never replied in time). With a v2 worker
@@ -903,6 +905,9 @@ var PoolWorker = class {
903
905
  getShedCount() {
904
906
  return this.shedCount;
905
907
  }
908
+ getTimedOutCount() {
909
+ return this.timedOutCount;
910
+ }
906
911
  isReady() {
907
912
  return this.ready;
908
913
  }
@@ -1197,6 +1202,7 @@ var PoolWorker = class {
1197
1202
  });
1198
1203
  return;
1199
1204
  }
1205
+ this.timedOutCount++;
1200
1206
  this.log.error("inference request timed out", {
1201
1207
  ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1202
1208
  meta: {
@@ -1468,6 +1474,15 @@ var SharedInferencePool = class {
1468
1474
  getDeadlineExpiredShedCount() {
1469
1475
  return this.deadlineExpiredShedCount;
1470
1476
  }
1477
+ /**
1478
+ * Requests that timed out client-side, summed over the live workers. A
1479
+ * worker restart resets its share; the inference-timeout guard treats a
1480
+ * drop as "the new count is the delta". This is the metric 1.2.218's
1481
+ * regression was invisible on for 4 h 19 min (2026-09-04).
1482
+ */
1483
+ getTimedOutCount() {
1484
+ return this.workers.reduce((sum, w) => sum + w.getTimedOutCount(), 0);
1485
+ }
1471
1486
  getHandle(modelIndex) {
1472
1487
  return new PoolHandle(this, modelIndex);
1473
1488
  }
@@ -2502,12 +2517,14 @@ var EngineFactory = class {
2502
2517
  inFlight: 0,
2503
2518
  shed: 0,
2504
2519
  dropped: 0,
2505
- deadlineExpired: 0
2520
+ deadlineExpired: 0,
2521
+ timedOut: 0
2506
2522
  };
2507
2523
  return {
2508
2524
  ...this.pool.getBacklog(),
2509
2525
  dropped: this.pool.getDroppedResponseCount(),
2510
- deadlineExpired: this.pool.getDeadlineExpiredShedCount()
2526
+ deadlineExpired: this.pool.getDeadlineExpiredShedCount(),
2527
+ timedOut: this.pool.getTimedOutCount()
2511
2528
  };
2512
2529
  }
2513
2530
  /** Python-side per-worker memory diagnostics (see SharedInferencePool). */
@@ -3001,6 +3018,235 @@ async function samplePool(pool, log, reportedDead) {
3001
3018
  };
3002
3019
  }
3003
3020
  //#endregion
3021
+ //#region src/detection-pipeline/engine/inference-timeout-guard.ts
3022
+ /**
3023
+ * Inference-timeout regression guard — "did this update make the GPU time
3024
+ * out more than the last one did?", asked by the node itself.
3025
+ *
3026
+ * ## Why
3027
+ *
3028
+ * `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
3029
+ * fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
3030
+ * lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
3031
+ * seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
3032
+ * next deploy, and the regression was found a day later by an audit that
3033
+ * counted log lines per version by hand (2026-09-05). A green typecheck, a
3034
+ * green suite and a 200 on `/health` all held throughout.
3035
+ *
3036
+ * ## What it does
3037
+ *
3038
+ * Every five minutes the guard records how many inference requests timed out
3039
+ * on this node under the current FINGERPRINT (the closure version). The last
3040
+ * hour of buckets is persisted in the addon store, so a restart hands the
3041
+ * previous fingerprint's rate to the next one. Once the new fingerprint has a
3042
+ * quarter hour of its own, the two hourly rates are compared:
3043
+ *
3044
+ * - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
3045
+ * → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
3046
+ * on the finding id) and an ERROR line naming both versions and rates;
3047
+ * - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
3048
+ * - the same fingerprint (a plain restart) → nothing to compare.
3049
+ *
3050
+ * The rate is per hour, not per frame: it is the unit the audit measured and
3051
+ * the unit an operator reads. A traffic doubling doubles the rate, which is
3052
+ * why the ratio is five and not two.
3053
+ *
3054
+ * Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
3055
+ * `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
3056
+ * timer, the store and the emit.
3057
+ */
3058
+ var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
3059
+ /** Addon-store key the ledger persists under (the addon's own settings blob). */
3060
+ var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
3061
+ var HOUR_MS = 60 * 6e4;
3062
+ function recordTimeoutBucket(ledger, bucket) {
3063
+ const buckets = [...ledger.buckets, bucket];
3064
+ return {
3065
+ version: ledger.version,
3066
+ buckets: buckets.length > 12 ? buckets.slice(buckets.length - 12) : buckets
3067
+ };
3068
+ }
3069
+ /** Timeouts per hour over the ledger's buckets; `null` under the minimum. */
3070
+ function timeoutsPerHour(ledger) {
3071
+ if (ledger === null || ledger.buckets.length < 3) return null;
3072
+ return ledger.buckets.reduce((sum, b) => sum + b.timeouts, 0) / (ledger.buckets.length * TIMEOUT_GUARD_BUCKET_MS) * HOUR_MS;
3073
+ }
3074
+ function evaluateTimeoutRegression(input) {
3075
+ const { previous, current } = input;
3076
+ const previousPerHour = timeoutsPerHour(previous);
3077
+ const currentPerHour = timeoutsPerHour(current);
3078
+ const base = {
3079
+ previousVersion: previous?.version ?? null,
3080
+ currentVersion: current.version,
3081
+ previousPerHour,
3082
+ currentPerHour
3083
+ };
3084
+ if (previous === null || previous.version === current.version || previousPerHour === null || currentPerHour === null) return {
3085
+ kind: "not-comparable",
3086
+ ...base
3087
+ };
3088
+ if (currentPerHour >= 20 && currentPerHour >= Math.max(previousPerHour, 20 / 5) * 5) return {
3089
+ kind: "regressed",
3090
+ ...base
3091
+ };
3092
+ if (previousPerHour >= 20 && currentPerHour * 5 <= previousPerHour) return {
3093
+ kind: "improved",
3094
+ ...base
3095
+ };
3096
+ return {
3097
+ kind: "quiet",
3098
+ ...base
3099
+ };
3100
+ }
3101
+ /**
3102
+ * The closure version, read from the closure the runner was started from.
3103
+ *
3104
+ * The forked runner script is `<closure>/node_modules/@camstack/system/dist/
3105
+ * addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
3106
+ * `@camstack/server/package.json` sits two directories up. This is read from
3107
+ * disk rather than asked of `server-management` because that cap is
3108
+ * server-provided and the parent refuses to route it for a forked addon —
3109
+ * "no provider registered for cap server-management", measured on all
3110
+ * three nodes on 2026-09-05. A file the process was started from cannot
3111
+ * be refused.
3112
+ */
3113
+ function closureVersionFromRunnerPath(runnerScript, fs) {
3114
+ if (runnerScript === void 0 || runnerScript.length === 0) return null;
3115
+ let dir = (0, node_path.dirname)(runnerScript);
3116
+ for (let depth = 0; depth < 12; depth += 1) {
3117
+ const candidate = (0, node_path.join)(dir, "node_modules", "@camstack", "server", "package.json");
3118
+ const raw = fs.readFile(candidate);
3119
+ if (raw !== null) {
3120
+ try {
3121
+ const parsed = JSON.parse(raw);
3122
+ const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
3123
+ if (typeof version === "string" && version.length > 0) return `server@${version}`;
3124
+ } catch {
3125
+ return null;
3126
+ }
3127
+ return null;
3128
+ }
3129
+ const parent = (0, node_path.dirname)(dir);
3130
+ if (parent === dir) break;
3131
+ dir = parent;
3132
+ }
3133
+ return null;
3134
+ }
3135
+ function readStoredLedgers(blob) {
3136
+ const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
3137
+ if (typeof raw !== "object" || raw === null) return {};
3138
+ const r = raw;
3139
+ return {
3140
+ ...isLedger(r.current) ? { current: r.current } : {},
3141
+ ...isLedger(r.previous) ? { previous: r.previous } : {}
3142
+ };
3143
+ }
3144
+ function isLedger(value) {
3145
+ if (typeof value !== "object" || value === null) return false;
3146
+ const v = value;
3147
+ return typeof v.version === "string" && Array.isArray(v.buckets) && v.buckets.every((b) => typeof b === "object" && b !== null && typeof b.at === "number" && typeof b.timeouts === "number");
3148
+ }
3149
+ function createInferenceTimeoutGuard(deps) {
3150
+ const now = deps.now ?? (() => Date.now());
3151
+ const findingId = `inference-timeout-regression:${deps.nodeId}`;
3152
+ let timer = null;
3153
+ let current = null;
3154
+ let previous = null;
3155
+ let lastTotal = 0;
3156
+ let raised = false;
3157
+ let improvementLogged = false;
3158
+ const ensureLedger = async () => {
3159
+ if (current !== null) return true;
3160
+ const fingerprint = await deps.resolveFingerprint();
3161
+ if (fingerprint === null) return false;
3162
+ const stored = readStoredLedgers(await deps.readStore());
3163
+ if (stored.current !== void 0 && stored.current.version === fingerprint) {
3164
+ current = stored.current;
3165
+ previous = stored.previous ?? null;
3166
+ } else {
3167
+ current = {
3168
+ version: fingerprint,
3169
+ buckets: []
3170
+ };
3171
+ previous = stored.current ?? stored.previous ?? null;
3172
+ }
3173
+ lastTotal = deps.readTotalTimeouts();
3174
+ deps.log.info("inference timeout guard armed", { meta: {
3175
+ fingerprint,
3176
+ previousVersion: previous?.version ?? null,
3177
+ previousPerHour: timeoutsPerHour(previous),
3178
+ bucketMs: TIMEOUT_GUARD_BUCKET_MS
3179
+ } });
3180
+ return true;
3181
+ };
3182
+ const tick = async () => {
3183
+ try {
3184
+ if (!await ensureLedger() || current === null) return;
3185
+ const total = deps.readTotalTimeouts();
3186
+ const delta = total >= lastTotal ? total - lastTotal : total;
3187
+ lastTotal = total;
3188
+ current = recordTimeoutBucket(current, {
3189
+ at: now(),
3190
+ timeouts: delta
3191
+ });
3192
+ await deps.writeStore({ [INFERENCE_TIMEOUT_LEDGER_KEY]: {
3193
+ current,
3194
+ ...previous !== null ? { previous } : {}
3195
+ } });
3196
+ const verdict = evaluateTimeoutRegression({
3197
+ previous,
3198
+ current
3199
+ });
3200
+ const meta = {
3201
+ previousVersion: verdict.previousVersion,
3202
+ currentVersion: verdict.currentVersion,
3203
+ previousPerHour: verdict.previousPerHour,
3204
+ currentPerHour: verdict.currentPerHour
3205
+ };
3206
+ if (verdict.kind === "regressed") {
3207
+ if (!raised) {
3208
+ raised = true;
3209
+ const message = `inference requests time out ${Math.round(verdict.currentPerHour ?? 0)}/h under ${verdict.currentVersion}, against ${Math.round(verdict.previousPerHour ?? 0)}/h under ${verdict.previousVersion ?? "unknown"} — the update regressed inference on ${deps.nodeId}`;
3210
+ deps.log.error("inference timeout REGRESSION after update", { meta });
3211
+ deps.emitFailed({
3212
+ findingId,
3213
+ severity: "error",
3214
+ title: "Inference timeouts regressed after update",
3215
+ message
3216
+ });
3217
+ }
3218
+ return;
3219
+ }
3220
+ if (raised && verdict.kind !== "not-comparable") {
3221
+ raised = false;
3222
+ deps.log.info("inference timeout regression cleared", { meta });
3223
+ deps.emitRecovered(findingId);
3224
+ }
3225
+ if (verdict.kind === "improved" && !improvementLogged) {
3226
+ improvementLogged = true;
3227
+ deps.log.info("inference timeouts improved after update", { meta });
3228
+ }
3229
+ } catch (err) {
3230
+ deps.log.warn("inference timeout guard tick failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
3231
+ }
3232
+ };
3233
+ return {
3234
+ start: () => {
3235
+ if (timer !== null) return;
3236
+ timer = setInterval(() => {
3237
+ tick();
3238
+ }, TIMEOUT_GUARD_BUCKET_MS);
3239
+ timer.unref?.();
3240
+ ensureLedger();
3241
+ },
3242
+ tick,
3243
+ stop: () => {
3244
+ if (timer !== null) clearInterval(timer);
3245
+ timer = null;
3246
+ }
3247
+ };
3248
+ }
3249
+ //#endregion
3004
3250
  //#region src/detection-pipeline/engine-provisioner.ts
3005
3251
  /** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
3006
3252
  var BACKOFF_SCHEDULE_MS = [
@@ -6179,6 +6425,11 @@ function applyZoneRuleGate(result, zones, rules) {
6179
6425
  * This is the main provider that consumers (DetectionWiring, Benchmark, tRPC)
6180
6426
  * interact with. It manages the engine factory, pipeline executor, and config persistence.
6181
6427
  */
6428
+ /** Event source of the inference-timeout guard's liveness findings. */
6429
+ var TIMEOUT_GUARD_SOURCE = {
6430
+ type: "addon",
6431
+ id: "detection-pipeline"
6432
+ };
6182
6433
  var KEY_TEMPLATES = "pipelineTemplates";
6183
6434
  function pythonModuleForBackend(backend) {
6184
6435
  switch (backend) {
@@ -6606,6 +6857,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
6606
6857
  * restart action. Started in `init()`, stopped in `shutdown()`.
6607
6858
  */
6608
6859
  poolMemoryGuard = null;
6860
+ /** Compares this node's inference-timeout rate across closure updates. */
6861
+ inferenceTimeoutGuard = null;
6609
6862
  /** Watchdog view of the live pools: the node-default pool plus every
6610
6863
  * per-device pool. Keys are stable per pool identity so the watchdog's
6611
6864
  * baseline tracks one pool across sweeps. */
@@ -6702,6 +6955,30 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
6702
6955
  }, this.log);
6703
6956
  this.poolMemoryGuard.start();
6704
6957
  }
6958
+ if (!this.inferenceTimeoutGuard) {
6959
+ const nodeId = this.localProbeNodeId();
6960
+ this.inferenceTimeoutGuard = createInferenceTimeoutGuard({
6961
+ readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
6962
+ readStore: () => this.readStore(),
6963
+ writeStore: (patch) => this.writeStore(patch),
6964
+ resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
6965
+ try {
6966
+ return node_fs.readFileSync(path, "utf8");
6967
+ } catch {
6968
+ return null;
6969
+ }
6970
+ } }),
6971
+ emitFailed: (finding) => {
6972
+ this.eventBus?.emit(require_dist.createEvent(require_dist.EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
6973
+ },
6974
+ emitRecovered: (findingId) => {
6975
+ this.eventBus?.emit(require_dist.createEvent(require_dist.EventCategory.SystemLivenessRecovered, TIMEOUT_GUARD_SOURCE, { findingId }));
6976
+ },
6977
+ nodeId,
6978
+ log: this.log.child("inference-timeout-guard")
6979
+ });
6980
+ this.inferenceTimeoutGuard.start();
6981
+ }
6705
6982
  }
6706
6983
  /**
6707
6984
  * Lazy-install the pip requirements file matching `engine.backend` into
@@ -8623,6 +8900,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8623
8900
  async shutdown() {
8624
8901
  this.poolMemoryGuard?.stop();
8625
8902
  this.poolMemoryGuard = null;
8903
+ this.inferenceTimeoutGuard?.stop();
8904
+ this.inferenceTimeoutGuard = null;
8626
8905
  await this.evictOverrideCache("shutdown");
8627
8906
  if (this.engineFactory) {
8628
8907
  await this.engineFactory.dispose();
@@ -7,6 +7,7 @@ import { n as startEventLoopStallMonitor } from "../event-loop-stall-monitor-DXL
7
7
  import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-zLBrzuFc.mjs";
8
8
  import * as fs from "node:fs";
9
9
  import * as path$1 from "node:path";
10
+ import { dirname, join } from "node:path";
10
11
  import { spawn } from "node:child_process";
11
12
  import sharp from "sharp";
12
13
  import * as os from "node:os";
@@ -862,6 +863,8 @@ var PoolWorker = class {
862
863
  * Surfaced so "the camera is quiet" and "the camera is saturated" can be
863
864
  * told apart without reading the logs. */
864
865
  shedCount = 0;
866
+ /** Requests that hit their client-side deadline (`inference request timed out`). */
867
+ timedOutCount = 0;
865
868
  /**
866
869
  * Live requests whose deadline fired IN FLIGHT and were answered dropped by
867
870
  * the local watchdog (the worker never replied in time). With a v2 worker
@@ -896,6 +899,9 @@ var PoolWorker = class {
896
899
  getShedCount() {
897
900
  return this.shedCount;
898
901
  }
902
+ getTimedOutCount() {
903
+ return this.timedOutCount;
904
+ }
899
905
  isReady() {
900
906
  return this.ready;
901
907
  }
@@ -1190,6 +1196,7 @@ var PoolWorker = class {
1190
1196
  });
1191
1197
  return;
1192
1198
  }
1199
+ this.timedOutCount++;
1193
1200
  this.log.error("inference request timed out", {
1194
1201
  ...deviceId !== void 0 ? { tags: { deviceId } } : {},
1195
1202
  meta: {
@@ -1461,6 +1468,15 @@ var SharedInferencePool = class {
1461
1468
  getDeadlineExpiredShedCount() {
1462
1469
  return this.deadlineExpiredShedCount;
1463
1470
  }
1471
+ /**
1472
+ * Requests that timed out client-side, summed over the live workers. A
1473
+ * worker restart resets its share; the inference-timeout guard treats a
1474
+ * drop as "the new count is the delta". This is the metric 1.2.218's
1475
+ * regression was invisible on for 4 h 19 min (2026-09-04).
1476
+ */
1477
+ getTimedOutCount() {
1478
+ return this.workers.reduce((sum, w) => sum + w.getTimedOutCount(), 0);
1479
+ }
1464
1480
  getHandle(modelIndex) {
1465
1481
  return new PoolHandle(this, modelIndex);
1466
1482
  }
@@ -2495,12 +2511,14 @@ var EngineFactory = class {
2495
2511
  inFlight: 0,
2496
2512
  shed: 0,
2497
2513
  dropped: 0,
2498
- deadlineExpired: 0
2514
+ deadlineExpired: 0,
2515
+ timedOut: 0
2499
2516
  };
2500
2517
  return {
2501
2518
  ...this.pool.getBacklog(),
2502
2519
  dropped: this.pool.getDroppedResponseCount(),
2503
- deadlineExpired: this.pool.getDeadlineExpiredShedCount()
2520
+ deadlineExpired: this.pool.getDeadlineExpiredShedCount(),
2521
+ timedOut: this.pool.getTimedOutCount()
2504
2522
  };
2505
2523
  }
2506
2524
  /** Python-side per-worker memory diagnostics (see SharedInferencePool). */
@@ -2994,6 +3012,235 @@ async function samplePool(pool, log, reportedDead) {
2994
3012
  };
2995
3013
  }
2996
3014
  //#endregion
3015
+ //#region src/detection-pipeline/engine/inference-timeout-guard.ts
3016
+ /**
3017
+ * Inference-timeout regression guard — "did this update make the GPU time
3018
+ * out more than the last one did?", asked by the node itself.
3019
+ *
3020
+ * ## Why
3021
+ *
3022
+ * `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
3023
+ * fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
3024
+ * lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
3025
+ * seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
3026
+ * next deploy, and the regression was found a day later by an audit that
3027
+ * counted log lines per version by hand (2026-09-05). A green typecheck, a
3028
+ * green suite and a 200 on `/health` all held throughout.
3029
+ *
3030
+ * ## What it does
3031
+ *
3032
+ * Every five minutes the guard records how many inference requests timed out
3033
+ * on this node under the current FINGERPRINT (the closure version). The last
3034
+ * hour of buckets is persisted in the addon store, so a restart hands the
3035
+ * previous fingerprint's rate to the next one. Once the new fingerprint has a
3036
+ * quarter hour of its own, the two hourly rates are compared:
3037
+ *
3038
+ * - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
3039
+ * → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
3040
+ * on the finding id) and an ERROR line naming both versions and rates;
3041
+ * - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
3042
+ * - the same fingerprint (a plain restart) → nothing to compare.
3043
+ *
3044
+ * The rate is per hour, not per frame: it is the unit the audit measured and
3045
+ * the unit an operator reads. A traffic doubling doubles the rate, which is
3046
+ * why the ratio is five and not two.
3047
+ *
3048
+ * Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
3049
+ * `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
3050
+ * timer, the store and the emit.
3051
+ */
3052
+ var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
3053
+ /** Addon-store key the ledger persists under (the addon's own settings blob). */
3054
+ var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
3055
+ var HOUR_MS = 60 * 6e4;
3056
+ function recordTimeoutBucket(ledger, bucket) {
3057
+ const buckets = [...ledger.buckets, bucket];
3058
+ return {
3059
+ version: ledger.version,
3060
+ buckets: buckets.length > 12 ? buckets.slice(buckets.length - 12) : buckets
3061
+ };
3062
+ }
3063
+ /** Timeouts per hour over the ledger's buckets; `null` under the minimum. */
3064
+ function timeoutsPerHour(ledger) {
3065
+ if (ledger === null || ledger.buckets.length < 3) return null;
3066
+ return ledger.buckets.reduce((sum, b) => sum + b.timeouts, 0) / (ledger.buckets.length * TIMEOUT_GUARD_BUCKET_MS) * HOUR_MS;
3067
+ }
3068
+ function evaluateTimeoutRegression(input) {
3069
+ const { previous, current } = input;
3070
+ const previousPerHour = timeoutsPerHour(previous);
3071
+ const currentPerHour = timeoutsPerHour(current);
3072
+ const base = {
3073
+ previousVersion: previous?.version ?? null,
3074
+ currentVersion: current.version,
3075
+ previousPerHour,
3076
+ currentPerHour
3077
+ };
3078
+ if (previous === null || previous.version === current.version || previousPerHour === null || currentPerHour === null) return {
3079
+ kind: "not-comparable",
3080
+ ...base
3081
+ };
3082
+ if (currentPerHour >= 20 && currentPerHour >= Math.max(previousPerHour, 20 / 5) * 5) return {
3083
+ kind: "regressed",
3084
+ ...base
3085
+ };
3086
+ if (previousPerHour >= 20 && currentPerHour * 5 <= previousPerHour) return {
3087
+ kind: "improved",
3088
+ ...base
3089
+ };
3090
+ return {
3091
+ kind: "quiet",
3092
+ ...base
3093
+ };
3094
+ }
3095
+ /**
3096
+ * The closure version, read from the closure the runner was started from.
3097
+ *
3098
+ * The forked runner script is `<closure>/node_modules/@camstack/system/dist/
3099
+ * addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
3100
+ * `@camstack/server/package.json` sits two directories up. This is read from
3101
+ * disk rather than asked of `server-management` because that cap is
3102
+ * server-provided and the parent refuses to route it for a forked addon —
3103
+ * "no provider registered for cap server-management", measured on all
3104
+ * three nodes on 2026-09-05. A file the process was started from cannot
3105
+ * be refused.
3106
+ */
3107
+ function closureVersionFromRunnerPath(runnerScript, fs) {
3108
+ if (runnerScript === void 0 || runnerScript.length === 0) return null;
3109
+ let dir = dirname(runnerScript);
3110
+ for (let depth = 0; depth < 12; depth += 1) {
3111
+ const candidate = join(dir, "node_modules", "@camstack", "server", "package.json");
3112
+ const raw = fs.readFile(candidate);
3113
+ if (raw !== null) {
3114
+ try {
3115
+ const parsed = JSON.parse(raw);
3116
+ const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
3117
+ if (typeof version === "string" && version.length > 0) return `server@${version}`;
3118
+ } catch {
3119
+ return null;
3120
+ }
3121
+ return null;
3122
+ }
3123
+ const parent = dirname(dir);
3124
+ if (parent === dir) break;
3125
+ dir = parent;
3126
+ }
3127
+ return null;
3128
+ }
3129
+ function readStoredLedgers(blob) {
3130
+ const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
3131
+ if (typeof raw !== "object" || raw === null) return {};
3132
+ const r = raw;
3133
+ return {
3134
+ ...isLedger(r.current) ? { current: r.current } : {},
3135
+ ...isLedger(r.previous) ? { previous: r.previous } : {}
3136
+ };
3137
+ }
3138
+ function isLedger(value) {
3139
+ if (typeof value !== "object" || value === null) return false;
3140
+ const v = value;
3141
+ return typeof v.version === "string" && Array.isArray(v.buckets) && v.buckets.every((b) => typeof b === "object" && b !== null && typeof b.at === "number" && typeof b.timeouts === "number");
3142
+ }
3143
+ function createInferenceTimeoutGuard(deps) {
3144
+ const now = deps.now ?? (() => Date.now());
3145
+ const findingId = `inference-timeout-regression:${deps.nodeId}`;
3146
+ let timer = null;
3147
+ let current = null;
3148
+ let previous = null;
3149
+ let lastTotal = 0;
3150
+ let raised = false;
3151
+ let improvementLogged = false;
3152
+ const ensureLedger = async () => {
3153
+ if (current !== null) return true;
3154
+ const fingerprint = await deps.resolveFingerprint();
3155
+ if (fingerprint === null) return false;
3156
+ const stored = readStoredLedgers(await deps.readStore());
3157
+ if (stored.current !== void 0 && stored.current.version === fingerprint) {
3158
+ current = stored.current;
3159
+ previous = stored.previous ?? null;
3160
+ } else {
3161
+ current = {
3162
+ version: fingerprint,
3163
+ buckets: []
3164
+ };
3165
+ previous = stored.current ?? stored.previous ?? null;
3166
+ }
3167
+ lastTotal = deps.readTotalTimeouts();
3168
+ deps.log.info("inference timeout guard armed", { meta: {
3169
+ fingerprint,
3170
+ previousVersion: previous?.version ?? null,
3171
+ previousPerHour: timeoutsPerHour(previous),
3172
+ bucketMs: TIMEOUT_GUARD_BUCKET_MS
3173
+ } });
3174
+ return true;
3175
+ };
3176
+ const tick = async () => {
3177
+ try {
3178
+ if (!await ensureLedger() || current === null) return;
3179
+ const total = deps.readTotalTimeouts();
3180
+ const delta = total >= lastTotal ? total - lastTotal : total;
3181
+ lastTotal = total;
3182
+ current = recordTimeoutBucket(current, {
3183
+ at: now(),
3184
+ timeouts: delta
3185
+ });
3186
+ await deps.writeStore({ [INFERENCE_TIMEOUT_LEDGER_KEY]: {
3187
+ current,
3188
+ ...previous !== null ? { previous } : {}
3189
+ } });
3190
+ const verdict = evaluateTimeoutRegression({
3191
+ previous,
3192
+ current
3193
+ });
3194
+ const meta = {
3195
+ previousVersion: verdict.previousVersion,
3196
+ currentVersion: verdict.currentVersion,
3197
+ previousPerHour: verdict.previousPerHour,
3198
+ currentPerHour: verdict.currentPerHour
3199
+ };
3200
+ if (verdict.kind === "regressed") {
3201
+ if (!raised) {
3202
+ raised = true;
3203
+ const message = `inference requests time out ${Math.round(verdict.currentPerHour ?? 0)}/h under ${verdict.currentVersion}, against ${Math.round(verdict.previousPerHour ?? 0)}/h under ${verdict.previousVersion ?? "unknown"} — the update regressed inference on ${deps.nodeId}`;
3204
+ deps.log.error("inference timeout REGRESSION after update", { meta });
3205
+ deps.emitFailed({
3206
+ findingId,
3207
+ severity: "error",
3208
+ title: "Inference timeouts regressed after update",
3209
+ message
3210
+ });
3211
+ }
3212
+ return;
3213
+ }
3214
+ if (raised && verdict.kind !== "not-comparable") {
3215
+ raised = false;
3216
+ deps.log.info("inference timeout regression cleared", { meta });
3217
+ deps.emitRecovered(findingId);
3218
+ }
3219
+ if (verdict.kind === "improved" && !improvementLogged) {
3220
+ improvementLogged = true;
3221
+ deps.log.info("inference timeouts improved after update", { meta });
3222
+ }
3223
+ } catch (err) {
3224
+ deps.log.warn("inference timeout guard tick failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
3225
+ }
3226
+ };
3227
+ return {
3228
+ start: () => {
3229
+ if (timer !== null) return;
3230
+ timer = setInterval(() => {
3231
+ tick();
3232
+ }, TIMEOUT_GUARD_BUCKET_MS);
3233
+ timer.unref?.();
3234
+ ensureLedger();
3235
+ },
3236
+ tick,
3237
+ stop: () => {
3238
+ if (timer !== null) clearInterval(timer);
3239
+ timer = null;
3240
+ }
3241
+ };
3242
+ }
3243
+ //#endregion
2997
3244
  //#region src/detection-pipeline/engine-provisioner.ts
2998
3245
  /** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
2999
3246
  var BACKOFF_SCHEDULE_MS = [
@@ -6172,6 +6419,11 @@ function applyZoneRuleGate(result, zones, rules) {
6172
6419
  * This is the main provider that consumers (DetectionWiring, Benchmark, tRPC)
6173
6420
  * interact with. It manages the engine factory, pipeline executor, and config persistence.
6174
6421
  */
6422
+ /** Event source of the inference-timeout guard's liveness findings. */
6423
+ var TIMEOUT_GUARD_SOURCE = {
6424
+ type: "addon",
6425
+ id: "detection-pipeline"
6426
+ };
6175
6427
  var KEY_TEMPLATES = "pipelineTemplates";
6176
6428
  function pythonModuleForBackend(backend) {
6177
6429
  switch (backend) {
@@ -6599,6 +6851,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
6599
6851
  * restart action. Started in `init()`, stopped in `shutdown()`.
6600
6852
  */
6601
6853
  poolMemoryGuard = null;
6854
+ /** Compares this node's inference-timeout rate across closure updates. */
6855
+ inferenceTimeoutGuard = null;
6602
6856
  /** Watchdog view of the live pools: the node-default pool plus every
6603
6857
  * per-device pool. Keys are stable per pool identity so the watchdog's
6604
6858
  * baseline tracks one pool across sweeps. */
@@ -6695,6 +6949,30 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
6695
6949
  }, this.log);
6696
6950
  this.poolMemoryGuard.start();
6697
6951
  }
6952
+ if (!this.inferenceTimeoutGuard) {
6953
+ const nodeId = this.localProbeNodeId();
6954
+ this.inferenceTimeoutGuard = createInferenceTimeoutGuard({
6955
+ readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
6956
+ readStore: () => this.readStore(),
6957
+ writeStore: (patch) => this.writeStore(patch),
6958
+ resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
6959
+ try {
6960
+ return fs.readFileSync(path, "utf8");
6961
+ } catch {
6962
+ return null;
6963
+ }
6964
+ } }),
6965
+ emitFailed: (finding) => {
6966
+ this.eventBus?.emit(createEvent(EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
6967
+ },
6968
+ emitRecovered: (findingId) => {
6969
+ this.eventBus?.emit(createEvent(EventCategory.SystemLivenessRecovered, TIMEOUT_GUARD_SOURCE, { findingId }));
6970
+ },
6971
+ nodeId,
6972
+ log: this.log.child("inference-timeout-guard")
6973
+ });
6974
+ this.inferenceTimeoutGuard.start();
6975
+ }
6698
6976
  }
6699
6977
  /**
6700
6978
  * Lazy-install the pip requirements file matching `engine.backend` into
@@ -8616,6 +8894,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
8616
8894
  async shutdown() {
8617
8895
  this.poolMemoryGuard?.stop();
8618
8896
  this.poolMemoryGuard = null;
8897
+ this.inferenceTimeoutGuard?.stop();
8898
+ this.inferenceTimeoutGuard = null;
8619
8899
  await this.evictOverrideCache("shutdown");
8620
8900
  if (this.engineFactory) {
8621
8901
  await this.engineFactory.dispose();
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-pipeline",
3
- "version": "1.2.180",
3
+ "version": "1.2.182",
4
4
  "description": "Pipeline bundle — runner, detection, motion, audio + stream broker. Multi-entry npm package shipping pipeline addons under a single bundle.",
5
5
  "keywords": [
6
6
  "camstack",