@camstack/addon-pipeline 1.2.180 → 1.2.182
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -869,6 +869,8 @@ var PoolWorker = class {
|
|
|
869
869
|
* Surfaced so "the camera is quiet" and "the camera is saturated" can be
|
|
870
870
|
* told apart without reading the logs. */
|
|
871
871
|
shedCount = 0;
|
|
872
|
+
/** Requests that hit their client-side deadline (`inference request timed out`). */
|
|
873
|
+
timedOutCount = 0;
|
|
872
874
|
/**
|
|
873
875
|
* Live requests whose deadline fired IN FLIGHT and were answered dropped by
|
|
874
876
|
* the local watchdog (the worker never replied in time). With a v2 worker
|
|
@@ -903,6 +905,9 @@ var PoolWorker = class {
|
|
|
903
905
|
getShedCount() {
|
|
904
906
|
return this.shedCount;
|
|
905
907
|
}
|
|
908
|
+
getTimedOutCount() {
|
|
909
|
+
return this.timedOutCount;
|
|
910
|
+
}
|
|
906
911
|
isReady() {
|
|
907
912
|
return this.ready;
|
|
908
913
|
}
|
|
@@ -1197,6 +1202,7 @@ var PoolWorker = class {
|
|
|
1197
1202
|
});
|
|
1198
1203
|
return;
|
|
1199
1204
|
}
|
|
1205
|
+
this.timedOutCount++;
|
|
1200
1206
|
this.log.error("inference request timed out", {
|
|
1201
1207
|
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1202
1208
|
meta: {
|
|
@@ -1468,6 +1474,15 @@ var SharedInferencePool = class {
|
|
|
1468
1474
|
getDeadlineExpiredShedCount() {
|
|
1469
1475
|
return this.deadlineExpiredShedCount;
|
|
1470
1476
|
}
|
|
1477
|
+
/**
|
|
1478
|
+
* Requests that timed out client-side, summed over the live workers. A
|
|
1479
|
+
* worker restart resets its share; the inference-timeout guard treats a
|
|
1480
|
+
* drop as "the new count is the delta". This is the metric 1.2.218's
|
|
1481
|
+
* regression was invisible on for 4 h 19 min (2026-09-04).
|
|
1482
|
+
*/
|
|
1483
|
+
getTimedOutCount() {
|
|
1484
|
+
return this.workers.reduce((sum, w) => sum + w.getTimedOutCount(), 0);
|
|
1485
|
+
}
|
|
1471
1486
|
getHandle(modelIndex) {
|
|
1472
1487
|
return new PoolHandle(this, modelIndex);
|
|
1473
1488
|
}
|
|
@@ -2502,12 +2517,14 @@ var EngineFactory = class {
|
|
|
2502
2517
|
inFlight: 0,
|
|
2503
2518
|
shed: 0,
|
|
2504
2519
|
dropped: 0,
|
|
2505
|
-
deadlineExpired: 0
|
|
2520
|
+
deadlineExpired: 0,
|
|
2521
|
+
timedOut: 0
|
|
2506
2522
|
};
|
|
2507
2523
|
return {
|
|
2508
2524
|
...this.pool.getBacklog(),
|
|
2509
2525
|
dropped: this.pool.getDroppedResponseCount(),
|
|
2510
|
-
deadlineExpired: this.pool.getDeadlineExpiredShedCount()
|
|
2526
|
+
deadlineExpired: this.pool.getDeadlineExpiredShedCount(),
|
|
2527
|
+
timedOut: this.pool.getTimedOutCount()
|
|
2511
2528
|
};
|
|
2512
2529
|
}
|
|
2513
2530
|
/** Python-side per-worker memory diagnostics (see SharedInferencePool). */
|
|
@@ -3001,6 +3018,235 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
3001
3018
|
};
|
|
3002
3019
|
}
|
|
3003
3020
|
//#endregion
|
|
3021
|
+
//#region src/detection-pipeline/engine/inference-timeout-guard.ts
|
|
3022
|
+
/**
|
|
3023
|
+
* Inference-timeout regression guard — "did this update make the GPU time
|
|
3024
|
+
* out more than the last one did?", asked by the node itself.
|
|
3025
|
+
*
|
|
3026
|
+
* ## Why
|
|
3027
|
+
*
|
|
3028
|
+
* `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
|
|
3029
|
+
* fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
|
|
3030
|
+
* lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
|
|
3031
|
+
* seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
|
|
3032
|
+
* next deploy, and the regression was found a day later by an audit that
|
|
3033
|
+
* counted log lines per version by hand (2026-09-05). A green typecheck, a
|
|
3034
|
+
* green suite and a 200 on `/health` all held throughout.
|
|
3035
|
+
*
|
|
3036
|
+
* ## What it does
|
|
3037
|
+
*
|
|
3038
|
+
* Every five minutes the guard records how many inference requests timed out
|
|
3039
|
+
* on this node under the current FINGERPRINT (the closure version). The last
|
|
3040
|
+
* hour of buckets is persisted in the addon store, so a restart hands the
|
|
3041
|
+
* previous fingerprint's rate to the next one. Once the new fingerprint has a
|
|
3042
|
+
* quarter hour of its own, the two hourly rates are compared:
|
|
3043
|
+
*
|
|
3044
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
|
|
3045
|
+
* → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
|
|
3046
|
+
* on the finding id) and an ERROR line naming both versions and rates;
|
|
3047
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
|
|
3048
|
+
* - the same fingerprint (a plain restart) → nothing to compare.
|
|
3049
|
+
*
|
|
3050
|
+
* The rate is per hour, not per frame: it is the unit the audit measured and
|
|
3051
|
+
* the unit an operator reads. A traffic doubling doubles the rate, which is
|
|
3052
|
+
* why the ratio is five and not two.
|
|
3053
|
+
*
|
|
3054
|
+
* Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
|
|
3055
|
+
* `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
|
|
3056
|
+
* timer, the store and the emit.
|
|
3057
|
+
*/
|
|
3058
|
+
var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
|
|
3059
|
+
/** Addon-store key the ledger persists under (the addon's own settings blob). */
|
|
3060
|
+
var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
|
|
3061
|
+
var HOUR_MS = 60 * 6e4;
|
|
3062
|
+
function recordTimeoutBucket(ledger, bucket) {
|
|
3063
|
+
const buckets = [...ledger.buckets, bucket];
|
|
3064
|
+
return {
|
|
3065
|
+
version: ledger.version,
|
|
3066
|
+
buckets: buckets.length > 12 ? buckets.slice(buckets.length - 12) : buckets
|
|
3067
|
+
};
|
|
3068
|
+
}
|
|
3069
|
+
/** Timeouts per hour over the ledger's buckets; `null` under the minimum. */
|
|
3070
|
+
function timeoutsPerHour(ledger) {
|
|
3071
|
+
if (ledger === null || ledger.buckets.length < 3) return null;
|
|
3072
|
+
return ledger.buckets.reduce((sum, b) => sum + b.timeouts, 0) / (ledger.buckets.length * TIMEOUT_GUARD_BUCKET_MS) * HOUR_MS;
|
|
3073
|
+
}
|
|
3074
|
+
function evaluateTimeoutRegression(input) {
|
|
3075
|
+
const { previous, current } = input;
|
|
3076
|
+
const previousPerHour = timeoutsPerHour(previous);
|
|
3077
|
+
const currentPerHour = timeoutsPerHour(current);
|
|
3078
|
+
const base = {
|
|
3079
|
+
previousVersion: previous?.version ?? null,
|
|
3080
|
+
currentVersion: current.version,
|
|
3081
|
+
previousPerHour,
|
|
3082
|
+
currentPerHour
|
|
3083
|
+
};
|
|
3084
|
+
if (previous === null || previous.version === current.version || previousPerHour === null || currentPerHour === null) return {
|
|
3085
|
+
kind: "not-comparable",
|
|
3086
|
+
...base
|
|
3087
|
+
};
|
|
3088
|
+
if (currentPerHour >= 20 && currentPerHour >= Math.max(previousPerHour, 20 / 5) * 5) return {
|
|
3089
|
+
kind: "regressed",
|
|
3090
|
+
...base
|
|
3091
|
+
};
|
|
3092
|
+
if (previousPerHour >= 20 && currentPerHour * 5 <= previousPerHour) return {
|
|
3093
|
+
kind: "improved",
|
|
3094
|
+
...base
|
|
3095
|
+
};
|
|
3096
|
+
return {
|
|
3097
|
+
kind: "quiet",
|
|
3098
|
+
...base
|
|
3099
|
+
};
|
|
3100
|
+
}
|
|
3101
|
+
/**
|
|
3102
|
+
* The closure version, read from the closure the runner was started from.
|
|
3103
|
+
*
|
|
3104
|
+
* The forked runner script is `<closure>/node_modules/@camstack/system/dist/
|
|
3105
|
+
* addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
|
|
3106
|
+
* `@camstack/server/package.json` sits two directories up. This is read from
|
|
3107
|
+
* disk rather than asked of `server-management` because that cap is
|
|
3108
|
+
* server-provided and the parent refuses to route it for a forked addon —
|
|
3109
|
+
* "no provider registered for cap server-management", measured on all
|
|
3110
|
+
* three nodes on 2026-09-05. A file the process was started from cannot
|
|
3111
|
+
* be refused.
|
|
3112
|
+
*/
|
|
3113
|
+
function closureVersionFromRunnerPath(runnerScript, fs) {
|
|
3114
|
+
if (runnerScript === void 0 || runnerScript.length === 0) return null;
|
|
3115
|
+
let dir = (0, node_path.dirname)(runnerScript);
|
|
3116
|
+
for (let depth = 0; depth < 12; depth += 1) {
|
|
3117
|
+
const candidate = (0, node_path.join)(dir, "node_modules", "@camstack", "server", "package.json");
|
|
3118
|
+
const raw = fs.readFile(candidate);
|
|
3119
|
+
if (raw !== null) {
|
|
3120
|
+
try {
|
|
3121
|
+
const parsed = JSON.parse(raw);
|
|
3122
|
+
const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
|
|
3123
|
+
if (typeof version === "string" && version.length > 0) return `server@${version}`;
|
|
3124
|
+
} catch {
|
|
3125
|
+
return null;
|
|
3126
|
+
}
|
|
3127
|
+
return null;
|
|
3128
|
+
}
|
|
3129
|
+
const parent = (0, node_path.dirname)(dir);
|
|
3130
|
+
if (parent === dir) break;
|
|
3131
|
+
dir = parent;
|
|
3132
|
+
}
|
|
3133
|
+
return null;
|
|
3134
|
+
}
|
|
3135
|
+
function readStoredLedgers(blob) {
|
|
3136
|
+
const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
|
|
3137
|
+
if (typeof raw !== "object" || raw === null) return {};
|
|
3138
|
+
const r = raw;
|
|
3139
|
+
return {
|
|
3140
|
+
...isLedger(r.current) ? { current: r.current } : {},
|
|
3141
|
+
...isLedger(r.previous) ? { previous: r.previous } : {}
|
|
3142
|
+
};
|
|
3143
|
+
}
|
|
3144
|
+
function isLedger(value) {
|
|
3145
|
+
if (typeof value !== "object" || value === null) return false;
|
|
3146
|
+
const v = value;
|
|
3147
|
+
return typeof v.version === "string" && Array.isArray(v.buckets) && v.buckets.every((b) => typeof b === "object" && b !== null && typeof b.at === "number" && typeof b.timeouts === "number");
|
|
3148
|
+
}
|
|
3149
|
+
function createInferenceTimeoutGuard(deps) {
|
|
3150
|
+
const now = deps.now ?? (() => Date.now());
|
|
3151
|
+
const findingId = `inference-timeout-regression:${deps.nodeId}`;
|
|
3152
|
+
let timer = null;
|
|
3153
|
+
let current = null;
|
|
3154
|
+
let previous = null;
|
|
3155
|
+
let lastTotal = 0;
|
|
3156
|
+
let raised = false;
|
|
3157
|
+
let improvementLogged = false;
|
|
3158
|
+
const ensureLedger = async () => {
|
|
3159
|
+
if (current !== null) return true;
|
|
3160
|
+
const fingerprint = await deps.resolveFingerprint();
|
|
3161
|
+
if (fingerprint === null) return false;
|
|
3162
|
+
const stored = readStoredLedgers(await deps.readStore());
|
|
3163
|
+
if (stored.current !== void 0 && stored.current.version === fingerprint) {
|
|
3164
|
+
current = stored.current;
|
|
3165
|
+
previous = stored.previous ?? null;
|
|
3166
|
+
} else {
|
|
3167
|
+
current = {
|
|
3168
|
+
version: fingerprint,
|
|
3169
|
+
buckets: []
|
|
3170
|
+
};
|
|
3171
|
+
previous = stored.current ?? stored.previous ?? null;
|
|
3172
|
+
}
|
|
3173
|
+
lastTotal = deps.readTotalTimeouts();
|
|
3174
|
+
deps.log.info("inference timeout guard armed", { meta: {
|
|
3175
|
+
fingerprint,
|
|
3176
|
+
previousVersion: previous?.version ?? null,
|
|
3177
|
+
previousPerHour: timeoutsPerHour(previous),
|
|
3178
|
+
bucketMs: TIMEOUT_GUARD_BUCKET_MS
|
|
3179
|
+
} });
|
|
3180
|
+
return true;
|
|
3181
|
+
};
|
|
3182
|
+
const tick = async () => {
|
|
3183
|
+
try {
|
|
3184
|
+
if (!await ensureLedger() || current === null) return;
|
|
3185
|
+
const total = deps.readTotalTimeouts();
|
|
3186
|
+
const delta = total >= lastTotal ? total - lastTotal : total;
|
|
3187
|
+
lastTotal = total;
|
|
3188
|
+
current = recordTimeoutBucket(current, {
|
|
3189
|
+
at: now(),
|
|
3190
|
+
timeouts: delta
|
|
3191
|
+
});
|
|
3192
|
+
await deps.writeStore({ [INFERENCE_TIMEOUT_LEDGER_KEY]: {
|
|
3193
|
+
current,
|
|
3194
|
+
...previous !== null ? { previous } : {}
|
|
3195
|
+
} });
|
|
3196
|
+
const verdict = evaluateTimeoutRegression({
|
|
3197
|
+
previous,
|
|
3198
|
+
current
|
|
3199
|
+
});
|
|
3200
|
+
const meta = {
|
|
3201
|
+
previousVersion: verdict.previousVersion,
|
|
3202
|
+
currentVersion: verdict.currentVersion,
|
|
3203
|
+
previousPerHour: verdict.previousPerHour,
|
|
3204
|
+
currentPerHour: verdict.currentPerHour
|
|
3205
|
+
};
|
|
3206
|
+
if (verdict.kind === "regressed") {
|
|
3207
|
+
if (!raised) {
|
|
3208
|
+
raised = true;
|
|
3209
|
+
const message = `inference requests time out ${Math.round(verdict.currentPerHour ?? 0)}/h under ${verdict.currentVersion}, against ${Math.round(verdict.previousPerHour ?? 0)}/h under ${verdict.previousVersion ?? "unknown"} — the update regressed inference on ${deps.nodeId}`;
|
|
3210
|
+
deps.log.error("inference timeout REGRESSION after update", { meta });
|
|
3211
|
+
deps.emitFailed({
|
|
3212
|
+
findingId,
|
|
3213
|
+
severity: "error",
|
|
3214
|
+
title: "Inference timeouts regressed after update",
|
|
3215
|
+
message
|
|
3216
|
+
});
|
|
3217
|
+
}
|
|
3218
|
+
return;
|
|
3219
|
+
}
|
|
3220
|
+
if (raised && verdict.kind !== "not-comparable") {
|
|
3221
|
+
raised = false;
|
|
3222
|
+
deps.log.info("inference timeout regression cleared", { meta });
|
|
3223
|
+
deps.emitRecovered(findingId);
|
|
3224
|
+
}
|
|
3225
|
+
if (verdict.kind === "improved" && !improvementLogged) {
|
|
3226
|
+
improvementLogged = true;
|
|
3227
|
+
deps.log.info("inference timeouts improved after update", { meta });
|
|
3228
|
+
}
|
|
3229
|
+
} catch (err) {
|
|
3230
|
+
deps.log.warn("inference timeout guard tick failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
3231
|
+
}
|
|
3232
|
+
};
|
|
3233
|
+
return {
|
|
3234
|
+
start: () => {
|
|
3235
|
+
if (timer !== null) return;
|
|
3236
|
+
timer = setInterval(() => {
|
|
3237
|
+
tick();
|
|
3238
|
+
}, TIMEOUT_GUARD_BUCKET_MS);
|
|
3239
|
+
timer.unref?.();
|
|
3240
|
+
ensureLedger();
|
|
3241
|
+
},
|
|
3242
|
+
tick,
|
|
3243
|
+
stop: () => {
|
|
3244
|
+
if (timer !== null) clearInterval(timer);
|
|
3245
|
+
timer = null;
|
|
3246
|
+
}
|
|
3247
|
+
};
|
|
3248
|
+
}
|
|
3249
|
+
//#endregion
|
|
3004
3250
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
3005
3251
|
/** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
|
|
3006
3252
|
var BACKOFF_SCHEDULE_MS = [
|
|
@@ -6179,6 +6425,11 @@ function applyZoneRuleGate(result, zones, rules) {
|
|
|
6179
6425
|
* This is the main provider that consumers (DetectionWiring, Benchmark, tRPC)
|
|
6180
6426
|
* interact with. It manages the engine factory, pipeline executor, and config persistence.
|
|
6181
6427
|
*/
|
|
6428
|
+
/** Event source of the inference-timeout guard's liveness findings. */
|
|
6429
|
+
var TIMEOUT_GUARD_SOURCE = {
|
|
6430
|
+
type: "addon",
|
|
6431
|
+
id: "detection-pipeline"
|
|
6432
|
+
};
|
|
6182
6433
|
var KEY_TEMPLATES = "pipelineTemplates";
|
|
6183
6434
|
function pythonModuleForBackend(backend) {
|
|
6184
6435
|
switch (backend) {
|
|
@@ -6606,6 +6857,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6606
6857
|
* restart action. Started in `init()`, stopped in `shutdown()`.
|
|
6607
6858
|
*/
|
|
6608
6859
|
poolMemoryGuard = null;
|
|
6860
|
+
/** Compares this node's inference-timeout rate across closure updates. */
|
|
6861
|
+
inferenceTimeoutGuard = null;
|
|
6609
6862
|
/** Watchdog view of the live pools: the node-default pool plus every
|
|
6610
6863
|
* per-device pool. Keys are stable per pool identity so the watchdog's
|
|
6611
6864
|
* baseline tracks one pool across sweeps. */
|
|
@@ -6702,6 +6955,30 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6702
6955
|
}, this.log);
|
|
6703
6956
|
this.poolMemoryGuard.start();
|
|
6704
6957
|
}
|
|
6958
|
+
if (!this.inferenceTimeoutGuard) {
|
|
6959
|
+
const nodeId = this.localProbeNodeId();
|
|
6960
|
+
this.inferenceTimeoutGuard = createInferenceTimeoutGuard({
|
|
6961
|
+
readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
|
|
6962
|
+
readStore: () => this.readStore(),
|
|
6963
|
+
writeStore: (patch) => this.writeStore(patch),
|
|
6964
|
+
resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
|
|
6965
|
+
try {
|
|
6966
|
+
return node_fs.readFileSync(path, "utf8");
|
|
6967
|
+
} catch {
|
|
6968
|
+
return null;
|
|
6969
|
+
}
|
|
6970
|
+
} }),
|
|
6971
|
+
emitFailed: (finding) => {
|
|
6972
|
+
this.eventBus?.emit(require_dist.createEvent(require_dist.EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
|
|
6973
|
+
},
|
|
6974
|
+
emitRecovered: (findingId) => {
|
|
6975
|
+
this.eventBus?.emit(require_dist.createEvent(require_dist.EventCategory.SystemLivenessRecovered, TIMEOUT_GUARD_SOURCE, { findingId }));
|
|
6976
|
+
},
|
|
6977
|
+
nodeId,
|
|
6978
|
+
log: this.log.child("inference-timeout-guard")
|
|
6979
|
+
});
|
|
6980
|
+
this.inferenceTimeoutGuard.start();
|
|
6981
|
+
}
|
|
6705
6982
|
}
|
|
6706
6983
|
/**
|
|
6707
6984
|
* Lazy-install the pip requirements file matching `engine.backend` into
|
|
@@ -8623,6 +8900,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8623
8900
|
async shutdown() {
|
|
8624
8901
|
this.poolMemoryGuard?.stop();
|
|
8625
8902
|
this.poolMemoryGuard = null;
|
|
8903
|
+
this.inferenceTimeoutGuard?.stop();
|
|
8904
|
+
this.inferenceTimeoutGuard = null;
|
|
8626
8905
|
await this.evictOverrideCache("shutdown");
|
|
8627
8906
|
if (this.engineFactory) {
|
|
8628
8907
|
await this.engineFactory.dispose();
|
|
@@ -7,6 +7,7 @@ import { n as startEventLoopStallMonitor } from "../event-loop-stall-monitor-DXL
|
|
|
7
7
|
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-zLBrzuFc.mjs";
|
|
8
8
|
import * as fs from "node:fs";
|
|
9
9
|
import * as path$1 from "node:path";
|
|
10
|
+
import { dirname, join } from "node:path";
|
|
10
11
|
import { spawn } from "node:child_process";
|
|
11
12
|
import sharp from "sharp";
|
|
12
13
|
import * as os from "node:os";
|
|
@@ -862,6 +863,8 @@ var PoolWorker = class {
|
|
|
862
863
|
* Surfaced so "the camera is quiet" and "the camera is saturated" can be
|
|
863
864
|
* told apart without reading the logs. */
|
|
864
865
|
shedCount = 0;
|
|
866
|
+
/** Requests that hit their client-side deadline (`inference request timed out`). */
|
|
867
|
+
timedOutCount = 0;
|
|
865
868
|
/**
|
|
866
869
|
* Live requests whose deadline fired IN FLIGHT and were answered dropped by
|
|
867
870
|
* the local watchdog (the worker never replied in time). With a v2 worker
|
|
@@ -896,6 +899,9 @@ var PoolWorker = class {
|
|
|
896
899
|
getShedCount() {
|
|
897
900
|
return this.shedCount;
|
|
898
901
|
}
|
|
902
|
+
getTimedOutCount() {
|
|
903
|
+
return this.timedOutCount;
|
|
904
|
+
}
|
|
899
905
|
isReady() {
|
|
900
906
|
return this.ready;
|
|
901
907
|
}
|
|
@@ -1190,6 +1196,7 @@ var PoolWorker = class {
|
|
|
1190
1196
|
});
|
|
1191
1197
|
return;
|
|
1192
1198
|
}
|
|
1199
|
+
this.timedOutCount++;
|
|
1193
1200
|
this.log.error("inference request timed out", {
|
|
1194
1201
|
...deviceId !== void 0 ? { tags: { deviceId } } : {},
|
|
1195
1202
|
meta: {
|
|
@@ -1461,6 +1468,15 @@ var SharedInferencePool = class {
|
|
|
1461
1468
|
getDeadlineExpiredShedCount() {
|
|
1462
1469
|
return this.deadlineExpiredShedCount;
|
|
1463
1470
|
}
|
|
1471
|
+
/**
|
|
1472
|
+
* Requests that timed out client-side, summed over the live workers. A
|
|
1473
|
+
* worker restart resets its share; the inference-timeout guard treats a
|
|
1474
|
+
* drop as "the new count is the delta". This is the metric 1.2.218's
|
|
1475
|
+
* regression was invisible on for 4 h 19 min (2026-09-04).
|
|
1476
|
+
*/
|
|
1477
|
+
getTimedOutCount() {
|
|
1478
|
+
return this.workers.reduce((sum, w) => sum + w.getTimedOutCount(), 0);
|
|
1479
|
+
}
|
|
1464
1480
|
getHandle(modelIndex) {
|
|
1465
1481
|
return new PoolHandle(this, modelIndex);
|
|
1466
1482
|
}
|
|
@@ -2495,12 +2511,14 @@ var EngineFactory = class {
|
|
|
2495
2511
|
inFlight: 0,
|
|
2496
2512
|
shed: 0,
|
|
2497
2513
|
dropped: 0,
|
|
2498
|
-
deadlineExpired: 0
|
|
2514
|
+
deadlineExpired: 0,
|
|
2515
|
+
timedOut: 0
|
|
2499
2516
|
};
|
|
2500
2517
|
return {
|
|
2501
2518
|
...this.pool.getBacklog(),
|
|
2502
2519
|
dropped: this.pool.getDroppedResponseCount(),
|
|
2503
|
-
deadlineExpired: this.pool.getDeadlineExpiredShedCount()
|
|
2520
|
+
deadlineExpired: this.pool.getDeadlineExpiredShedCount(),
|
|
2521
|
+
timedOut: this.pool.getTimedOutCount()
|
|
2504
2522
|
};
|
|
2505
2523
|
}
|
|
2506
2524
|
/** Python-side per-worker memory diagnostics (see SharedInferencePool). */
|
|
@@ -2994,6 +3012,235 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
2994
3012
|
};
|
|
2995
3013
|
}
|
|
2996
3014
|
//#endregion
|
|
3015
|
+
//#region src/detection-pipeline/engine/inference-timeout-guard.ts
|
|
3016
|
+
/**
|
|
3017
|
+
* Inference-timeout regression guard — "did this update make the GPU time
|
|
3018
|
+
* out more than the last one did?", asked by the node itself.
|
|
3019
|
+
*
|
|
3020
|
+
* ## Why
|
|
3021
|
+
*
|
|
3022
|
+
* `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
|
|
3023
|
+
* fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
|
|
3024
|
+
* lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
|
|
3025
|
+
* seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
|
|
3026
|
+
* next deploy, and the regression was found a day later by an audit that
|
|
3027
|
+
* counted log lines per version by hand (2026-09-05). A green typecheck, a
|
|
3028
|
+
* green suite and a 200 on `/health` all held throughout.
|
|
3029
|
+
*
|
|
3030
|
+
* ## What it does
|
|
3031
|
+
*
|
|
3032
|
+
* Every five minutes the guard records how many inference requests timed out
|
|
3033
|
+
* on this node under the current FINGERPRINT (the closure version). The last
|
|
3034
|
+
* hour of buckets is persisted in the addon store, so a restart hands the
|
|
3035
|
+
* previous fingerprint's rate to the next one. Once the new fingerprint has a
|
|
3036
|
+
* quarter hour of its own, the two hourly rates are compared:
|
|
3037
|
+
*
|
|
3038
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
|
|
3039
|
+
* → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
|
|
3040
|
+
* on the finding id) and an ERROR line naming both versions and rates;
|
|
3041
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
|
|
3042
|
+
* - the same fingerprint (a plain restart) → nothing to compare.
|
|
3043
|
+
*
|
|
3044
|
+
* The rate is per hour, not per frame: it is the unit the audit measured and
|
|
3045
|
+
* the unit an operator reads. A traffic doubling doubles the rate, which is
|
|
3046
|
+
* why the ratio is five and not two.
|
|
3047
|
+
*
|
|
3048
|
+
* Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
|
|
3049
|
+
* `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
|
|
3050
|
+
* timer, the store and the emit.
|
|
3051
|
+
*/
|
|
3052
|
+
var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
|
|
3053
|
+
/** Addon-store key the ledger persists under (the addon's own settings blob). */
|
|
3054
|
+
var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
|
|
3055
|
+
var HOUR_MS = 60 * 6e4;
|
|
3056
|
+
function recordTimeoutBucket(ledger, bucket) {
|
|
3057
|
+
const buckets = [...ledger.buckets, bucket];
|
|
3058
|
+
return {
|
|
3059
|
+
version: ledger.version,
|
|
3060
|
+
buckets: buckets.length > 12 ? buckets.slice(buckets.length - 12) : buckets
|
|
3061
|
+
};
|
|
3062
|
+
}
|
|
3063
|
+
/** Timeouts per hour over the ledger's buckets; `null` under the minimum. */
|
|
3064
|
+
function timeoutsPerHour(ledger) {
|
|
3065
|
+
if (ledger === null || ledger.buckets.length < 3) return null;
|
|
3066
|
+
return ledger.buckets.reduce((sum, b) => sum + b.timeouts, 0) / (ledger.buckets.length * TIMEOUT_GUARD_BUCKET_MS) * HOUR_MS;
|
|
3067
|
+
}
|
|
3068
|
+
function evaluateTimeoutRegression(input) {
|
|
3069
|
+
const { previous, current } = input;
|
|
3070
|
+
const previousPerHour = timeoutsPerHour(previous);
|
|
3071
|
+
const currentPerHour = timeoutsPerHour(current);
|
|
3072
|
+
const base = {
|
|
3073
|
+
previousVersion: previous?.version ?? null,
|
|
3074
|
+
currentVersion: current.version,
|
|
3075
|
+
previousPerHour,
|
|
3076
|
+
currentPerHour
|
|
3077
|
+
};
|
|
3078
|
+
if (previous === null || previous.version === current.version || previousPerHour === null || currentPerHour === null) return {
|
|
3079
|
+
kind: "not-comparable",
|
|
3080
|
+
...base
|
|
3081
|
+
};
|
|
3082
|
+
if (currentPerHour >= 20 && currentPerHour >= Math.max(previousPerHour, 20 / 5) * 5) return {
|
|
3083
|
+
kind: "regressed",
|
|
3084
|
+
...base
|
|
3085
|
+
};
|
|
3086
|
+
if (previousPerHour >= 20 && currentPerHour * 5 <= previousPerHour) return {
|
|
3087
|
+
kind: "improved",
|
|
3088
|
+
...base
|
|
3089
|
+
};
|
|
3090
|
+
return {
|
|
3091
|
+
kind: "quiet",
|
|
3092
|
+
...base
|
|
3093
|
+
};
|
|
3094
|
+
}
|
|
3095
|
+
/**
|
|
3096
|
+
* The closure version, read from the closure the runner was started from.
|
|
3097
|
+
*
|
|
3098
|
+
* The forked runner script is `<closure>/node_modules/@camstack/system/dist/
|
|
3099
|
+
* addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
|
|
3100
|
+
* `@camstack/server/package.json` sits two directories up. This is read from
|
|
3101
|
+
* disk rather than asked of `server-management` because that cap is
|
|
3102
|
+
* server-provided and the parent refuses to route it for a forked addon —
|
|
3103
|
+
* "no provider registered for cap server-management", measured on all
|
|
3104
|
+
* three nodes on 2026-09-05. A file the process was started from cannot
|
|
3105
|
+
* be refused.
|
|
3106
|
+
*/
|
|
3107
|
+
function closureVersionFromRunnerPath(runnerScript, fs) {
|
|
3108
|
+
if (runnerScript === void 0 || runnerScript.length === 0) return null;
|
|
3109
|
+
let dir = dirname(runnerScript);
|
|
3110
|
+
for (let depth = 0; depth < 12; depth += 1) {
|
|
3111
|
+
const candidate = join(dir, "node_modules", "@camstack", "server", "package.json");
|
|
3112
|
+
const raw = fs.readFile(candidate);
|
|
3113
|
+
if (raw !== null) {
|
|
3114
|
+
try {
|
|
3115
|
+
const parsed = JSON.parse(raw);
|
|
3116
|
+
const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
|
|
3117
|
+
if (typeof version === "string" && version.length > 0) return `server@${version}`;
|
|
3118
|
+
} catch {
|
|
3119
|
+
return null;
|
|
3120
|
+
}
|
|
3121
|
+
return null;
|
|
3122
|
+
}
|
|
3123
|
+
const parent = dirname(dir);
|
|
3124
|
+
if (parent === dir) break;
|
|
3125
|
+
dir = parent;
|
|
3126
|
+
}
|
|
3127
|
+
return null;
|
|
3128
|
+
}
|
|
3129
|
+
function readStoredLedgers(blob) {
|
|
3130
|
+
const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
|
|
3131
|
+
if (typeof raw !== "object" || raw === null) return {};
|
|
3132
|
+
const r = raw;
|
|
3133
|
+
return {
|
|
3134
|
+
...isLedger(r.current) ? { current: r.current } : {},
|
|
3135
|
+
...isLedger(r.previous) ? { previous: r.previous } : {}
|
|
3136
|
+
};
|
|
3137
|
+
}
|
|
3138
|
+
function isLedger(value) {
|
|
3139
|
+
if (typeof value !== "object" || value === null) return false;
|
|
3140
|
+
const v = value;
|
|
3141
|
+
return typeof v.version === "string" && Array.isArray(v.buckets) && v.buckets.every((b) => typeof b === "object" && b !== null && typeof b.at === "number" && typeof b.timeouts === "number");
|
|
3142
|
+
}
|
|
3143
|
+
function createInferenceTimeoutGuard(deps) {
|
|
3144
|
+
const now = deps.now ?? (() => Date.now());
|
|
3145
|
+
const findingId = `inference-timeout-regression:${deps.nodeId}`;
|
|
3146
|
+
let timer = null;
|
|
3147
|
+
let current = null;
|
|
3148
|
+
let previous = null;
|
|
3149
|
+
let lastTotal = 0;
|
|
3150
|
+
let raised = false;
|
|
3151
|
+
let improvementLogged = false;
|
|
3152
|
+
const ensureLedger = async () => {
|
|
3153
|
+
if (current !== null) return true;
|
|
3154
|
+
const fingerprint = await deps.resolveFingerprint();
|
|
3155
|
+
if (fingerprint === null) return false;
|
|
3156
|
+
const stored = readStoredLedgers(await deps.readStore());
|
|
3157
|
+
if (stored.current !== void 0 && stored.current.version === fingerprint) {
|
|
3158
|
+
current = stored.current;
|
|
3159
|
+
previous = stored.previous ?? null;
|
|
3160
|
+
} else {
|
|
3161
|
+
current = {
|
|
3162
|
+
version: fingerprint,
|
|
3163
|
+
buckets: []
|
|
3164
|
+
};
|
|
3165
|
+
previous = stored.current ?? stored.previous ?? null;
|
|
3166
|
+
}
|
|
3167
|
+
lastTotal = deps.readTotalTimeouts();
|
|
3168
|
+
deps.log.info("inference timeout guard armed", { meta: {
|
|
3169
|
+
fingerprint,
|
|
3170
|
+
previousVersion: previous?.version ?? null,
|
|
3171
|
+
previousPerHour: timeoutsPerHour(previous),
|
|
3172
|
+
bucketMs: TIMEOUT_GUARD_BUCKET_MS
|
|
3173
|
+
} });
|
|
3174
|
+
return true;
|
|
3175
|
+
};
|
|
3176
|
+
const tick = async () => {
|
|
3177
|
+
try {
|
|
3178
|
+
if (!await ensureLedger() || current === null) return;
|
|
3179
|
+
const total = deps.readTotalTimeouts();
|
|
3180
|
+
const delta = total >= lastTotal ? total - lastTotal : total;
|
|
3181
|
+
lastTotal = total;
|
|
3182
|
+
current = recordTimeoutBucket(current, {
|
|
3183
|
+
at: now(),
|
|
3184
|
+
timeouts: delta
|
|
3185
|
+
});
|
|
3186
|
+
await deps.writeStore({ [INFERENCE_TIMEOUT_LEDGER_KEY]: {
|
|
3187
|
+
current,
|
|
3188
|
+
...previous !== null ? { previous } : {}
|
|
3189
|
+
} });
|
|
3190
|
+
const verdict = evaluateTimeoutRegression({
|
|
3191
|
+
previous,
|
|
3192
|
+
current
|
|
3193
|
+
});
|
|
3194
|
+
const meta = {
|
|
3195
|
+
previousVersion: verdict.previousVersion,
|
|
3196
|
+
currentVersion: verdict.currentVersion,
|
|
3197
|
+
previousPerHour: verdict.previousPerHour,
|
|
3198
|
+
currentPerHour: verdict.currentPerHour
|
|
3199
|
+
};
|
|
3200
|
+
if (verdict.kind === "regressed") {
|
|
3201
|
+
if (!raised) {
|
|
3202
|
+
raised = true;
|
|
3203
|
+
const message = `inference requests time out ${Math.round(verdict.currentPerHour ?? 0)}/h under ${verdict.currentVersion}, against ${Math.round(verdict.previousPerHour ?? 0)}/h under ${verdict.previousVersion ?? "unknown"} — the update regressed inference on ${deps.nodeId}`;
|
|
3204
|
+
deps.log.error("inference timeout REGRESSION after update", { meta });
|
|
3205
|
+
deps.emitFailed({
|
|
3206
|
+
findingId,
|
|
3207
|
+
severity: "error",
|
|
3208
|
+
title: "Inference timeouts regressed after update",
|
|
3209
|
+
message
|
|
3210
|
+
});
|
|
3211
|
+
}
|
|
3212
|
+
return;
|
|
3213
|
+
}
|
|
3214
|
+
if (raised && verdict.kind !== "not-comparable") {
|
|
3215
|
+
raised = false;
|
|
3216
|
+
deps.log.info("inference timeout regression cleared", { meta });
|
|
3217
|
+
deps.emitRecovered(findingId);
|
|
3218
|
+
}
|
|
3219
|
+
if (verdict.kind === "improved" && !improvementLogged) {
|
|
3220
|
+
improvementLogged = true;
|
|
3221
|
+
deps.log.info("inference timeouts improved after update", { meta });
|
|
3222
|
+
}
|
|
3223
|
+
} catch (err) {
|
|
3224
|
+
deps.log.warn("inference timeout guard tick failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
3225
|
+
}
|
|
3226
|
+
};
|
|
3227
|
+
return {
|
|
3228
|
+
start: () => {
|
|
3229
|
+
if (timer !== null) return;
|
|
3230
|
+
timer = setInterval(() => {
|
|
3231
|
+
tick();
|
|
3232
|
+
}, TIMEOUT_GUARD_BUCKET_MS);
|
|
3233
|
+
timer.unref?.();
|
|
3234
|
+
ensureLedger();
|
|
3235
|
+
},
|
|
3236
|
+
tick,
|
|
3237
|
+
stop: () => {
|
|
3238
|
+
if (timer !== null) clearInterval(timer);
|
|
3239
|
+
timer = null;
|
|
3240
|
+
}
|
|
3241
|
+
};
|
|
3242
|
+
}
|
|
3243
|
+
//#endregion
|
|
2997
3244
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
2998
3245
|
/** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
|
|
2999
3246
|
var BACKOFF_SCHEDULE_MS = [
|
|
@@ -6172,6 +6419,11 @@ function applyZoneRuleGate(result, zones, rules) {
|
|
|
6172
6419
|
* This is the main provider that consumers (DetectionWiring, Benchmark, tRPC)
|
|
6173
6420
|
* interact with. It manages the engine factory, pipeline executor, and config persistence.
|
|
6174
6421
|
*/
|
|
6422
|
+
/** Event source of the inference-timeout guard's liveness findings. */
|
|
6423
|
+
var TIMEOUT_GUARD_SOURCE = {
|
|
6424
|
+
type: "addon",
|
|
6425
|
+
id: "detection-pipeline"
|
|
6426
|
+
};
|
|
6175
6427
|
var KEY_TEMPLATES = "pipelineTemplates";
|
|
6176
6428
|
function pythonModuleForBackend(backend) {
|
|
6177
6429
|
switch (backend) {
|
|
@@ -6599,6 +6851,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6599
6851
|
* restart action. Started in `init()`, stopped in `shutdown()`.
|
|
6600
6852
|
*/
|
|
6601
6853
|
poolMemoryGuard = null;
|
|
6854
|
+
/** Compares this node's inference-timeout rate across closure updates. */
|
|
6855
|
+
inferenceTimeoutGuard = null;
|
|
6602
6856
|
/** Watchdog view of the live pools: the node-default pool plus every
|
|
6603
6857
|
* per-device pool. Keys are stable per pool identity so the watchdog's
|
|
6604
6858
|
* baseline tracks one pool across sweeps. */
|
|
@@ -6695,6 +6949,30 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6695
6949
|
}, this.log);
|
|
6696
6950
|
this.poolMemoryGuard.start();
|
|
6697
6951
|
}
|
|
6952
|
+
if (!this.inferenceTimeoutGuard) {
|
|
6953
|
+
const nodeId = this.localProbeNodeId();
|
|
6954
|
+
this.inferenceTimeoutGuard = createInferenceTimeoutGuard({
|
|
6955
|
+
readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
|
|
6956
|
+
readStore: () => this.readStore(),
|
|
6957
|
+
writeStore: (patch) => this.writeStore(patch),
|
|
6958
|
+
resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
|
|
6959
|
+
try {
|
|
6960
|
+
return fs.readFileSync(path, "utf8");
|
|
6961
|
+
} catch {
|
|
6962
|
+
return null;
|
|
6963
|
+
}
|
|
6964
|
+
} }),
|
|
6965
|
+
emitFailed: (finding) => {
|
|
6966
|
+
this.eventBus?.emit(createEvent(EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
|
|
6967
|
+
},
|
|
6968
|
+
emitRecovered: (findingId) => {
|
|
6969
|
+
this.eventBus?.emit(createEvent(EventCategory.SystemLivenessRecovered, TIMEOUT_GUARD_SOURCE, { findingId }));
|
|
6970
|
+
},
|
|
6971
|
+
nodeId,
|
|
6972
|
+
log: this.log.child("inference-timeout-guard")
|
|
6973
|
+
});
|
|
6974
|
+
this.inferenceTimeoutGuard.start();
|
|
6975
|
+
}
|
|
6698
6976
|
}
|
|
6699
6977
|
/**
|
|
6700
6978
|
* Lazy-install the pip requirements file matching `engine.backend` into
|
|
@@ -8616,6 +8894,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8616
8894
|
async shutdown() {
|
|
8617
8895
|
this.poolMemoryGuard?.stop();
|
|
8618
8896
|
this.poolMemoryGuard = null;
|
|
8897
|
+
this.inferenceTimeoutGuard?.stop();
|
|
8898
|
+
this.inferenceTimeoutGuard = null;
|
|
8619
8899
|
await this.evictOverrideCache("shutdown");
|
|
8620
8900
|
if (this.engineFactory) {
|
|
8621
8901
|
await this.engineFactory.dispose();
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@camstack/addon-pipeline",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.182",
|
|
4
4
|
"description": "Pipeline bundle — runner, detection, motion, audio + stream broker. Multi-entry npm package shipping pipeline addons under a single bundle.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"camstack",
|