@camstack/addon-pipeline 1.2.181 → 1.2.182
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -3019,6 +3019,42 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
3019
3019
|
}
|
|
3020
3020
|
//#endregion
|
|
3021
3021
|
//#region src/detection-pipeline/engine/inference-timeout-guard.ts
|
|
3022
|
+
/**
|
|
3023
|
+
* Inference-timeout regression guard — "did this update make the GPU time
|
|
3024
|
+
* out more than the last one did?", asked by the node itself.
|
|
3025
|
+
*
|
|
3026
|
+
* ## Why
|
|
3027
|
+
*
|
|
3028
|
+
* `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
|
|
3029
|
+
* fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
|
|
3030
|
+
* lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
|
|
3031
|
+
* seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
|
|
3032
|
+
* next deploy, and the regression was found a day later by an audit that
|
|
3033
|
+
* counted log lines per version by hand (2026-09-05). A green typecheck, a
|
|
3034
|
+
* green suite and a 200 on `/health` all held throughout.
|
|
3035
|
+
*
|
|
3036
|
+
* ## What it does
|
|
3037
|
+
*
|
|
3038
|
+
* Every five minutes the guard records how many inference requests timed out
|
|
3039
|
+
* on this node under the current FINGERPRINT (the closure version). The last
|
|
3040
|
+
* hour of buckets is persisted in the addon store, so a restart hands the
|
|
3041
|
+
* previous fingerprint's rate to the next one. Once the new fingerprint has a
|
|
3042
|
+
* quarter hour of its own, the two hourly rates are compared:
|
|
3043
|
+
*
|
|
3044
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
|
|
3045
|
+
* → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
|
|
3046
|
+
* on the finding id) and an ERROR line naming both versions and rates;
|
|
3047
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
|
|
3048
|
+
* - the same fingerprint (a plain restart) → nothing to compare.
|
|
3049
|
+
*
|
|
3050
|
+
* The rate is per hour, not per frame: it is the unit the audit measured and
|
|
3051
|
+
* the unit an operator reads. A traffic doubling doubles the rate, which is
|
|
3052
|
+
* why the ratio is five and not two.
|
|
3053
|
+
*
|
|
3054
|
+
* Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
|
|
3055
|
+
* `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
|
|
3056
|
+
* timer, the store and the emit.
|
|
3057
|
+
*/
|
|
3022
3058
|
var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
|
|
3023
3059
|
/** Addon-store key the ledger persists under (the addon's own settings blob). */
|
|
3024
3060
|
var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
|
|
@@ -3062,6 +3098,40 @@ function evaluateTimeoutRegression(input) {
|
|
|
3062
3098
|
...base
|
|
3063
3099
|
};
|
|
3064
3100
|
}
|
|
3101
|
+
/**
|
|
3102
|
+
* The closure version, read from the closure the runner was started from.
|
|
3103
|
+
*
|
|
3104
|
+
* The forked runner script is `<closure>/node_modules/@camstack/system/dist/
|
|
3105
|
+
* addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
|
|
3106
|
+
* `@camstack/server/package.json` sits two directories up. This is read from
|
|
3107
|
+
* disk rather than asked of `server-management` because that cap is
|
|
3108
|
+
* server-provided and the parent refuses to route it for a forked addon —
|
|
3109
|
+
* "no provider registered for cap server-management", measured on all
|
|
3110
|
+
* three nodes on 2026-09-05. A file the process was started from cannot
|
|
3111
|
+
* be refused.
|
|
3112
|
+
*/
|
|
3113
|
+
function closureVersionFromRunnerPath(runnerScript, fs) {
|
|
3114
|
+
if (runnerScript === void 0 || runnerScript.length === 0) return null;
|
|
3115
|
+
let dir = (0, node_path.dirname)(runnerScript);
|
|
3116
|
+
for (let depth = 0; depth < 12; depth += 1) {
|
|
3117
|
+
const candidate = (0, node_path.join)(dir, "node_modules", "@camstack", "server", "package.json");
|
|
3118
|
+
const raw = fs.readFile(candidate);
|
|
3119
|
+
if (raw !== null) {
|
|
3120
|
+
try {
|
|
3121
|
+
const parsed = JSON.parse(raw);
|
|
3122
|
+
const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
|
|
3123
|
+
if (typeof version === "string" && version.length > 0) return `server@${version}`;
|
|
3124
|
+
} catch {
|
|
3125
|
+
return null;
|
|
3126
|
+
}
|
|
3127
|
+
return null;
|
|
3128
|
+
}
|
|
3129
|
+
const parent = (0, node_path.dirname)(dir);
|
|
3130
|
+
if (parent === dir) break;
|
|
3131
|
+
dir = parent;
|
|
3132
|
+
}
|
|
3133
|
+
return null;
|
|
3134
|
+
}
|
|
3065
3135
|
function readStoredLedgers(blob) {
|
|
3066
3136
|
const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
|
|
3067
3137
|
if (typeof raw !== "object" || raw === null) return {};
|
|
@@ -6891,12 +6961,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6891
6961
|
readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
|
|
6892
6962
|
readStore: () => this.readStore(),
|
|
6893
6963
|
writeStore: (patch) => this.writeStore(patch),
|
|
6894
|
-
resolveFingerprint: async () => {
|
|
6895
|
-
|
|
6896
|
-
|
|
6897
|
-
|
|
6898
|
-
|
|
6899
|
-
|
|
6964
|
+
resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
|
|
6965
|
+
try {
|
|
6966
|
+
return node_fs.readFileSync(path, "utf8");
|
|
6967
|
+
} catch {
|
|
6968
|
+
return null;
|
|
6969
|
+
}
|
|
6970
|
+
} }),
|
|
6900
6971
|
emitFailed: (finding) => {
|
|
6901
6972
|
this.eventBus?.emit(require_dist.createEvent(require_dist.EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
|
|
6902
6973
|
},
|
|
@@ -7,6 +7,7 @@ import { n as startEventLoopStallMonitor } from "../event-loop-stall-monitor-DXL
|
|
|
7
7
|
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-zLBrzuFc.mjs";
|
|
8
8
|
import * as fs from "node:fs";
|
|
9
9
|
import * as path$1 from "node:path";
|
|
10
|
+
import { dirname, join } from "node:path";
|
|
10
11
|
import { spawn } from "node:child_process";
|
|
11
12
|
import sharp from "sharp";
|
|
12
13
|
import * as os from "node:os";
|
|
@@ -3012,6 +3013,42 @@ async function samplePool(pool, log, reportedDead) {
|
|
|
3012
3013
|
}
|
|
3013
3014
|
//#endregion
|
|
3014
3015
|
//#region src/detection-pipeline/engine/inference-timeout-guard.ts
|
|
3016
|
+
/**
|
|
3017
|
+
* Inference-timeout regression guard — "did this update make the GPU time
|
|
3018
|
+
* out more than the last one did?", asked by the node itself.
|
|
3019
|
+
*
|
|
3020
|
+
* ## Why
|
|
3021
|
+
*
|
|
3022
|
+
* `@camstack/server` 1.2.218 reintroduced the inference timeouts 1.2.216 had
|
|
3023
|
+
* fixed: ~3 760 `inference request timed out` an hour for 4 h 19 min (16 215
|
|
3024
|
+
* lines, ~19 000 `runInference failed`), on a cluster whose previous hour had
|
|
3025
|
+
* seen ~5. Nothing said so. 1.2.219 fixed it again as a side effect of the
|
|
3026
|
+
* next deploy, and the regression was found a day later by an audit that
|
|
3027
|
+
* counted log lines per version by hand (2026-09-05). A green typecheck, a
|
|
3028
|
+
* green suite and a 200 on `/health` all held throughout.
|
|
3029
|
+
*
|
|
3030
|
+
* ## What it does
|
|
3031
|
+
*
|
|
3032
|
+
* Every five minutes the guard records how many inference requests timed out
|
|
3033
|
+
* on this node under the current FINGERPRINT (the closure version). The last
|
|
3034
|
+
* hour of buckets is persisted in the addon store, so a restart hands the
|
|
3035
|
+
* previous fingerprint's rate to the next one. Once the new fingerprint has a
|
|
3036
|
+
* quarter hour of its own, the two hourly rates are compared:
|
|
3037
|
+
*
|
|
3038
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× worse AND above `TIMEOUT_GUARD_FLOOR_PER_HOUR`
|
|
3039
|
+
* → `system.liveness-failed` (AlertCenter raises a persistent alert keyed
|
|
3040
|
+
* on the finding id) and an ERROR line naming both versions and rates;
|
|
3041
|
+
* - ≥ `TIMEOUT_GUARD_RATIO`× better → an INFO line, no alert;
|
|
3042
|
+
* - the same fingerprint (a plain restart) → nothing to compare.
|
|
3043
|
+
*
|
|
3044
|
+
* The rate is per hour, not per frame: it is the unit the audit measured and
|
|
3045
|
+
* the unit an operator reads. A traffic doubling doubles the rate, which is
|
|
3046
|
+
* why the ratio is five and not two.
|
|
3047
|
+
*
|
|
3048
|
+
* Pure core (`recordTimeoutBucket`, `evaluateTimeoutRegression`) tested in
|
|
3049
|
+
* `__tests__/inference-timeout-guard.spec.ts`; the runtime wrapper owns the
|
|
3050
|
+
* timer, the store and the emit.
|
|
3051
|
+
*/
|
|
3015
3052
|
var TIMEOUT_GUARD_BUCKET_MS = 5 * 6e4;
|
|
3016
3053
|
/** Addon-store key the ledger persists under (the addon's own settings blob). */
|
|
3017
3054
|
var INFERENCE_TIMEOUT_LEDGER_KEY = "_inferenceTimeoutLedger";
|
|
@@ -3055,6 +3092,40 @@ function evaluateTimeoutRegression(input) {
|
|
|
3055
3092
|
...base
|
|
3056
3093
|
};
|
|
3057
3094
|
}
|
|
3095
|
+
/**
|
|
3096
|
+
* The closure version, read from the closure the runner was started from.
|
|
3097
|
+
*
|
|
3098
|
+
* The forked runner script is `<closure>/node_modules/@camstack/system/dist/
|
|
3099
|
+
* addon-runner.js` (hub: `/data/server-root/current/…`); the closure's own
|
|
3100
|
+
* `@camstack/server/package.json` sits two directories up. This is read from
|
|
3101
|
+
* disk rather than asked of `server-management` because that cap is
|
|
3102
|
+
* server-provided and the parent refuses to route it for a forked addon —
|
|
3103
|
+
* "no provider registered for cap server-management", measured on all
|
|
3104
|
+
* three nodes on 2026-09-05. A file the process was started from cannot
|
|
3105
|
+
* be refused.
|
|
3106
|
+
*/
|
|
3107
|
+
function closureVersionFromRunnerPath(runnerScript, fs) {
|
|
3108
|
+
if (runnerScript === void 0 || runnerScript.length === 0) return null;
|
|
3109
|
+
let dir = dirname(runnerScript);
|
|
3110
|
+
for (let depth = 0; depth < 12; depth += 1) {
|
|
3111
|
+
const candidate = join(dir, "node_modules", "@camstack", "server", "package.json");
|
|
3112
|
+
const raw = fs.readFile(candidate);
|
|
3113
|
+
if (raw !== null) {
|
|
3114
|
+
try {
|
|
3115
|
+
const parsed = JSON.parse(raw);
|
|
3116
|
+
const version = typeof parsed === "object" && parsed !== null ? parsed.version : void 0;
|
|
3117
|
+
if (typeof version === "string" && version.length > 0) return `server@${version}`;
|
|
3118
|
+
} catch {
|
|
3119
|
+
return null;
|
|
3120
|
+
}
|
|
3121
|
+
return null;
|
|
3122
|
+
}
|
|
3123
|
+
const parent = dirname(dir);
|
|
3124
|
+
if (parent === dir) break;
|
|
3125
|
+
dir = parent;
|
|
3126
|
+
}
|
|
3127
|
+
return null;
|
|
3128
|
+
}
|
|
3058
3129
|
function readStoredLedgers(blob) {
|
|
3059
3130
|
const raw = blob[INFERENCE_TIMEOUT_LEDGER_KEY];
|
|
3060
3131
|
if (typeof raw !== "object" || raw === null) return {};
|
|
@@ -6884,12 +6955,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6884
6955
|
readTotalTimeouts: () => this.listGuardedPools().reduce((sum, pool) => sum + pool.factory.getPoolBacklog().timedOut, 0),
|
|
6885
6956
|
readStore: () => this.readStore(),
|
|
6886
6957
|
writeStore: (patch) => this.writeStore(patch),
|
|
6887
|
-
resolveFingerprint: async () => {
|
|
6888
|
-
|
|
6889
|
-
|
|
6890
|
-
|
|
6891
|
-
|
|
6892
|
-
|
|
6958
|
+
resolveFingerprint: async () => closureVersionFromRunnerPath(process.argv[1], { readFile: (path) => {
|
|
6959
|
+
try {
|
|
6960
|
+
return fs.readFileSync(path, "utf8");
|
|
6961
|
+
} catch {
|
|
6962
|
+
return null;
|
|
6963
|
+
}
|
|
6964
|
+
} }),
|
|
6893
6965
|
emitFailed: (finding) => {
|
|
6894
6966
|
this.eventBus?.emit(createEvent(EventCategory.SystemLivenessFailed, TIMEOUT_GUARD_SOURCE, finding));
|
|
6895
6967
|
},
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@camstack/addon-pipeline",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.182",
|
|
4
4
|
"description": "Pipeline bundle — runner, detection, motion, audio + stream broker. Multi-entry npm package shipping pipeline addons under a single bundle.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"camstack",
|