@camstack/addon-pipeline 1.2.130 → 1.2.131
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{addon-utils-DH78p79i.js → addon-utils-DCQ5AP_v.js} +1 -1
- package/dist/audio-analyzer/index.js +3 -3
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/detection-pipeline/index.js +332 -20
- package/dist/detection-pipeline/index.mjs +330 -18
- package/dist/{dist-DOw-izoI.js → dist-CqoMNE1e.js} +1817 -1712
- package/dist/{dist-Wt81pN5Y.mjs → dist-DO0UVhOk.mjs} +1817 -1712
- package/dist/{event-loop-stall-monitor-CtQCOZqm.mjs → event-loop-stall-monitor-D0Juf7Dn.mjs} +1 -1
- package/dist/{event-loop-stall-monitor-Bld7jNCb.js → event-loop-stall-monitor-eM7_yYfn.js} +1 -1
- package/dist/{lazy-sharp-BbNI2ZD-.js → lazy-sharp-BGrpS8yi.js} +1 -1
- package/dist/motion-wasm/index.js +2 -2
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/pipeline-runner/index.js +4 -4
- package/dist/pipeline-runner/index.mjs +3 -3
- package/dist/{process-memory-1_D3EZT9.mjs → process-memory-Bac3FuG6.mjs} +1 -1
- package/dist/{process-memory-D9H5qjDl.js → process-memory-xAPjvSw9.js} +1 -1
- package/dist/recorder/index.js +95 -4
- package/dist/recorder/index.mjs +94 -3
- package/dist/{segment-demux-js-C2jBBqGW.js → segment-demux-js-Cu0oDMUL.js} +1 -1
- package/dist/{segment-demux-js-BW-Q-j2A.mjs → segment-demux-js-skhsAZx6.mjs} +1 -1
- package/dist/session-decode/decode-worker-child.js +2 -2
- package/dist/session-decode/decode-worker-child.mjs +1 -1
- package/dist/stream-broker/_stub.js +1 -1
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-Dw-iwP4y.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-DGQkDEYw.mjs} +3 -3
- package/dist/stream-broker/{_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-D36MoVV1.mjs → _virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CQKwHIHI.mjs} +1 -1
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-DEVCEYgu.mjs → hostInit-D7FwwZgj.mjs} +3 -3
- package/dist/stream-broker/index.js +2 -2
- package/dist/stream-broker/index.mjs +2 -2
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-nAZl_Rsv.js → worker-protocol-G3HXTY8z.js} +1 -1
- package/dist/{worker-protocol-rpnr7Ldd.mjs → worker-protocol-_Ujvj1hl.mjs} +1 -1
- package/package.json +1 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { n as __require } from "../chunk-DnnnRqeS.mjs";
|
|
2
|
-
import { $ as pickClusterStepModels, D as YAMNET_TO_MACRO, Dt as hydrateSchema, Et as createEvent, Ft as sleep, G as evaluateZoneRules, Gt as string, I as defaultDeviceFor$1, Kt as union, Lt as array, Mt as parseJsonUnknown, St as BaseAddon, Ut as object, V as detectionPipelineCapability, W as enumerateInferenceDevices, Z as overlayClusterStepSettings, c as DEFAULT_CLUSTER_STEP_MODELS, dt as runtimeDevices$1, et as pickClusterStepSettings, gt as errMsg, jt as nodePin, l as DEFAULT_CLUSTER_STEP_SETTINGS, lt as resolvePoolMemoryPolicy, mt as supportedRuntimes$1, p as DEVICE_BACKEND_TO_FORMAT, q as inferModelProvider, qt as EventCategory, rt as pipelineExecutorCapability, st as resolveClusterStepModelId, t as APPLE_SA_TO_MACRO, x as PoolMemoryWatchdog } from "../dist-
|
|
3
|
-
import { a as ALL_PIPELINE_STEPS, c as getDefaultModelForFormatFromDef, d as resolveModelForFormat, f as landmarkPrecisionVerdict, l as getStep, n as startEventLoopStallMonitor, o as ALL_STEPS, r as localFrameRegistry, s as getDefaultModelForFormat, u as getStepDefinition } from "../event-loop-stall-monitor-
|
|
2
|
+
import { $ as pickClusterStepModels, D as YAMNET_TO_MACRO, Dt as hydrateSchema, Et as createEvent, Ft as sleep, G as evaluateZoneRules, Gt as string, I as defaultDeviceFor$1, Kt as union, Lt as array, Mt as parseJsonUnknown, St as BaseAddon, Ut as object, V as detectionPipelineCapability, W as enumerateInferenceDevices, Z as overlayClusterStepSettings, c as DEFAULT_CLUSTER_STEP_MODELS, dt as runtimeDevices$1, et as pickClusterStepSettings, gt as errMsg, jt as nodePin, l as DEFAULT_CLUSTER_STEP_SETTINGS, lt as resolvePoolMemoryPolicy, mt as supportedRuntimes$1, p as DEVICE_BACKEND_TO_FORMAT, q as inferModelProvider, qt as EventCategory, rt as pipelineExecutorCapability, st as resolveClusterStepModelId, t as APPLE_SA_TO_MACRO, x as PoolMemoryWatchdog } from "../dist-DO0UVhOk.mjs";
|
|
3
|
+
import { a as ALL_PIPELINE_STEPS, c as getDefaultModelForFormatFromDef, d as resolveModelForFormat, f as landmarkPrecisionVerdict, l as getStep, n as startEventLoopStallMonitor, o as ALL_STEPS, r as localFrameRegistry, s as getDefaultModelForFormat, u as getStepDefinition } from "../event-loop-stall-monitor-D0Juf7Dn.mjs";
|
|
4
4
|
import { t as getSharp } from "../lazy-sharp-BtT5x-qP.mjs";
|
|
5
|
-
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-
|
|
5
|
+
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-Bac3FuG6.mjs";
|
|
6
6
|
import { a as ensureModel, o as isModelDownloaded, r as deleteModelFromDisk } from "../addon-utils-CgBCF-mE.mjs";
|
|
7
7
|
import sharp from "sharp";
|
|
8
8
|
import { spawn } from "node:child_process";
|
|
@@ -901,15 +901,20 @@ var PoolWorker = class {
|
|
|
901
901
|
} });
|
|
902
902
|
this.rejectAll(err);
|
|
903
903
|
});
|
|
904
|
-
this.process
|
|
905
|
-
|
|
906
|
-
this.log.error("Worker process exited", { meta: {
|
|
907
|
-
worker: this.opts.workerLabel,
|
|
908
|
-
code
|
|
909
|
-
} });
|
|
910
|
-
this.rejectAll(/* @__PURE__ */ new Error(`Worker process exited with code ${code}`));
|
|
911
|
-
}
|
|
904
|
+
const spawnedProcess = this.process;
|
|
905
|
+
spawnedProcess.on("exit", (code, signal) => {
|
|
912
906
|
this.ready = false;
|
|
907
|
+
if (this.process !== spawnedProcess) return;
|
|
908
|
+
this.log.error("Worker process exited", { meta: {
|
|
909
|
+
worker: this.opts.workerLabel,
|
|
910
|
+
pid: spawnedProcess.pid ?? null,
|
|
911
|
+
runtime: this.opts.poolRuntime,
|
|
912
|
+
device: this.opts.device ?? "default",
|
|
913
|
+
code,
|
|
914
|
+
signal,
|
|
915
|
+
inFlight: this.pending.size
|
|
916
|
+
} });
|
|
917
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
|
|
913
918
|
});
|
|
914
919
|
this.process.stdout.on("data", (chunk) => {
|
|
915
920
|
const t0 = Date.now();
|
|
@@ -1037,6 +1042,7 @@ var PoolWorker = class {
|
|
|
1037
1042
|
this.process = null;
|
|
1038
1043
|
this.ready = false;
|
|
1039
1044
|
await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
|
|
1045
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
|
|
1040
1046
|
}
|
|
1041
1047
|
/**
|
|
1042
1048
|
* Shed an inference request when the worker already has {@link
|
|
@@ -2349,6 +2355,23 @@ var EngineFactory = class {
|
|
|
2349
2355
|
active: l.active
|
|
2350
2356
|
}));
|
|
2351
2357
|
}
|
|
2358
|
+
/**
|
|
2359
|
+
* Is this factory's pool alive and able to answer?
|
|
2360
|
+
*
|
|
2361
|
+
* `SharedInferencePool.isReady()` and `PoolWorker.isReady()` have always
|
|
2362
|
+
* existed; this factory surfaced `getPoolPids()` and `getPoolBacklog()` and
|
|
2363
|
+
* **not** readiness, so the one signal every caller needed sat one layer
|
|
2364
|
+
* below all of them and was never lifted. That is why a pool whose Python
|
|
2365
|
+
* worker was SIGABRT'd 1.3 s after `Pool ready` was handed back to 370 000
|
|
2366
|
+
* consecutive dispatches over 31 hours: `resolveDeviceFactory` had nothing
|
|
2367
|
+
* to ask.
|
|
2368
|
+
*
|
|
2369
|
+
* A factory that never initialised answers `false` — a pool that is not
|
|
2370
|
+
* there cannot serve a frame, and the caller's job is the same either way.
|
|
2371
|
+
*/
|
|
2372
|
+
isReady() {
|
|
2373
|
+
return this.pool?.isReady() ?? false;
|
|
2374
|
+
}
|
|
2352
2375
|
/** Native pid of the underlying Python pool, if any. */
|
|
2353
2376
|
getPoolPid() {
|
|
2354
2377
|
return this.pool?.getPid() ?? null;
|
|
@@ -2655,24 +2678,160 @@ var IdlePoolReaper = class {
|
|
|
2655
2678
|
}
|
|
2656
2679
|
};
|
|
2657
2680
|
//#endregion
|
|
2681
|
+
//#region src/detection-pipeline/engine/inference-device-liveness.ts
|
|
2682
|
+
/**
|
|
2683
|
+
* A dispatch refused because its inference device is not usable.
|
|
2684
|
+
*
|
|
2685
|
+
* Distinct from `PoolWorker[w0]: not initialized`, which said the same thing
|
|
2686
|
+
* for three different states and so distinguished none of them. This one names
|
|
2687
|
+
* the device, the state and the underlying death, so a single rejected frame
|
|
2688
|
+
* is decidable without correlating logs.
|
|
2689
|
+
*/
|
|
2690
|
+
var InferenceDeviceUnusableError = class extends Error {
|
|
2691
|
+
deviceKey;
|
|
2692
|
+
state;
|
|
2693
|
+
constructor(unhealthy) {
|
|
2694
|
+
super(`inference device "${unhealthy.deviceKey}" is ${unhealthy.state} after ${unhealthy.deaths} pool death(s): ${unhealthy.lastError}`);
|
|
2695
|
+
this.name = "InferenceDeviceUnusableError";
|
|
2696
|
+
this.deviceKey = unhealthy.deviceKey;
|
|
2697
|
+
this.state = unhealthy.state;
|
|
2698
|
+
}
|
|
2699
|
+
};
|
|
2700
|
+
/**
|
|
2701
|
+
* Defaults. Three deaths inside ten minutes trips the breaker, so a
|
|
2702
|
+
* deterministic crash costs **three** Python spawns instead of one per frame,
|
|
2703
|
+
* and a flaky-once-an-hour pool still self-heals forever.
|
|
2704
|
+
*/
|
|
2705
|
+
var DEFAULT_INFERENCE_DEVICE_LIVENESS = {
|
|
2706
|
+
windowMs: 600 * 1e3,
|
|
2707
|
+
maxDeaths: 3,
|
|
2708
|
+
maxBackoffMs: 3e4
|
|
2709
|
+
};
|
|
2710
|
+
var InferenceDeviceLivenessSupervisor = class {
|
|
2711
|
+
opts;
|
|
2712
|
+
state = /* @__PURE__ */ new Map();
|
|
2713
|
+
constructor(opts = DEFAULT_INFERENCE_DEVICE_LIVENESS) {
|
|
2714
|
+
this.opts = opts;
|
|
2715
|
+
}
|
|
2716
|
+
/**
|
|
2717
|
+
* Record one pool death for `deviceKey` and decide what happens next.
|
|
2718
|
+
* `now` is injectable for deterministic tests. Never throws.
|
|
2719
|
+
*/
|
|
2720
|
+
recordDeath(deviceKey, error, now = Date.now()) {
|
|
2721
|
+
const cutoff = now - this.opts.windowMs;
|
|
2722
|
+
const recent = [...(this.state.get(deviceKey)?.deaths ?? []).filter((t) => t >= cutoff), now];
|
|
2723
|
+
if (recent.length >= this.opts.maxDeaths) {
|
|
2724
|
+
this.state.set(deviceKey, {
|
|
2725
|
+
deaths: recent,
|
|
2726
|
+
failed: true,
|
|
2727
|
+
retryAtMs: Number.POSITIVE_INFINITY,
|
|
2728
|
+
lastError: error,
|
|
2729
|
+
since: now
|
|
2730
|
+
});
|
|
2731
|
+
return {
|
|
2732
|
+
action: "failed",
|
|
2733
|
+
deathsInWindow: recent.length
|
|
2734
|
+
};
|
|
2735
|
+
}
|
|
2736
|
+
const backoffMs = Math.min(this.opts.maxBackoffMs, 500 * 2 ** Math.min(6, Math.max(0, recent.length - 1)));
|
|
2737
|
+
this.state.set(deviceKey, {
|
|
2738
|
+
deaths: recent,
|
|
2739
|
+
failed: false,
|
|
2740
|
+
retryAtMs: now + backoffMs,
|
|
2741
|
+
lastError: error,
|
|
2742
|
+
since: now
|
|
2743
|
+
});
|
|
2744
|
+
return {
|
|
2745
|
+
action: "rebuild",
|
|
2746
|
+
backoffMs,
|
|
2747
|
+
deathsInWindow: recent.length
|
|
2748
|
+
};
|
|
2749
|
+
}
|
|
2750
|
+
/** Has the breaker tripped for this device? */
|
|
2751
|
+
isFailed(deviceKey) {
|
|
2752
|
+
return this.state.get(deviceKey)?.failed ?? false;
|
|
2753
|
+
}
|
|
2754
|
+
/**
|
|
2755
|
+
* Why this device cannot be used right now, or `null` when it may be
|
|
2756
|
+
* (re)built. A device inside its backoff answers `'backoff'`; a device whose
|
|
2757
|
+
* budget is spent answers `'failed'` forever.
|
|
2758
|
+
*/
|
|
2759
|
+
unusableReason(deviceKey, now = Date.now()) {
|
|
2760
|
+
const s = this.state.get(deviceKey);
|
|
2761
|
+
if (!s) return null;
|
|
2762
|
+
if (s.failed) return {
|
|
2763
|
+
deviceKey,
|
|
2764
|
+
state: "failed",
|
|
2765
|
+
since: s.since,
|
|
2766
|
+
deaths: s.deaths.length,
|
|
2767
|
+
lastError: s.lastError
|
|
2768
|
+
};
|
|
2769
|
+
if (now < s.retryAtMs) return {
|
|
2770
|
+
deviceKey,
|
|
2771
|
+
state: "backoff",
|
|
2772
|
+
since: s.since,
|
|
2773
|
+
deaths: s.deaths.length,
|
|
2774
|
+
lastError: s.lastError
|
|
2775
|
+
};
|
|
2776
|
+
return null;
|
|
2777
|
+
}
|
|
2778
|
+
/**
|
|
2779
|
+
* Every device the breaker currently refuses, sorted by key.
|
|
2780
|
+
*
|
|
2781
|
+
* This is what the health channel publishes. `'backoff'` entries are
|
|
2782
|
+
* included deliberately: a device mid-backoff cannot serve a frame either,
|
|
2783
|
+
* and the reader's own arm/apply reluctance (D49) is what keeps a single
|
|
2784
|
+
* transient observation from excluding it.
|
|
2785
|
+
*/
|
|
2786
|
+
unhealthy(now = Date.now()) {
|
|
2787
|
+
const out = [];
|
|
2788
|
+
for (const deviceKey of [...this.state.keys()].toSorted()) {
|
|
2789
|
+
const reason = this.unusableReason(deviceKey, now);
|
|
2790
|
+
if (reason) out.push(reason);
|
|
2791
|
+
}
|
|
2792
|
+
return out;
|
|
2793
|
+
}
|
|
2794
|
+
/**
|
|
2795
|
+
* Operator re-arm: forget everything about `deviceKey` so the next dispatch
|
|
2796
|
+
* builds a fresh pool with a full budget. Returns whether there was anything
|
|
2797
|
+
* to forget — a terminal state nobody can clear is a silent fault, and this
|
|
2798
|
+
* is the clearing.
|
|
2799
|
+
*/
|
|
2800
|
+
rearm(deviceKey) {
|
|
2801
|
+
return this.state.delete(deviceKey);
|
|
2802
|
+
}
|
|
2803
|
+
/** Forget every device (shutdown / full engine re-spin). */
|
|
2804
|
+
reset() {
|
|
2805
|
+
this.state.clear();
|
|
2806
|
+
}
|
|
2807
|
+
};
|
|
2808
|
+
//#endregion
|
|
2658
2809
|
//#region src/detection-pipeline/engine/pool-memory-guard.ts
|
|
2659
2810
|
function createDetectionPoolMemoryGuard(source, logger) {
|
|
2660
2811
|
const log = logger.child("pool-memory");
|
|
2812
|
+
/** Pools already reported dead — the `pool is DEAD` line fires on the EDGE.
|
|
2813
|
+
* A pool nobody dispatches to is never condemned, so without this gate one
|
|
2814
|
+
* idle corpse writes an ERROR every sweep forever, which is how a real
|
|
2815
|
+
* signal becomes noise. Cleared when the pool comes back or goes away. */
|
|
2816
|
+
const reportedDead = /* @__PURE__ */ new Set();
|
|
2661
2817
|
return new PoolMemoryWatchdog({
|
|
2662
2818
|
policy: resolvePoolMemoryPolicy(process.env),
|
|
2663
2819
|
log,
|
|
2664
2820
|
restart: (key) => source.restartPool(key),
|
|
2665
2821
|
sample: async () => {
|
|
2666
2822
|
const out = [];
|
|
2823
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2667
2824
|
for (const pool of source.listPools()) {
|
|
2668
|
-
|
|
2825
|
+
seen.add(pool.key);
|
|
2826
|
+
const telemetry = await samplePool(pool, log, reportedDead);
|
|
2669
2827
|
if (telemetry) out.push(telemetry);
|
|
2670
2828
|
}
|
|
2829
|
+
for (const key of [...reportedDead]) if (!seen.has(key)) reportedDead.delete(key);
|
|
2671
2830
|
return out;
|
|
2672
2831
|
}
|
|
2673
2832
|
});
|
|
2674
2833
|
}
|
|
2675
|
-
async function samplePool(pool, log) {
|
|
2834
|
+
async function samplePool(pool, log, reportedDead) {
|
|
2676
2835
|
const pids = pool.factory.getPoolPids();
|
|
2677
2836
|
if (pids.length === 0) return null;
|
|
2678
2837
|
let maxRssBytes = 0;
|
|
@@ -2696,7 +2855,18 @@ async function samplePool(pool, log) {
|
|
|
2696
2855
|
threads: mem.threads
|
|
2697
2856
|
});
|
|
2698
2857
|
}
|
|
2699
|
-
if (workers.length === 0)
|
|
2858
|
+
if (workers.length === 0) {
|
|
2859
|
+
if (!pool.factory.isReady()) {
|
|
2860
|
+
if (!reportedDead.has(pool.key)) {
|
|
2861
|
+
reportedDead.add(pool.key);
|
|
2862
|
+
log.error("inference pool is DEAD — every worker pid is gone and the pool is not ready", { meta: {
|
|
2863
|
+
poolKey: pool.key,
|
|
2864
|
+
pids
|
|
2865
|
+
} });
|
|
2866
|
+
}
|
|
2867
|
+
} else reportedDead.delete(pool.key);
|
|
2868
|
+
return null;
|
|
2869
|
+
}
|
|
2700
2870
|
const backlog = pool.factory.getPoolBacklog();
|
|
2701
2871
|
const memStats = await pool.factory.getPoolMemStats().catch((err) => err instanceof Error ? err.message : String(err));
|
|
2702
2872
|
return {
|
|
@@ -8401,7 +8571,16 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8401
8571
|
* If global steps exist, initializes with them. Otherwise, creates an empty pool.
|
|
8402
8572
|
*/
|
|
8403
8573
|
async ensureEngineFactory() {
|
|
8404
|
-
|
|
8574
|
+
const defaultDeviceKey = deviceKeyOf(this.currentEngine);
|
|
8575
|
+
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8576
|
+
if (this.engineFactory) {
|
|
8577
|
+
if (this.engineFactory.isReady()) return;
|
|
8578
|
+
const dead = this.engineFactory;
|
|
8579
|
+
this.engineFactory = null;
|
|
8580
|
+
this.noteDeviceDeath(defaultDeviceKey, "pool worker is not ready");
|
|
8581
|
+
await dead.dispose().catch(() => void 0);
|
|
8582
|
+
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8583
|
+
}
|
|
8405
8584
|
const inflight = this.engineFactoryInflight;
|
|
8406
8585
|
if (inflight) {
|
|
8407
8586
|
await inflight;
|
|
@@ -8423,11 +8602,12 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8423
8602
|
} catch (err) {
|
|
8424
8603
|
await factory.dispose().catch(() => void 0);
|
|
8425
8604
|
this.log.error("Node-default pool failed to initialize", { meta: {
|
|
8426
|
-
deviceKey:
|
|
8605
|
+
deviceKey: defaultDeviceKey,
|
|
8427
8606
|
backend: this.currentEngine.backend,
|
|
8428
8607
|
device: this.currentEngine.device ?? null,
|
|
8429
8608
|
error: errMsg(err)
|
|
8430
8609
|
} });
|
|
8610
|
+
this.noteDeviceDeath(defaultDeviceKey, errMsg(err));
|
|
8431
8611
|
throw err;
|
|
8432
8612
|
}
|
|
8433
8613
|
this.engineFactory = factory;
|
|
@@ -8476,16 +8656,147 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8476
8656
|
* fail (the loser can't claim the already-held device).
|
|
8477
8657
|
*/
|
|
8478
8658
|
deviceFactoryInflight = /* @__PURE__ */ new Map();
|
|
8659
|
+
/**
|
|
8660
|
+
* Per-`deviceKey` restart budget for the Python pools (D6 discipline).
|
|
8661
|
+
*
|
|
8662
|
+
* In-memory and per-runner ON PURPOSE. The terminal state means "this
|
|
8663
|
+
* accelerator aborted N times in a row a few minutes ago"; a runner respawn
|
|
8664
|
+
* re-derives it in three dispatches if it is still true, and persisting it
|
|
8665
|
+
* would outlive the driver update that fixes it. The operator's explicit
|
|
8666
|
+
* re-arm is {@link rearmInferenceDevice}, and the state is never invisible:
|
|
8667
|
+
* it is published on {@link getInferenceDeviceHealth}.
|
|
8668
|
+
*/
|
|
8669
|
+
deviceLiveness = new InferenceDeviceLivenessSupervisor();
|
|
8670
|
+
/**
|
|
8671
|
+
* Throw if `deviceKey` is currently refused — before a single Python process
|
|
8672
|
+
* is spawned, a model index is allocated, or a frame is copied.
|
|
8673
|
+
*
|
|
8674
|
+
* The cost of NOT asking, measured: each frame of the 2026-08-25 storm ran
|
|
8675
|
+
* the full `models to load` → `allocateIndex` → `loadModel` → throw cycle at
|
|
8676
|
+
* 13 Hz for 31 hours, leaking one model index per frame — and past 254
|
|
8677
|
+
* (`encodeModelByte`) that pool could never have served inference again even
|
|
8678
|
+
* if its worker had come back.
|
|
8679
|
+
*/
|
|
8680
|
+
refuseIfDeviceUnusable(deviceKey) {
|
|
8681
|
+
const unhealthy = this.deviceLiveness.unusableReason(deviceKey);
|
|
8682
|
+
if (unhealthy) throw new InferenceDeviceUnusableError(unhealthy);
|
|
8683
|
+
}
|
|
8684
|
+
/**
|
|
8685
|
+
* Record one pool death against the budget and SAY what was decided.
|
|
8686
|
+
*
|
|
8687
|
+
* One line per state change, not one per dispatch: the refusals that follow
|
|
8688
|
+
* are already carried by the rejected frame's own error. The `failed`
|
|
8689
|
+
* transition is an ERROR because it is the terminal state — an operator who
|
|
8690
|
+
* never sees this line has a silently half-dead node, which is the failure
|
|
8691
|
+
* this whole path exists to end.
|
|
8692
|
+
*/
|
|
8693
|
+
noteDeviceDeath(deviceKey, reason) {
|
|
8694
|
+
const decision = this.deviceLiveness.recordDeath(deviceKey, reason);
|
|
8695
|
+
if (decision.action === "failed") {
|
|
8696
|
+
this.log.error("inference device FAILED — restart budget exhausted, no pool will be spawned", { meta: {
|
|
8697
|
+
deviceKey,
|
|
8698
|
+
deathsInWindow: decision.deathsInWindow,
|
|
8699
|
+
reason,
|
|
8700
|
+
rearm: "pipelineExecutor.rearmInferenceDevice"
|
|
8701
|
+
} });
|
|
8702
|
+
return;
|
|
8703
|
+
}
|
|
8704
|
+
this.log.warn("inference pool died — rebuilding under the restart budget", { meta: {
|
|
8705
|
+
deviceKey,
|
|
8706
|
+
deathsInWindow: decision.deathsInWindow,
|
|
8707
|
+
backoffMs: decision.backoffMs,
|
|
8708
|
+
reason
|
|
8709
|
+
} });
|
|
8710
|
+
}
|
|
8711
|
+
/**
|
|
8712
|
+
* Retire a dead per-device pool: drop it from the map FIRST (so a concurrent
|
|
8713
|
+
* dispatch rebuilds fresh through the single-flight guard rather than
|
|
8714
|
+
* grabbing the corpse), cancel its idle timer, charge the budget, then
|
|
8715
|
+
* dispose — which rejects whatever was still in flight on it with a reason
|
|
8716
|
+
* rather than letting those requests sit until their own deadlines.
|
|
8717
|
+
*/
|
|
8718
|
+
async condemnDeviceFactory(deviceKey, factory, reason) {
|
|
8719
|
+
if (this.factoriesByDevice.get(deviceKey) === factory) {
|
|
8720
|
+
this.factoriesByDevice.delete(deviceKey);
|
|
8721
|
+
this.deviceReaper.cancel(deviceKey);
|
|
8722
|
+
this.noteDeviceDeath(deviceKey, reason);
|
|
8723
|
+
}
|
|
8724
|
+
await factory.dispose().catch(() => void 0);
|
|
8725
|
+
}
|
|
8726
|
+
/**
|
|
8727
|
+
* Every inference device this node currently refuses, and why.
|
|
8728
|
+
*
|
|
8729
|
+
* The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
|
|
8730
|
+
* entirely inside this addon and reached no routing decision at any level:
|
|
8731
|
+
* the capability gate is keyed on model FORMAT and so is permanently blind
|
|
8732
|
+
* between `openvino:gpu` and `openvino:npu`, and the orchestrator's live
|
|
8733
|
+
* probe answers about HARDWARE, which was present the whole time. So the
|
|
8734
|
+
* balancer kept handing this node cameras by rotation — 20 sessions onto a
|
|
8735
|
+
* dead pool in 20 minutes — and every one of them lost its frames.
|
|
8736
|
+
*
|
|
8737
|
+
* Two sources, deliberately merged into one answer:
|
|
8738
|
+
* - the breaker's own `failed` / `backoff` states, and
|
|
8739
|
+
* - any factory currently cached but NOT ready. A pool nobody has dispatched
|
|
8740
|
+
* to since it died has not yet been condemned by the dispatch path, and a
|
|
8741
|
+
* device with no traffic is exactly the one that must stop being chosen
|
|
8742
|
+
* BEFORE the traffic arrives. Charging the budget is the dispatch path's
|
|
8743
|
+
* job; reporting is free and costs no spawn.
|
|
8744
|
+
*
|
|
8745
|
+
* Read-only and synchronous over in-memory state: it never probes, never
|
|
8746
|
+
* spawns, and never throws — the reader's whole safety argument rests on a
|
|
8747
|
+
* failed read changing nothing, and a read that can fail for its own reasons
|
|
8748
|
+
* would poison that.
|
|
8749
|
+
*/
|
|
8750
|
+
async getInferenceDeviceHealth() {
|
|
8751
|
+
const byKey = /* @__PURE__ */ new Map();
|
|
8752
|
+
for (const entry of this.deviceLiveness.unhealthy()) byKey.set(entry.deviceKey, entry);
|
|
8753
|
+
const live = [...this.factoriesByDevice, ...this.engineFactory ? [[deviceKeyOf(this.currentEngine), this.engineFactory]] : []];
|
|
8754
|
+
for (const [deviceKey, factory] of live) {
|
|
8755
|
+
if (factory.isReady() || byKey.has(deviceKey)) continue;
|
|
8756
|
+
byKey.set(deviceKey, {
|
|
8757
|
+
deviceKey,
|
|
8758
|
+
state: "backoff",
|
|
8759
|
+
since: Date.now(),
|
|
8760
|
+
deaths: 0,
|
|
8761
|
+
lastError: "pool worker is not ready"
|
|
8762
|
+
});
|
|
8763
|
+
}
|
|
8764
|
+
return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
|
|
8765
|
+
}
|
|
8766
|
+
/**
|
|
8767
|
+
* Operator re-arm of a terminally failed inference device: forget the budget
|
|
8768
|
+
* so the next dispatch builds a fresh pool.
|
|
8769
|
+
*
|
|
8770
|
+
* A terminal state with no way out is a silent fault dressed as a safety
|
|
8771
|
+
* feature. This is the way out, and it is the ONLY automatic-looking one on
|
|
8772
|
+
* offer: there is deliberately no timed probation, because the crash this
|
|
8773
|
+
* bounds is deterministic and a probation would simply respawn Python
|
|
8774
|
+
* forever at a slower rate. Re-arming a device that was never failed is a
|
|
8775
|
+
* no-op, reported as such.
|
|
8776
|
+
*/
|
|
8777
|
+
async rearmInferenceDevice(input) {
|
|
8778
|
+
const rearmed = this.deviceLiveness.rearm(input.deviceKey);
|
|
8779
|
+
this.log.info("inference device re-armed by operator", { meta: {
|
|
8780
|
+
deviceKey: input.deviceKey,
|
|
8781
|
+
rearmed
|
|
8782
|
+
} });
|
|
8783
|
+
return { rearmed };
|
|
8784
|
+
}
|
|
8479
8785
|
async resolveDeviceFactory(deviceKey) {
|
|
8480
8786
|
const engine = resolveDeviceEngine(deviceKey);
|
|
8481
8787
|
if (enginesEqual(engine, this.currentEngine)) {
|
|
8482
8788
|
await this.ensureEngineFactory();
|
|
8483
8789
|
return this.engineFactory;
|
|
8484
8790
|
}
|
|
8791
|
+
this.refuseIfDeviceUnusable(deviceKey);
|
|
8485
8792
|
const existing = this.factoriesByDevice.get(deviceKey);
|
|
8486
8793
|
if (existing) {
|
|
8487
|
-
|
|
8488
|
-
|
|
8794
|
+
if (existing.isReady()) {
|
|
8795
|
+
this.deviceReaper.touch(deviceKey);
|
|
8796
|
+
return existing;
|
|
8797
|
+
}
|
|
8798
|
+
await this.condemnDeviceFactory(deviceKey, existing, "pool worker is not ready");
|
|
8799
|
+
this.refuseIfDeviceUnusable(deviceKey);
|
|
8489
8800
|
}
|
|
8490
8801
|
const inflight = this.deviceFactoryInflight.get(deviceKey);
|
|
8491
8802
|
if (inflight) return inflight;
|
|
@@ -8510,6 +8821,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8510
8821
|
error: err instanceof Error ? err.message : String(err)
|
|
8511
8822
|
} });
|
|
8512
8823
|
await factory.dispose().catch(() => void 0);
|
|
8824
|
+
this.noteDeviceDeath(deviceKey, errMsg(err));
|
|
8513
8825
|
throw err;
|
|
8514
8826
|
}
|
|
8515
8827
|
this.factoriesByDevice.set(deviceKey, factory);
|