@camstack/addon-pipeline 1.2.294 → 1.2.296
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +241 -0
- package/dist/audio-analyzer/index.js +2 -2
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/{default-detection-model-0dPKRKUD.mjs → default-detection-model-Co578D8C.mjs} +181 -99
- package/dist/{default-detection-model-D24AJOTn.js → default-detection-model-D1daTtqT.js} +181 -99
- package/dist/detection-pipeline/index.js +1301 -531
- package/dist/detection-pipeline/index.mjs +1301 -531
- package/dist/{dist-CJR259Xf.js → dist-8up-f2TX.js} +3688 -2698
- package/dist/{dist-RXbmRAwP.mjs → dist-CCd0Q3nr.mjs} +3676 -2698
- package/dist/motion-wasm/index.js +1 -1
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/{node-atmRSHPk.mjs → node-DgMSXSWP.mjs} +1 -1
- package/dist/{node-DWg9zbY1.js → node-lpQgHes9.js} +1 -1
- package/dist/pipeline-runner/index.js +975 -270
- package/dist/pipeline-runner/index.mjs +975 -270
- package/dist/{process-memory-BJUXvTjd.js → process-memory-CX_92V_r.js} +1 -1
- package/dist/{process-memory-BgFOHFnx.mjs → process-memory-DFC_O5zE.mjs} +1 -1
- package/dist/recorder/index.js +14 -6
- package/dist/recorder/index.mjs +14 -6
- package/dist/{segment-demux-js-C_fPJub3.js → segment-demux-js-DzBx6NN2.js} +1 -1
- package/dist/{segment-demux-js-G7wFpHzn.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
- package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
- package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
- package/dist/stream-broker/_stub.js +2 -2
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-6IyM-BIn.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-BBYHWS3M.mjs → hostInit-BPtppL3W.mjs} +2 -2
- package/dist/stream-broker/index.js +4 -4
- package/dist/stream-broker/index.mjs +4 -4
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
- package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
- package/package.json +3 -2
- package/python/inference_pool.py +422 -64
- package/python/postprocessors/__init__.py +2 -0
- package/python/postprocessors/ssd.py +73 -17
- package/python/postprocessors/test_ssd.py +205 -0
- package/python/postprocessors/test_yunet.py +292 -0
- package/python/postprocessors/testdata/ssdlite_mobiledet_outputs.json +1 -0
- package/python/postprocessors/yunet.py +275 -0
- package/python/test_inference_pool_compile_off_loop.py +414 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-6IHzlLJ_.mjs +0 -26
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DoyA71_q.mjs +0 -26
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { n as __require } from "../chunk-DnnnRqeS.mjs";
|
|
2
|
-
import {
|
|
3
|
-
import { s as readProcessCost } from "../node-
|
|
4
|
-
import { a as localFrameRegistry, c as ALL_STEPS, d as getStepDefinition, f as resolveModelForFormat, l as getDefaultModelForFormat, m as landmarkPrecisionVerdict, n as resolveDefaultDetectionModel, r as macroGateVerdict, s as ALL_PIPELINE_STEPS, u as getStep } from "../default-detection-model-
|
|
2
|
+
import { At as pickClusterStepSettings, Cn as parseJsonUnknown, Dt as overlayClusterStepSettings, En as sleep, Gt as resolvePoolMemoryPolicy, H as YAMNET_TO_MACRO, Ln as string, Mt as pickInvalidClusterStepModels, On as array, Pn as object, Pt as pipelineExecutorCapability, Qt as supportedRuntimes$1, Rn as union, Sn as nodePin, Vt as resolveClusterStepModelId, cn as BaseAddon, et as defaultDeviceFor$1, gt as inferModelProvider, it as detectionPipelineCapability, j as PoolMemoryWatchdog, kt as pickClusterStepModels, lt as evaluateZoneRules, m as DEFAULT_CLUSTER_STEP_SETTINGS, mn as createEvent, p as DEFAULT_CLUSTER_STEP_MODELS, qt as runtimeDevices$1, st as enumerateInferenceDevices, t as APPLE_SA_TO_MACRO, tn as errMsg, vn as hydrateSchema, vt as loadContributionCapability, y as DEVICE_BACKEND_TO_FORMAT, zn as EventCategory } from "../dist-CCd0Q3nr.mjs";
|
|
3
|
+
import { s as readProcessCost } from "../node-DgMSXSWP.mjs";
|
|
4
|
+
import { a as localFrameRegistry, c as ALL_STEPS, d as getStepDefinition, f as resolveModelForFormat, l as getDefaultModelForFormat, m as landmarkPrecisionVerdict, n as resolveDefaultDetectionModel, r as macroGateVerdict, s as ALL_PIPELINE_STEPS, u as getStep } from "../default-detection-model-Co578D8C.mjs";
|
|
5
5
|
import { t as getSharp } from "../lazy-sharp-DXsqwpph.mjs";
|
|
6
|
-
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-
|
|
6
|
+
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-DFC_O5zE.mjs";
|
|
7
7
|
import * as os from "node:os";
|
|
8
8
|
import * as fs from "node:fs";
|
|
9
9
|
import * as path$1 from "node:path";
|
|
@@ -415,6 +415,12 @@ var ClusterModelSource = class {
|
|
|
415
415
|
const next = pickClusterStepModels(view);
|
|
416
416
|
const nextSettings = pickClusterStepSettings(view);
|
|
417
417
|
this.lastReadAtMs = this.now();
|
|
418
|
+
const invalid = pickInvalidClusterStepModels(view);
|
|
419
|
+
if (invalid.length > 0) this.logger.warn("cluster step model row holds a value that is not a model id — running the catalog default for that step; no re-embed pass will run on it", { meta: {
|
|
420
|
+
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
421
|
+
invalid,
|
|
422
|
+
inUse: Object.fromEntries(invalid.map((e) => [e.stepId, next[e.stepId]]))
|
|
423
|
+
} });
|
|
418
424
|
const changed = Object.keys(next).filter((stepId) => next[stepId] !== this.models[stepId]);
|
|
419
425
|
const settingsChanged = Object.keys(nextSettings).some((stepId) => {
|
|
420
426
|
const from = this.settings[stepId] ?? {};
|
|
@@ -562,6 +568,98 @@ var DeviceOverrideMirror = class {
|
|
|
562
568
|
}
|
|
563
569
|
};
|
|
564
570
|
//#endregion
|
|
571
|
+
//#region src/detection-pipeline/engine/pool-worker-health.ts
|
|
572
|
+
/**
|
|
573
|
+
* The one table of load bounds (D653, fix round 1).
|
|
574
|
+
*
|
|
575
|
+
* The soft bound decides how soon the operator HEARS about a slow compile; the
|
|
576
|
+
* hard bound decides when a compile is given up for hung. A single 120 s bound
|
|
577
|
+
* that also killed the compile was wrong twice: a slow but finite compile was
|
|
578
|
+
* killed before it could write its cache, so every respawn faced the same cold
|
|
579
|
+
* compile and three of them walked the device to `failed`; and on CoreML the
|
|
580
|
+
* bound was shorter than the cross-process compile-cache lock wait alone.
|
|
581
|
+
*/
|
|
582
|
+
var POOL_MODEL_LOAD_BOUNDS = {
|
|
583
|
+
openvino: {
|
|
584
|
+
softMs: 12e4,
|
|
585
|
+
hardMs: 6e5,
|
|
586
|
+
why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
|
|
587
|
+
gilMayBeHeldDuringCompile: false,
|
|
588
|
+
gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
|
|
589
|
+
},
|
|
590
|
+
coreml: {
|
|
591
|
+
softMs: 3e5,
|
|
592
|
+
hardMs: 9e5,
|
|
593
|
+
why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
|
|
594
|
+
gilMayBeHeldDuringCompile: true,
|
|
595
|
+
gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
|
|
596
|
+
},
|
|
597
|
+
onnxruntime: {
|
|
598
|
+
softMs: 12e4,
|
|
599
|
+
hardMs: 6e5,
|
|
600
|
+
why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
|
|
601
|
+
gilMayBeHeldDuringCompile: false,
|
|
602
|
+
gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
|
|
603
|
+
},
|
|
604
|
+
edgetpu: {
|
|
605
|
+
softMs: 12e4,
|
|
606
|
+
hardMs: 6e5,
|
|
607
|
+
why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
|
|
608
|
+
gilMayBeHeldDuringCompile: false,
|
|
609
|
+
gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
|
|
610
|
+
}
|
|
611
|
+
};
|
|
612
|
+
/**
|
|
613
|
+
* The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
|
|
614
|
+
* so the worker's NAMED answer always lands first.
|
|
615
|
+
*/
|
|
616
|
+
var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
|
|
617
|
+
function chargesDeviceBudget(reason) {
|
|
618
|
+
return reason !== "compile-hung";
|
|
619
|
+
}
|
|
620
|
+
/**
|
|
621
|
+
* Tracks live inference outcomes on one worker and says when its executor has
|
|
622
|
+
* produced nothing for too long. Consulted on demand — no timer.
|
|
623
|
+
*
|
|
624
|
+
* A SHED (`dropped`) does not count as a result: the Python loop answers sheds
|
|
625
|
+
* itself, so a worker whose executor threads are hung keeps shedding happily.
|
|
626
|
+
*/
|
|
627
|
+
var InferStallDetector = class {
|
|
628
|
+
unanswered = 0;
|
|
629
|
+
lastResultAt;
|
|
630
|
+
/**
|
|
631
|
+
* When the current streak of unanswered requests began — the first one after
|
|
632
|
+
* a result. IDLE time is not silence: quiet cameras send nothing, and a
|
|
633
|
+
* worker that was asked nothing has failed nothing. Measuring from the last
|
|
634
|
+
* result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
|
|
635
|
+
* cameras) poison a healthy worker and charge the device.
|
|
636
|
+
*/
|
|
637
|
+
streakStartedAt = null;
|
|
638
|
+
constructor(now) {
|
|
639
|
+
this.lastResultAt = now;
|
|
640
|
+
}
|
|
641
|
+
/** A reply that carried a real result (or a real error) from the executor. */
|
|
642
|
+
noteResult(now) {
|
|
643
|
+
this.unanswered = 0;
|
|
644
|
+
this.lastResultAt = now;
|
|
645
|
+
this.streakStartedAt = null;
|
|
646
|
+
}
|
|
647
|
+
/**
|
|
648
|
+
* A live request that ended without a result (deadline expiry or a shed).
|
|
649
|
+
* Returns the stall when the worker has crossed both thresholds.
|
|
650
|
+
*/
|
|
651
|
+
noteUnanswered(now) {
|
|
652
|
+
this.unanswered += 1;
|
|
653
|
+
if (this.streakStartedAt === null) this.streakStartedAt = now;
|
|
654
|
+
const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
|
|
655
|
+
if (this.unanswered >= 20 && silentMs >= 3e4) return {
|
|
656
|
+
unanswered: this.unanswered,
|
|
657
|
+
silentMs
|
|
658
|
+
};
|
|
659
|
+
return null;
|
|
660
|
+
}
|
|
661
|
+
};
|
|
662
|
+
//#endregion
|
|
565
663
|
//#region src/detection-pipeline/engine/shared-inference-pool.ts
|
|
566
664
|
/**
|
|
567
665
|
* SharedInferencePool — TypeScript wrapper for inference_pool.py.
|
|
@@ -603,6 +701,27 @@ var RAW_FMT_CODE = {
|
|
|
603
701
|
gray: 2
|
|
604
702
|
};
|
|
605
703
|
/**
|
|
704
|
+
* A load the worker answered with a machine-readable failure class. The
|
|
705
|
+
* provider tells a `compile-timeout` (counted against THAT MODEL on the
|
|
706
|
+
* device) from an ordinary failure by `reason`, never by parsing text.
|
|
707
|
+
*/
|
|
708
|
+
var PoolModelLoadError = class extends Error {
|
|
709
|
+
reason;
|
|
710
|
+
/** The model's file stem, as the worker named it. */
|
|
711
|
+
model;
|
|
712
|
+
constructor(message, reason, model) {
|
|
713
|
+
super(message);
|
|
714
|
+
this.name = "PoolModelLoadError";
|
|
715
|
+
this.reason = reason;
|
|
716
|
+
this.model = model;
|
|
717
|
+
}
|
|
718
|
+
};
|
|
719
|
+
/**
|
|
720
|
+
* Request id the worker uses for UNSOLICITED events — a reply to no request
|
|
721
|
+
* (`compile-finished-late`). Never allocated to a request.
|
|
722
|
+
*/
|
|
723
|
+
var WORKER_EVENT_REQ_ID = 4294967295;
|
|
724
|
+
/**
|
|
606
725
|
* Per-inference-request reply timeout (ms). Turns a wedged request (worker
|
|
607
726
|
* alive but no reply) into a rejection so every caller settles — runtime frame
|
|
608
727
|
* dispatch drops the frame; the benchmark full-tree run rejects the wedged
|
|
@@ -611,6 +730,22 @@ var RAW_FMT_CODE = {
|
|
|
611
730
|
* large frame) never trips; override via env for constrained hardware.
|
|
612
731
|
*/
|
|
613
732
|
var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
|
|
733
|
+
/** Commands that compile a model — the only ones that get the load deadline. */
|
|
734
|
+
var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
|
|
735
|
+
/**
|
|
736
|
+
* Deadline of the liveness probe (`mem_stats`) sent after a command times out.
|
|
737
|
+
* The worker answers `mem_stats` on its event loop, never behind a compile
|
|
738
|
+
* (D653), so a healthy worker replies in milliseconds even mid-compile; ten
|
|
739
|
+
* seconds only has to outlast a GC pause or a burst of inference replies.
|
|
740
|
+
*/
|
|
741
|
+
var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
|
|
742
|
+
/**
|
|
743
|
+
* How long a poisoned worker may keep its in-flight LIVE requests before it is
|
|
744
|
+
* killed anyway. A poisoned worker takes no new work, and every live request
|
|
745
|
+
* carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
|
|
746
|
+
* by then; the second of slack keeps the backstop from racing the last reply.
|
|
747
|
+
*/
|
|
748
|
+
var POOL_POISON_DRAIN_SLACK_MS = 1e3;
|
|
614
749
|
/**
|
|
615
750
|
* Max inference requests outstanding to ONE worker before new ones are SHED.
|
|
616
751
|
*
|
|
@@ -798,6 +933,27 @@ var PoolWorker = class {
|
|
|
798
933
|
wireVersion = 1;
|
|
799
934
|
nextRequestId = 1;
|
|
800
935
|
ready = false;
|
|
936
|
+
/** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
|
|
937
|
+
* a poisoned worker is recycled, not revived. */
|
|
938
|
+
poisonVerdict = null;
|
|
939
|
+
/** The kill of a poisoned worker has been started. */
|
|
940
|
+
recycling = false;
|
|
941
|
+
/** Kills a poisoned worker whose drain did not finish in time. */
|
|
942
|
+
recycleBackstop = null;
|
|
943
|
+
/** A liveness probe is in flight — one at a time, never a probe storm. */
|
|
944
|
+
probeInFlight = false;
|
|
945
|
+
/** The child has exited (any cause). */
|
|
946
|
+
exited = false;
|
|
947
|
+
/** A compile past its soft bound, still running in the worker (D653). */
|
|
948
|
+
abandonedCompile = null;
|
|
949
|
+
/** Says when live inference has produced nothing for too long (D653). */
|
|
950
|
+
inferStall = new InferStallDetector(Date.now());
|
|
951
|
+
/**
|
|
952
|
+
* A load's HOST deadline expired with no reply since the last reply of any
|
|
953
|
+
* kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
|
|
954
|
+
* is over for this worker until something — anything — comes back.
|
|
955
|
+
*/
|
|
956
|
+
loadDeadlineMissed = false;
|
|
801
957
|
log;
|
|
802
958
|
opts;
|
|
803
959
|
constructor(opts) {
|
|
@@ -822,6 +978,24 @@ var PoolWorker = class {
|
|
|
822
978
|
isReady() {
|
|
823
979
|
return this.ready;
|
|
824
980
|
}
|
|
981
|
+
/** Why this worker was recycled, typed for the provider's budget, or `null`. */
|
|
982
|
+
getDeathCause() {
|
|
983
|
+
const v = this.poisonVerdict;
|
|
984
|
+
const message = this.getPoisonDescription();
|
|
985
|
+
if (v === null || message === null) return null;
|
|
986
|
+
return {
|
|
987
|
+
reason: v.reason,
|
|
988
|
+
chargesDeviceBudget: chargesDeviceBudget(v.reason),
|
|
989
|
+
message
|
|
990
|
+
};
|
|
991
|
+
}
|
|
992
|
+
/** Why this live worker was declared unusable, or `null` (D653). */
|
|
993
|
+
getPoisonDescription() {
|
|
994
|
+
const v = this.poisonVerdict;
|
|
995
|
+
if (v === null) return null;
|
|
996
|
+
const target = v.command.model !== null ? ` of ${v.command.model}` : "";
|
|
997
|
+
return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
|
|
998
|
+
}
|
|
825
999
|
async initialize(initialModels) {
|
|
826
1000
|
this.process = spawn(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
|
|
827
1001
|
"pipe",
|
|
@@ -845,7 +1019,11 @@ var PoolWorker = class {
|
|
|
845
1019
|
const spawnedProcess = this.process;
|
|
846
1020
|
spawnedProcess.on("exit", (code, signal) => {
|
|
847
1021
|
this.ready = false;
|
|
1022
|
+
this.exited = true;
|
|
1023
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1024
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
848
1025
|
if (this.process !== spawnedProcess) return;
|
|
1026
|
+
const poisonedBy = this.getPoisonDescription();
|
|
849
1027
|
this.log.error("Worker process exited", { meta: {
|
|
850
1028
|
worker: this.opts.workerLabel,
|
|
851
1029
|
pid: spawnedProcess.pid ?? null,
|
|
@@ -853,9 +1031,10 @@ var PoolWorker = class {
|
|
|
853
1031
|
device: this.opts.device ?? "default",
|
|
854
1032
|
code,
|
|
855
1033
|
signal,
|
|
856
|
-
inFlight: this.pending.size
|
|
1034
|
+
inFlight: this.pending.size,
|
|
1035
|
+
...poisonedBy !== null ? { poisonedBy } : {}
|
|
857
1036
|
} });
|
|
858
|
-
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
|
|
1037
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
|
|
859
1038
|
});
|
|
860
1039
|
this.process.stdout.on("data", (chunk) => {
|
|
861
1040
|
const t0 = Date.now();
|
|
@@ -869,6 +1048,7 @@ var PoolWorker = class {
|
|
|
869
1048
|
runtime: this.opts.poolRuntime,
|
|
870
1049
|
concurrency: this.opts.concurrency,
|
|
871
1050
|
protocolVersion: 2,
|
|
1051
|
+
modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
|
|
872
1052
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
873
1053
|
};
|
|
874
1054
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -891,6 +1071,7 @@ var PoolWorker = class {
|
|
|
891
1071
|
clearTimeout(timeout);
|
|
892
1072
|
if (result["status"] === "ready") {
|
|
893
1073
|
this.ready = true;
|
|
1074
|
+
this.inferStall.noteResult(Date.now());
|
|
894
1075
|
const loadedCount = result["models"];
|
|
895
1076
|
const startupMs = result["startupMs"];
|
|
896
1077
|
const workers = result["workers"] ?? 1;
|
|
@@ -913,7 +1094,7 @@ var PoolWorker = class {
|
|
|
913
1094
|
async infer(modelByte, jpeg, deviceId) {
|
|
914
1095
|
this.ensureReady();
|
|
915
1096
|
const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
|
|
916
|
-
return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
|
|
1097
|
+
return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
|
|
917
1098
|
}
|
|
918
1099
|
async inferRaw(modelByte, raw, width, height, format, deviceId) {
|
|
919
1100
|
this.ensureReady();
|
|
@@ -966,12 +1147,18 @@ var PoolWorker = class {
|
|
|
966
1147
|
const payload = Buffer.allocUnsafe(5);
|
|
967
1148
|
payload[0] = modelByte;
|
|
968
1149
|
payload.writeUInt32LE(frameId, 1);
|
|
969
|
-
return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
|
|
1150
|
+
return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
|
|
970
1151
|
}
|
|
971
1152
|
async sendCommand(cmd) {
|
|
972
1153
|
this.ensureReady();
|
|
1154
|
+
const command = describeCommand(cmd);
|
|
973
1155
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
974
|
-
|
|
1156
|
+
const sentAt = Date.now();
|
|
1157
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
|
|
1158
|
+
const raw = await this.dispatch(MSG_COMMAND, payload, { command });
|
|
1159
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
|
|
1160
|
+
this.inspectCommandReply(raw, command, sentAt);
|
|
1161
|
+
return raw;
|
|
975
1162
|
}
|
|
976
1163
|
/** Command whose reply shape is NOT the load/unload/replace/status envelope
|
|
977
1164
|
* (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
|
|
@@ -979,13 +1166,15 @@ var PoolWorker = class {
|
|
|
979
1166
|
async sendRawCommand(cmd) {
|
|
980
1167
|
this.ensureReady();
|
|
981
1168
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
982
|
-
return this.dispatch(MSG_COMMAND, payload);
|
|
1169
|
+
return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
|
|
983
1170
|
}
|
|
984
1171
|
async dispose() {
|
|
985
1172
|
const proc = this.process;
|
|
986
1173
|
if (!proc) return;
|
|
987
1174
|
this.process = null;
|
|
988
1175
|
this.ready = false;
|
|
1176
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1177
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
989
1178
|
await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
|
|
990
1179
|
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
|
|
991
1180
|
}
|
|
@@ -1028,8 +1217,11 @@ var PoolWorker = class {
|
|
|
1028
1217
|
}
|
|
1029
1218
|
/** Live inference gets seconds; commands and model loads keep the long
|
|
1030
1219
|
* timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
|
|
1031
|
-
deadlineFor(msgType) {
|
|
1032
|
-
|
|
1220
|
+
deadlineFor(msgType, opts = {}) {
|
|
1221
|
+
if (opts.timeoutMs !== void 0) return opts.timeoutMs;
|
|
1222
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
|
|
1223
|
+
if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
|
|
1224
|
+
return POOL_INFER_TIMEOUT_MS;
|
|
1033
1225
|
}
|
|
1034
1226
|
/**
|
|
1035
1227
|
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
@@ -1046,16 +1238,18 @@ var PoolWorker = class {
|
|
|
1046
1238
|
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1047
1239
|
return stamp;
|
|
1048
1240
|
}
|
|
1049
|
-
dispatch(msgType, payload,
|
|
1050
|
-
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1241
|
+
dispatch(msgType, payload, opts = {}) {
|
|
1242
|
+
const shed = this.shedIfSaturated(msgType, opts.deviceId);
|
|
1051
1243
|
if (shed) return Promise.resolve(shed);
|
|
1052
1244
|
const reqId = this.allocRequestId();
|
|
1053
1245
|
return new Promise((resolve, reject) => {
|
|
1054
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1246
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
|
|
1055
1247
|
this.pending.set(reqId, {
|
|
1056
1248
|
resolve,
|
|
1057
1249
|
reject,
|
|
1058
|
-
timer
|
|
1250
|
+
timer,
|
|
1251
|
+
...opts.command !== void 0 ? { command: opts.command } : {},
|
|
1252
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1059
1253
|
});
|
|
1060
1254
|
try {
|
|
1061
1255
|
const stamp = this.deadlineStamp(msgType);
|
|
@@ -1086,8 +1280,9 @@ var PoolWorker = class {
|
|
|
1086
1280
|
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1087
1281
|
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1088
1282
|
*/
|
|
1089
|
-
armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1090
|
-
const
|
|
1283
|
+
armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
|
|
1284
|
+
const { deviceId, command } = opts;
|
|
1285
|
+
const timeoutMs = this.deadlineFor(msgType, opts);
|
|
1091
1286
|
const timer = setTimeout(() => {
|
|
1092
1287
|
if (this.pending.delete(reqId)) {
|
|
1093
1288
|
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
@@ -1111,6 +1306,7 @@ var PoolWorker = class {
|
|
|
1111
1306
|
dropped: true,
|
|
1112
1307
|
shedReason: "deadline-expired"
|
|
1113
1308
|
});
|
|
1309
|
+
this.noteInferUnanswered();
|
|
1114
1310
|
return;
|
|
1115
1311
|
}
|
|
1116
1312
|
this.timedOutCount++;
|
|
@@ -1123,10 +1319,20 @@ var PoolWorker = class {
|
|
|
1123
1319
|
device: this.opts.device ?? "default",
|
|
1124
1320
|
inFlight: this.pending.size,
|
|
1125
1321
|
reqId,
|
|
1126
|
-
timeoutMs
|
|
1322
|
+
timeoutMs,
|
|
1323
|
+
...command !== void 0 ? {
|
|
1324
|
+
command: command.cmd,
|
|
1325
|
+
modelIndex: command.modelIndex,
|
|
1326
|
+
model: command.model
|
|
1327
|
+
} : {}
|
|
1127
1328
|
}
|
|
1128
1329
|
});
|
|
1129
|
-
|
|
1330
|
+
const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
|
|
1331
|
+
if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
|
|
1332
|
+
this.loadDeadlineMissed = true;
|
|
1333
|
+
reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
|
|
1334
|
+
} else reject(new Error(message));
|
|
1335
|
+
if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
|
|
1130
1336
|
}
|
|
1131
1337
|
}, timeoutMs);
|
|
1132
1338
|
timer.unref?.();
|
|
@@ -1137,11 +1343,12 @@ var PoolWorker = class {
|
|
|
1137
1343
|
if (shed) return Promise.resolve(shed);
|
|
1138
1344
|
const reqId = this.allocRequestId();
|
|
1139
1345
|
return new Promise((resolve, reject) => {
|
|
1140
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1346
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
|
|
1141
1347
|
this.pending.set(reqId, {
|
|
1142
1348
|
resolve,
|
|
1143
1349
|
reject,
|
|
1144
|
-
timer
|
|
1350
|
+
timer,
|
|
1351
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1145
1352
|
});
|
|
1146
1353
|
try {
|
|
1147
1354
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
@@ -1163,10 +1370,10 @@ var PoolWorker = class {
|
|
|
1163
1370
|
}
|
|
1164
1371
|
allocRequestId() {
|
|
1165
1372
|
let id = this.nextRequestId;
|
|
1166
|
-
this.nextRequestId = id >=
|
|
1373
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1167
1374
|
while (this.pending.has(id)) {
|
|
1168
1375
|
id = this.nextRequestId;
|
|
1169
|
-
this.nextRequestId = id >=
|
|
1376
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1170
1377
|
}
|
|
1171
1378
|
return id;
|
|
1172
1379
|
}
|
|
@@ -1181,6 +1388,8 @@ var PoolWorker = class {
|
|
|
1181
1388
|
this.process.stdin.write(payload);
|
|
1182
1389
|
}
|
|
1183
1390
|
ensureReady() {
|
|
1391
|
+
const poisonedBy = this.getPoisonDescription();
|
|
1392
|
+
if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
|
|
1184
1393
|
if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
|
|
1185
1394
|
}
|
|
1186
1395
|
/** Time spent in `Buffer.concat` since the last report. */
|
|
@@ -1220,23 +1429,273 @@ var PoolWorker = class {
|
|
|
1220
1429
|
const reqId = this.receiveBuffer.readUInt32LE(4);
|
|
1221
1430
|
const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
|
|
1222
1431
|
this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
|
|
1432
|
+
if (reqId === WORKER_EVENT_REQ_ID) {
|
|
1433
|
+
this.handleWorkerEvent(jsonBytes);
|
|
1434
|
+
continue;
|
|
1435
|
+
}
|
|
1223
1436
|
const entry = this.pending.get(reqId);
|
|
1224
1437
|
if (!entry) {
|
|
1225
1438
|
this.log.warn("Response for unknown request id", { meta: {
|
|
1226
1439
|
worker: this.opts.workerLabel,
|
|
1227
1440
|
reqId
|
|
1228
1441
|
} });
|
|
1442
|
+
this.noteLateReply(jsonBytes);
|
|
1229
1443
|
continue;
|
|
1230
1444
|
}
|
|
1231
1445
|
this.pending.delete(reqId);
|
|
1232
1446
|
if (entry.timer) clearTimeout(entry.timer);
|
|
1233
1447
|
try {
|
|
1234
1448
|
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1449
|
+
if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
|
|
1450
|
+
else this.inferStall.noteResult(Date.now());
|
|
1235
1451
|
entry.resolve(parsed);
|
|
1236
1452
|
} catch (err) {
|
|
1237
1453
|
entry.reject(err instanceof Error ? err : new Error(String(err)));
|
|
1238
1454
|
}
|
|
1239
1455
|
}
|
|
1456
|
+
this.noteAnyReply();
|
|
1457
|
+
if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
|
|
1458
|
+
}
|
|
1459
|
+
/**
|
|
1460
|
+
* Something came back from the worker: its loop was alive just now, so the
|
|
1461
|
+
* missed-deadline verdict is lifted.
|
|
1462
|
+
*/
|
|
1463
|
+
noteAnyReply() {
|
|
1464
|
+
this.loadDeadlineMissed = false;
|
|
1465
|
+
}
|
|
1466
|
+
/**
|
|
1467
|
+
* The invariant the load deadline depends on (D653 § 2): loads into one
|
|
1468
|
+
* pool are serialised, so at most ONE load is outstanding per worker. The
|
|
1469
|
+
* worker runs model commands in order and starts a load's soft bound only
|
|
1470
|
+
* when it begins, while the host's deadline for it runs from the SEND. A
|
|
1471
|
+
* load queued behind another could therefore see its host deadline fire
|
|
1472
|
+
* before the worker's named answer. Nothing in the provider issues that; if
|
|
1473
|
+
* anything ever does, this says so.
|
|
1474
|
+
*/
|
|
1475
|
+
warnIfLoadQueued(next) {
|
|
1476
|
+
const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
|
|
1477
|
+
if (ahead.length === 0) return;
|
|
1478
|
+
this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
|
|
1479
|
+
worker: this.opts.workerLabel,
|
|
1480
|
+
pid: this.getPid(),
|
|
1481
|
+
runtime: this.opts.poolRuntime,
|
|
1482
|
+
device: this.opts.device ?? "default",
|
|
1483
|
+
model: next.model,
|
|
1484
|
+
queuedBehind: ahead.map((c) => c.model)
|
|
1485
|
+
} });
|
|
1486
|
+
}
|
|
1487
|
+
hasPendingLoad() {
|
|
1488
|
+
for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
|
|
1489
|
+
return false;
|
|
1490
|
+
}
|
|
1491
|
+
/**
|
|
1492
|
+
* A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
|
|
1493
|
+
* or that an earlier one still has (`worker-poisoned`). The worker is not
|
|
1494
|
+
* recycled for it (fix round 1): the compile keeps running so a slow but
|
|
1495
|
+
* finite one writes its cache, the loaded models keep serving, and the worker
|
|
1496
|
+
* starts no new load. Only a compile still running at the HARD bound costs
|
|
1497
|
+
* the process.
|
|
1498
|
+
*/
|
|
1499
|
+
inspectCommandReply(raw, command, sentAt) {
|
|
1500
|
+
const reason = raw["reason"];
|
|
1501
|
+
if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
|
|
1502
|
+
if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
|
|
1503
|
+
const model = raw["modelId"];
|
|
1504
|
+
const abandoned = {
|
|
1505
|
+
...command,
|
|
1506
|
+
model: typeof model === "string" ? model : command.model
|
|
1507
|
+
};
|
|
1508
|
+
const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
|
|
1509
|
+
const reported = raw["compileElapsedMs"];
|
|
1510
|
+
const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
|
|
1511
|
+
const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
|
|
1512
|
+
hardTimer.unref?.();
|
|
1513
|
+
this.abandonedCompile = {
|
|
1514
|
+
command: abandoned,
|
|
1515
|
+
since: Date.now(),
|
|
1516
|
+
hardTimer
|
|
1517
|
+
};
|
|
1518
|
+
this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
|
|
1519
|
+
worker: this.opts.workerLabel,
|
|
1520
|
+
pid: this.getPid(),
|
|
1521
|
+
runtime: this.opts.poolRuntime,
|
|
1522
|
+
device: this.opts.device ?? "default",
|
|
1523
|
+
model: abandoned.model,
|
|
1524
|
+
modelIndex: abandoned.modelIndex,
|
|
1525
|
+
softMs: bounds.softMs,
|
|
1526
|
+
hardMs: bounds.hardMs,
|
|
1527
|
+
loads: "refused until it returns"
|
|
1528
|
+
} });
|
|
1529
|
+
}
|
|
1530
|
+
/** An unsolicited worker event (`compile-finished-late`). */
|
|
1531
|
+
handleWorkerEvent(jsonBytes) {
|
|
1532
|
+
let event;
|
|
1533
|
+
try {
|
|
1534
|
+
event = JSON.parse(jsonBytes.toString("utf8"));
|
|
1535
|
+
} catch {
|
|
1536
|
+
return;
|
|
1537
|
+
}
|
|
1538
|
+
if (event["event"] !== "compile-finished-late") return;
|
|
1539
|
+
const abandoned = this.abandonedCompile;
|
|
1540
|
+
if (abandoned !== null) clearTimeout(abandoned.hardTimer);
|
|
1541
|
+
this.abandonedCompile = null;
|
|
1542
|
+
this.inferStall.noteResult(Date.now());
|
|
1543
|
+
const ok = event["ok"] === true;
|
|
1544
|
+
const meta = {
|
|
1545
|
+
worker: this.opts.workerLabel,
|
|
1546
|
+
pid: this.getPid(),
|
|
1547
|
+
runtime: this.opts.poolRuntime,
|
|
1548
|
+
device: this.opts.device ?? "default",
|
|
1549
|
+
model: typeof event["modelId"] === "string" ? event["modelId"] : null,
|
|
1550
|
+
elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
|
|
1551
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1552
|
+
};
|
|
1553
|
+
if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
|
|
1554
|
+
else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
|
|
1555
|
+
this.opts.onCompileFinishedLate?.({
|
|
1556
|
+
model: meta.model,
|
|
1557
|
+
ok,
|
|
1558
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1559
|
+
});
|
|
1560
|
+
}
|
|
1561
|
+
/**
|
|
1562
|
+
* The GIL grace (D653 round 2): may a silent loop be a compile holding the
|
|
1563
|
+
* GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
|
|
1564
|
+
* on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
|
|
1565
|
+
*
|
|
1566
|
+
* NOT while a compile runs past its soft bound: the worker REPLIED
|
|
1567
|
+
* `compile-timeout`, so its loop is proven alive, and an inference hang
|
|
1568
|
+
* behind that compile — the 2026-09-26 incident exactly — must be seen on
|
|
1569
|
+
* the normal rule, not ~600-900 s later at the hard bound.
|
|
1570
|
+
*/
|
|
1571
|
+
gilGraceActive() {
|
|
1572
|
+
if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
|
|
1573
|
+
if (this.loadDeadlineMissed) return false;
|
|
1574
|
+
return this.hasPendingLoad();
|
|
1575
|
+
}
|
|
1576
|
+
/**
|
|
1577
|
+
* The reason a silent worker is recycled with: a freeze that began with a
|
|
1578
|
+
* load missing its deadline is that compile's doing (`compile-hung`, charged
|
|
1579
|
+
* to the model, not the device); anything else is the device's.
|
|
1580
|
+
*/
|
|
1581
|
+
silenceReason(fallback) {
|
|
1582
|
+
return this.loadDeadlineMissed ? "compile-hung" : fallback;
|
|
1583
|
+
}
|
|
1584
|
+
/** A live request ended with no result: judge the executor (D653, item 4). */
|
|
1585
|
+
noteInferUnanswered() {
|
|
1586
|
+
const stall = this.inferStall.noteUnanswered(Date.now());
|
|
1587
|
+
if (stall === null || this.gilGraceActive()) return;
|
|
1588
|
+
this.poison(this.silenceReason("infer-unresponsive"), {
|
|
1589
|
+
cmd: "infer",
|
|
1590
|
+
modelIndex: null,
|
|
1591
|
+
model: null
|
|
1592
|
+
}, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
|
|
1593
|
+
}
|
|
1594
|
+
/** A reply whose request was already abandoned: a real result still proves the executor runs. */
|
|
1595
|
+
noteLateReply(jsonBytes) {
|
|
1596
|
+
try {
|
|
1597
|
+
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1598
|
+
if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
|
|
1599
|
+
} catch {}
|
|
1600
|
+
}
|
|
1601
|
+
/**
|
|
1602
|
+
* Ask a worker whose command just timed out whether its loop still answers.
|
|
1603
|
+
* `mem_stats` is served on the loop, never behind a compile, so a worker
|
|
1604
|
+
* that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
|
|
1605
|
+
* slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
|
|
1606
|
+
* `mem_stats` lost while the process looked alive.
|
|
1607
|
+
*/
|
|
1608
|
+
probeLiveness(timedOut) {
|
|
1609
|
+
if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
|
|
1610
|
+
this.probeInFlight = true;
|
|
1611
|
+
this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
|
|
1612
|
+
command: {
|
|
1613
|
+
cmd: "mem_stats",
|
|
1614
|
+
modelIndex: null,
|
|
1615
|
+
model: null
|
|
1616
|
+
},
|
|
1617
|
+
timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
|
|
1618
|
+
probe: true
|
|
1619
|
+
}).then(() => {
|
|
1620
|
+
this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
|
|
1621
|
+
worker: this.opts.workerLabel,
|
|
1622
|
+
pid: this.getPid(),
|
|
1623
|
+
runtime: this.opts.poolRuntime,
|
|
1624
|
+
device: this.opts.device ?? "default",
|
|
1625
|
+
command: timedOut.cmd,
|
|
1626
|
+
modelIndex: timedOut.modelIndex,
|
|
1627
|
+
model: timedOut.model
|
|
1628
|
+
} });
|
|
1629
|
+
}, (err) => {
|
|
1630
|
+
if (this.gilGraceActive()) {
|
|
1631
|
+
this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
|
|
1632
|
+
worker: this.opts.workerLabel,
|
|
1633
|
+
pid: this.getPid(),
|
|
1634
|
+
runtime: this.opts.poolRuntime,
|
|
1635
|
+
device: this.opts.device ?? "default",
|
|
1636
|
+
command: timedOut.cmd,
|
|
1637
|
+
model: timedOut.model
|
|
1638
|
+
} });
|
|
1639
|
+
return;
|
|
1640
|
+
}
|
|
1641
|
+
this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1642
|
+
}).finally(() => {
|
|
1643
|
+
this.probeInFlight = false;
|
|
1644
|
+
});
|
|
1645
|
+
}
|
|
1646
|
+
/**
|
|
1647
|
+
* Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
|
|
1648
|
+
* and recycle it once it has drained.
|
|
1649
|
+
*
|
|
1650
|
+
* `ready = false` is what hands the pool back to the provider: its next
|
|
1651
|
+
* dispatch finds the factory not ready and condemns it under the per-device
|
|
1652
|
+
* restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
|
|
1653
|
+
* excludes) — the same path a crashed worker takes, except that a
|
|
1654
|
+
* `compile-hung` death is not charged to the device (see PoolPoisonReason).
|
|
1655
|
+
* An unresponsive worker is killed at once (it will answer nothing it
|
|
1656
|
+
* holds); a compile-hung one keeps serving its loaded models until its
|
|
1657
|
+
* in-flight requests are answered, bounded by the live deadline.
|
|
1658
|
+
*/
|
|
1659
|
+
poison(reason, command, detail) {
|
|
1660
|
+
if (this.poisonVerdict !== null || this.exited || this.process === null) return;
|
|
1661
|
+
this.poisonVerdict = {
|
|
1662
|
+
reason,
|
|
1663
|
+
command,
|
|
1664
|
+
detail,
|
|
1665
|
+
pid: this.getPid()
|
|
1666
|
+
};
|
|
1667
|
+
this.ready = false;
|
|
1668
|
+
const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
|
|
1669
|
+
this.log.error("pool worker POISONED — recycling it", { meta: {
|
|
1670
|
+
worker: this.opts.workerLabel,
|
|
1671
|
+
pid: this.getPid(),
|
|
1672
|
+
runtime: this.opts.poolRuntime,
|
|
1673
|
+
device: this.opts.device ?? "default",
|
|
1674
|
+
reason,
|
|
1675
|
+
command: command.cmd,
|
|
1676
|
+
modelIndex: command.modelIndex,
|
|
1677
|
+
model: command.model,
|
|
1678
|
+
detail,
|
|
1679
|
+
inFlight: this.pending.size,
|
|
1680
|
+
inFlightCommands
|
|
1681
|
+
} });
|
|
1682
|
+
if (reason !== "compile-hung" || this.pending.size === 0) {
|
|
1683
|
+
this.recycle();
|
|
1684
|
+
return;
|
|
1685
|
+
}
|
|
1686
|
+
this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
|
|
1687
|
+
this.recycleBackstop.unref?.();
|
|
1688
|
+
}
|
|
1689
|
+
/**
|
|
1690
|
+
* Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
|
|
1691
|
+
* reported and rejects whatever it still held — a deliberate dispose would
|
|
1692
|
+
* silence both, and this is not one.
|
|
1693
|
+
*/
|
|
1694
|
+
recycle() {
|
|
1695
|
+
if (this.recycling || this.exited || this.process === null) return;
|
|
1696
|
+
this.recycling = true;
|
|
1697
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1698
|
+
terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
|
|
1240
1699
|
}
|
|
1241
1700
|
rejectAll(err) {
|
|
1242
1701
|
const entries = [...this.pending.values()];
|
|
@@ -1283,8 +1742,10 @@ var SharedInferencePool = class {
|
|
|
1283
1742
|
this.tuning = options.tuning ?? null;
|
|
1284
1743
|
this.numWorkers = Math.max(1, options.numWorkers ?? 1);
|
|
1285
1744
|
this.device = options.device;
|
|
1745
|
+
this.onCompileFinishedLate = options.onCompileFinishedLate;
|
|
1286
1746
|
}
|
|
1287
1747
|
device;
|
|
1748
|
+
onCompileFinishedLate;
|
|
1288
1749
|
/** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
|
|
1289
1750
|
/** Summed backlog and shed count across the pool's workers. A rising
|
|
1290
1751
|
* `inFlight` with a rising `shed` is a worker falling behind; a rising
|
|
@@ -1323,7 +1784,8 @@ var SharedInferencePool = class {
|
|
|
1323
1784
|
tuning: this.tuning,
|
|
1324
1785
|
logger: this.log,
|
|
1325
1786
|
workerLabel: `w${i}`,
|
|
1326
|
-
...this.device ? { device: this.device } : {}
|
|
1787
|
+
...this.device ? { device: this.device } : {},
|
|
1788
|
+
...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
|
|
1327
1789
|
}));
|
|
1328
1790
|
const t0 = performance.now();
|
|
1329
1791
|
const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
|
|
@@ -1418,7 +1880,7 @@ var SharedInferencePool = class {
|
|
|
1418
1880
|
index,
|
|
1419
1881
|
config: serializeModelConfig(config)
|
|
1420
1882
|
})));
|
|
1421
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
1883
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1422
1884
|
if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
|
|
1423
1885
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1424
1886
|
}
|
|
@@ -1454,7 +1916,7 @@ var SharedInferencePool = class {
|
|
|
1454
1916
|
index,
|
|
1455
1917
|
config: serializeModelConfig(config)
|
|
1456
1918
|
})));
|
|
1457
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
1919
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1458
1920
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1459
1921
|
}
|
|
1460
1922
|
/**
|
|
@@ -1504,9 +1966,35 @@ var SharedInferencePool = class {
|
|
|
1504
1966
|
allocateIndex() {
|
|
1505
1967
|
return this.nextFreeIndex++;
|
|
1506
1968
|
}
|
|
1969
|
+
/**
|
|
1970
|
+
* Give back an index whose load FAILED, so the retry reuses it. Only the most
|
|
1971
|
+
* recent allocation can be returned. Loads into one pool are serialised by
|
|
1972
|
+
* the provider, so a failed load is normally the latest one; anything else
|
|
1973
|
+
* is left allocated rather than risk handing out a live slot twice.
|
|
1974
|
+
*/
|
|
1975
|
+
releaseIndex(index) {
|
|
1976
|
+
if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
|
|
1977
|
+
}
|
|
1507
1978
|
isReady() {
|
|
1508
1979
|
return this.workers.length > 0 && this.workers.every((w) => w.isReady());
|
|
1509
1980
|
}
|
|
1981
|
+
/**
|
|
1982
|
+
* Why a worker of this pool was declared unusable while alive (D653), or
|
|
1983
|
+
* `null`. The provider charges the restart budget with THIS — or, for a
|
|
1984
|
+
* `compile-hung` death, does not charge the device at all — so the
|
|
1985
|
+
* `inference device FAILED` line names the cause instead of "pool worker is
|
|
1986
|
+
* not ready".
|
|
1987
|
+
*/
|
|
1988
|
+
getDeathCause() {
|
|
1989
|
+
for (const w of this.workers) {
|
|
1990
|
+
const cause = w.getDeathCause();
|
|
1991
|
+
if (cause !== null) return {
|
|
1992
|
+
...cause,
|
|
1993
|
+
message: `worker ${cause.message}`
|
|
1994
|
+
};
|
|
1995
|
+
}
|
|
1996
|
+
return null;
|
|
1997
|
+
}
|
|
1510
1998
|
async dispose() {
|
|
1511
1999
|
await Promise.all(this.workers.map((w) => w.dispose()));
|
|
1512
2000
|
this.workers.length = 0;
|
|
@@ -1571,6 +2059,23 @@ var SharedInferencePool = class {
|
|
|
1571
2059
|
return found;
|
|
1572
2060
|
}
|
|
1573
2061
|
};
|
|
2062
|
+
/** A failed command's reply as one message: its reason class first, when it has one. */
|
|
2063
|
+
function describeCommandFailure(resp) {
|
|
2064
|
+
const error = resp.error ?? "unknown";
|
|
2065
|
+
return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
|
|
2066
|
+
}
|
|
2067
|
+
/** Name a command for the lines that must say which one hung (D653). */
|
|
2068
|
+
function describeCommand(cmd) {
|
|
2069
|
+
const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
|
|
2070
|
+
const index = cmd["index"];
|
|
2071
|
+
const config = cmd["config"];
|
|
2072
|
+
const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
|
|
2073
|
+
return {
|
|
2074
|
+
cmd: name,
|
|
2075
|
+
modelIndex: typeof index === "number" ? index : null,
|
|
2076
|
+
model: typeof modelPath === "string" && modelPath.length > 0 ? path$1.basename(modelPath, path$1.extname(modelPath)) : null
|
|
2077
|
+
};
|
|
2078
|
+
}
|
|
1574
2079
|
function serializeModelConfig(config) {
|
|
1575
2080
|
const result = {
|
|
1576
2081
|
path: config.path,
|
|
@@ -1696,12 +2201,13 @@ function resolveEffectivePostprocessorForEntry(definition, entry) {
|
|
|
1696
2201
|
//#region src/detection-pipeline/registry/model-knob-applicability.ts
|
|
1697
2202
|
/** Decodes that run NMS in the pipeline (and hence consult `nmsIouThreshold`).
|
|
1698
2203
|
* Mirrors the Python postprocessors that read the knob — yolo.py,
|
|
1699
|
-
* yolo_seg.py, rfdetr.py, scrfd.py. */
|
|
2204
|
+
* yolo_seg.py, rfdetr.py, scrfd.py, yunet.py. */
|
|
1700
2205
|
var NMS_CONSUMING_POSTPROCESSORS = new Set([
|
|
1701
2206
|
"yolo",
|
|
1702
2207
|
"yolo-seg",
|
|
1703
2208
|
"rfdetr",
|
|
1704
|
-
"scrfd"
|
|
2209
|
+
"scrfd",
|
|
2210
|
+
"yunet"
|
|
1705
2211
|
]);
|
|
1706
2212
|
/**
|
|
1707
2213
|
* Does this decode consult `nmsIouThreshold`?
|
|
@@ -1857,6 +2363,28 @@ function poolDecodeSettingsEqual(a, b) {
|
|
|
1857
2363
|
}
|
|
1858
2364
|
//#endregion
|
|
1859
2365
|
//#region src/detection-pipeline/engine/pipeline-model-manager.ts
|
|
2366
|
+
/**
|
|
2367
|
+
* A (step, model) load the pool rejected — carrying the pool's failure class
|
|
2368
|
+
* (`compile-timeout`, …) so the provider can count a timeout against THAT
|
|
2369
|
+
* model without parsing a message (D653).
|
|
2370
|
+
*/
|
|
2371
|
+
var StepVariantLoadError = class extends Error {
|
|
2372
|
+
stepId;
|
|
2373
|
+
modelId;
|
|
2374
|
+
poolIndex;
|
|
2375
|
+
reason;
|
|
2376
|
+
/** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
|
|
2377
|
+
poolModel;
|
|
2378
|
+
constructor(stepId, modelId, poolIndex, cause) {
|
|
2379
|
+
super(cause instanceof Error ? cause.message : String(cause));
|
|
2380
|
+
this.name = "StepVariantLoadError";
|
|
2381
|
+
this.stepId = stepId;
|
|
2382
|
+
this.modelId = modelId;
|
|
2383
|
+
this.poolIndex = poolIndex;
|
|
2384
|
+
this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
|
|
2385
|
+
this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
|
|
2386
|
+
}
|
|
2387
|
+
};
|
|
1860
2388
|
var PipelineModelManager = class {
|
|
1861
2389
|
pool;
|
|
1862
2390
|
source;
|
|
@@ -1868,11 +2396,20 @@ var PipelineModelManager = class {
|
|
|
1868
2396
|
lruClock = 0;
|
|
1869
2397
|
log;
|
|
1870
2398
|
maxModelsPerStep;
|
|
2399
|
+
deviceKey;
|
|
2400
|
+
/**
|
|
2401
|
+
* Loads issued and not yet answered, keyed `stepId::modelId`. A second
|
|
2402
|
+
* caller for the same pair JOINS the pending load instead of issuing its own
|
|
2403
|
+
* (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
|
|
2404
|
+
* compile, each allocating a pool index of its own.
|
|
2405
|
+
*/
|
|
2406
|
+
pendingLoads = /* @__PURE__ */ new Map();
|
|
1871
2407
|
constructor(pool, source, logger, options) {
|
|
1872
2408
|
this.pool = pool;
|
|
1873
2409
|
this.source = source;
|
|
1874
2410
|
this.log = logger;
|
|
1875
2411
|
this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
|
|
2412
|
+
this.deviceKey = options?.deviceKey ?? null;
|
|
1876
2413
|
}
|
|
1877
2414
|
/**
|
|
1878
2415
|
* Apply a new pipeline configuration — driven by the runtime config
|
|
@@ -2022,6 +2559,23 @@ var PipelineModelManager = class {
|
|
|
2022
2559
|
await this.reconcileDecode(existing, settings);
|
|
2023
2560
|
return existing;
|
|
2024
2561
|
}
|
|
2562
|
+
const key = `${stepId}::${modelId}`;
|
|
2563
|
+
const pending = this.pendingLoads.get(key);
|
|
2564
|
+
if (pending) {
|
|
2565
|
+
const joined = await pending;
|
|
2566
|
+
await this.reconcileDecode(joined, settings);
|
|
2567
|
+
return joined;
|
|
2568
|
+
}
|
|
2569
|
+
const load = this.issueLoad(perStep, stepId, modelId, settings);
|
|
2570
|
+
this.pendingLoads.set(key, load);
|
|
2571
|
+
try {
|
|
2572
|
+
return await load;
|
|
2573
|
+
} finally {
|
|
2574
|
+
this.pendingLoads.delete(key);
|
|
2575
|
+
}
|
|
2576
|
+
}
|
|
2577
|
+
/** Evict if at capacity, then send ONE load command and record the result. */
|
|
2578
|
+
async issueLoad(perStep, stepId, modelId, settings) {
|
|
2025
2579
|
while (perStep.size >= this.maxModelsPerStep) {
|
|
2026
2580
|
const evicted = this.pickEvictionTarget(stepId);
|
|
2027
2581
|
if (!evicted) break;
|
|
@@ -2033,16 +2587,31 @@ var PipelineModelManager = class {
|
|
|
2033
2587
|
cap: this.maxModelsPerStep
|
|
2034
2588
|
} });
|
|
2035
2589
|
}
|
|
2036
|
-
const index = this.pool.allocateIndex();
|
|
2037
2590
|
const config = this.source.buildConfig(stepId, modelId, settings);
|
|
2038
2591
|
const decode = poolDecodeSettingsOf(config);
|
|
2592
|
+
const index = this.pool.allocateIndex();
|
|
2039
2593
|
this.log.info("Loading step variant", { meta: {
|
|
2040
2594
|
step: stepId,
|
|
2041
2595
|
modelId,
|
|
2042
2596
|
poolIndex: index,
|
|
2597
|
+
deviceKey: this.deviceKey,
|
|
2043
2598
|
...decode
|
|
2044
2599
|
} });
|
|
2045
|
-
|
|
2600
|
+
let loadMs;
|
|
2601
|
+
try {
|
|
2602
|
+
({loadMs} = await this.pool.loadModel(index, config));
|
|
2603
|
+
} catch (err) {
|
|
2604
|
+
this.pool.releaseIndex(index);
|
|
2605
|
+
this.log.error("Step variant load failed", { meta: {
|
|
2606
|
+
step: stepId,
|
|
2607
|
+
modelId,
|
|
2608
|
+
poolIndex: index,
|
|
2609
|
+
deviceKey: this.deviceKey,
|
|
2610
|
+
reason: err instanceof PoolModelLoadError ? err.reason : null,
|
|
2611
|
+
error: err instanceof Error ? err.message : String(err)
|
|
2612
|
+
} });
|
|
2613
|
+
throw new StepVariantLoadError(stepId, modelId, index, err);
|
|
2614
|
+
}
|
|
2046
2615
|
this.log.info("Step variant loaded", { meta: {
|
|
2047
2616
|
step: stepId,
|
|
2048
2617
|
modelId,
|
|
@@ -2412,6 +2981,15 @@ var EngineFactory = class {
|
|
|
2412
2981
|
isReady() {
|
|
2413
2982
|
return this.pool?.isReady() ?? false;
|
|
2414
2983
|
}
|
|
2984
|
+
/**
|
|
2985
|
+
* Why this factory's pool stopped being usable while its process was alive —
|
|
2986
|
+
* a compile that never returned, or a worker that answered nothing (D653) —
|
|
2987
|
+
* or `null`. A pool that merely crashed says nothing here; its `Worker
|
|
2988
|
+
* process exited` line already names the signal.
|
|
2989
|
+
*/
|
|
2990
|
+
getDeathCause() {
|
|
2991
|
+
return this.pool?.getDeathCause() ?? null;
|
|
2992
|
+
}
|
|
2415
2993
|
/** Native pid of the underlying Python pool, if any. */
|
|
2416
2994
|
getPoolPid() {
|
|
2417
2995
|
return this.pool?.getPid() ?? null;
|
|
@@ -2538,12 +3116,13 @@ var EngineFactory = class {
|
|
|
2538
3116
|
concurrency,
|
|
2539
3117
|
tuning: resolvedTuning,
|
|
2540
3118
|
numWorkers,
|
|
2541
|
-
...this.opts.engine.device ? { device: this.opts.engine.device } : {}
|
|
3119
|
+
...this.opts.engine.device ? { device: this.opts.engine.device } : {},
|
|
3120
|
+
...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
|
|
2542
3121
|
});
|
|
2543
3122
|
this.poolManager = new PipelineModelManager(this.pool, {
|
|
2544
3123
|
buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
|
|
2545
3124
|
resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
|
|
2546
|
-
}, this.log.child("model-mgr"));
|
|
3125
|
+
}, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
|
|
2547
3126
|
await this.pool.initialize([]);
|
|
2548
3127
|
await this.poolManager.applyConfig(steps);
|
|
2549
3128
|
}
|
|
@@ -2659,6 +3238,175 @@ function buildPoolModelConfigForStep(inputs) {
|
|
|
2659
3238
|
};
|
|
2660
3239
|
}
|
|
2661
3240
|
//#endregion
|
|
3241
|
+
//#region src/detection-pipeline/engine/model-load-governor.ts
|
|
3242
|
+
/**
|
|
3243
|
+
* What the dispatch path may ask of a pool's model loads, and what it must say
|
|
3244
|
+
* when it may not (D653, fix round 1).
|
|
3245
|
+
*
|
|
3246
|
+
* Three pieces of state, consulted ON DEMAND by the dispatch that needs a
|
|
3247
|
+
* model — there is no timer anywhere in here:
|
|
3248
|
+
*
|
|
3249
|
+
* 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
|
|
3250
|
+
* failed load, so without it every frame of every camera re-issued the
|
|
3251
|
+
* failing load and wrote two ERRORs. After a failure the model is not
|
|
3252
|
+
* re-issued on that pool for an interval that doubles per consecutive
|
|
3253
|
+
* failure up to a cap; a success clears it.
|
|
3254
|
+
* 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
|
|
3255
|
+
* bound is not a crash of the device. The SAME model timing out
|
|
3256
|
+
* {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
|
|
3257
|
+
* since a slow compile now runs on and writes its cache, only a compile
|
|
3258
|
+
* that never finishes can do — refuses THAT MODEL there, by name. The
|
|
3259
|
+
* device keeps serving every other model.
|
|
3260
|
+
* 3. Which camera has already been TOLD about the current state of a model,
|
|
3261
|
+
* so each camera gets one line per state change, never one per frame.
|
|
3262
|
+
*/
|
|
3263
|
+
/** The first back-off after a failed load. */
|
|
3264
|
+
var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
|
|
3265
|
+
/** The back-off doubles per consecutive failure up to this. */
|
|
3266
|
+
var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
|
|
3267
|
+
/** Timeouts older than this no longer count toward a refusal. */
|
|
3268
|
+
var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
|
|
3269
|
+
var ModelLoadGovernor = class {
|
|
3270
|
+
/** Keyed by the pool OBJECT, so a respawned pool starts clean. */
|
|
3271
|
+
backoff = /* @__PURE__ */ new WeakMap();
|
|
3272
|
+
timeouts = /* @__PURE__ */ new Map();
|
|
3273
|
+
reported = /* @__PURE__ */ new Map();
|
|
3274
|
+
generation = 0;
|
|
3275
|
+
/** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
|
|
3276
|
+
check(pool, deviceKey, models, now = Date.now()) {
|
|
3277
|
+
for (const model of models) {
|
|
3278
|
+
const t = this.timeouts.get(timeoutKey(deviceKey, model));
|
|
3279
|
+
if (t?.refused === true) return {
|
|
3280
|
+
kind: "refused",
|
|
3281
|
+
model,
|
|
3282
|
+
timeouts: t.at.length,
|
|
3283
|
+
error: t.lastError
|
|
3284
|
+
};
|
|
3285
|
+
}
|
|
3286
|
+
const perPool = this.backoff.get(pool);
|
|
3287
|
+
if (perPool === void 0) return { kind: "go" };
|
|
3288
|
+
for (const model of models) {
|
|
3289
|
+
const b = perPool.get(model);
|
|
3290
|
+
if (b !== void 0 && now < b.retryAtMs) return {
|
|
3291
|
+
kind: "backoff",
|
|
3292
|
+
model,
|
|
3293
|
+
retryInMs: b.retryAtMs - now,
|
|
3294
|
+
failures: b.failures,
|
|
3295
|
+
error: b.error,
|
|
3296
|
+
generation: b.generation
|
|
3297
|
+
};
|
|
3298
|
+
}
|
|
3299
|
+
return { kind: "go" };
|
|
3300
|
+
}
|
|
3301
|
+
/** One load of `model` on `pool` failed. */
|
|
3302
|
+
recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
|
|
3303
|
+
let perPool = this.backoff.get(pool);
|
|
3304
|
+
if (perPool === void 0) {
|
|
3305
|
+
perPool = /* @__PURE__ */ new Map();
|
|
3306
|
+
this.backoff.set(pool, perPool);
|
|
3307
|
+
}
|
|
3308
|
+
const previousEntry = perPool.get(model);
|
|
3309
|
+
const failures = (previousEntry?.failures ?? 0) + 1;
|
|
3310
|
+
const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
|
|
3311
|
+
this.generation += 1;
|
|
3312
|
+
const generation = this.generation;
|
|
3313
|
+
perPool.set(model, {
|
|
3314
|
+
failures,
|
|
3315
|
+
retryAtMs: now + retryInMs,
|
|
3316
|
+
error,
|
|
3317
|
+
generation,
|
|
3318
|
+
poolModel: poolModel ?? previousEntry?.poolModel ?? null,
|
|
3319
|
+
reason
|
|
3320
|
+
});
|
|
3321
|
+
if (!compileTimeout) return {
|
|
3322
|
+
generation,
|
|
3323
|
+
retryInMs,
|
|
3324
|
+
refusedNow: false,
|
|
3325
|
+
compileTimeouts: 0
|
|
3326
|
+
};
|
|
3327
|
+
const key = timeoutKey(deviceKey, model);
|
|
3328
|
+
const previous = this.timeouts.get(key);
|
|
3329
|
+
const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
|
|
3330
|
+
const refused = at.length >= 2;
|
|
3331
|
+
this.timeouts.set(key, {
|
|
3332
|
+
at,
|
|
3333
|
+
refused,
|
|
3334
|
+
lastError: error
|
|
3335
|
+
});
|
|
3336
|
+
return {
|
|
3337
|
+
generation,
|
|
3338
|
+
retryInMs,
|
|
3339
|
+
refusedNow: refused && previous?.refused !== true,
|
|
3340
|
+
compileTimeouts: at.length
|
|
3341
|
+
};
|
|
3342
|
+
}
|
|
3343
|
+
/**
|
|
3344
|
+
* A compile that outlived its bound came back on `pool` (D653 round 3).
|
|
3345
|
+
*
|
|
3346
|
+
* `ok`: THAT model's cache is written and the worker loads again, so its
|
|
3347
|
+
* back-off is lifted — and only its: another model's failure on the same
|
|
3348
|
+
* pool is not the compile that just finished. Not ok: it is one more
|
|
3349
|
+
* failure of that model, and its back-off grows. Refusals by name are never
|
|
3350
|
+
* lifted here; a model refused for timing out twice stays refused until the
|
|
3351
|
+
* operator re-arms. Returns the governor keys it touched.
|
|
3352
|
+
*/
|
|
3353
|
+
settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
|
|
3354
|
+
const perPool = this.backoff.get(pool);
|
|
3355
|
+
if (perPool === void 0) return [];
|
|
3356
|
+
const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
|
|
3357
|
+
for (const key of refusedMeanwhile) perPool.delete(key);
|
|
3358
|
+
const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
|
|
3359
|
+
for (const key of own) if (ok) perPool.delete(key);
|
|
3360
|
+
else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
|
|
3361
|
+
return [...refusedMeanwhile, ...own];
|
|
3362
|
+
}
|
|
3363
|
+
/** `models` loaded on `pool`: forget their failures there. */
|
|
3364
|
+
recordSuccess(pool, deviceKey, models) {
|
|
3365
|
+
const perPool = this.backoff.get(pool);
|
|
3366
|
+
for (const model of models) {
|
|
3367
|
+
perPool?.delete(model);
|
|
3368
|
+
this.timeouts.delete(timeoutKey(deviceKey, model));
|
|
3369
|
+
}
|
|
3370
|
+
}
|
|
3371
|
+
/**
|
|
3372
|
+
* Has `camera` been told about this `state` of `model` on `deviceKey` yet?
|
|
3373
|
+
* Returns `true` exactly once per state change.
|
|
3374
|
+
*/
|
|
3375
|
+
shouldReport(camera, deviceKey, model, state) {
|
|
3376
|
+
const key = `${camera ?? "-"}|${deviceKey}|${model}`;
|
|
3377
|
+
if (this.reported.get(key) === state) return false;
|
|
3378
|
+
this.reported.set(key, state);
|
|
3379
|
+
return true;
|
|
3380
|
+
}
|
|
3381
|
+
/** Every model refused on some device. */
|
|
3382
|
+
refused() {
|
|
3383
|
+
const out = [];
|
|
3384
|
+
for (const [key, t] of this.timeouts) {
|
|
3385
|
+
if (!t.refused) continue;
|
|
3386
|
+
const [deviceKey = "", model = ""] = key.split("\0");
|
|
3387
|
+
out.push({
|
|
3388
|
+
deviceKey,
|
|
3389
|
+
model,
|
|
3390
|
+
timeouts: t.at.length,
|
|
3391
|
+
lastError: t.lastError
|
|
3392
|
+
});
|
|
3393
|
+
}
|
|
3394
|
+
return out;
|
|
3395
|
+
}
|
|
3396
|
+
/** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
|
|
3397
|
+
rearm(deviceKey) {
|
|
3398
|
+
let cleared = 0;
|
|
3399
|
+
for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
|
|
3400
|
+
this.timeouts.delete(key);
|
|
3401
|
+
cleared += 1;
|
|
3402
|
+
}
|
|
3403
|
+
return cleared;
|
|
3404
|
+
}
|
|
3405
|
+
};
|
|
3406
|
+
function timeoutKey(deviceKey, model) {
|
|
3407
|
+
return `${deviceKey}\u0000${model}`;
|
|
3408
|
+
}
|
|
3409
|
+
//#endregion
|
|
2662
3410
|
//#region src/detection-pipeline/engine/idle-pool-reaper.ts
|
|
2663
3411
|
var IdlePoolReaper = class {
|
|
2664
3412
|
lastUsed = /* @__PURE__ */ new Map();
|
|
@@ -3158,6 +3906,52 @@ function createInferenceTimeoutGuard(deps) {
|
|
|
3158
3906
|
};
|
|
3159
3907
|
}
|
|
3160
3908
|
//#endregion
|
|
3909
|
+
//#region src/detection-pipeline/model-pin.ts
|
|
3910
|
+
function requestedModels(steps) {
|
|
3911
|
+
const out = /* @__PURE__ */ new Map();
|
|
3912
|
+
const walk = (nodes) => {
|
|
3913
|
+
for (const step of nodes) {
|
|
3914
|
+
if (!step.enabled) continue;
|
|
3915
|
+
out.set(step.addonId, step.modelId);
|
|
3916
|
+
if (step.children !== void 0) walk(step.children);
|
|
3917
|
+
}
|
|
3918
|
+
};
|
|
3919
|
+
walk(steps);
|
|
3920
|
+
return out;
|
|
3921
|
+
}
|
|
3922
|
+
function checkReplayPin(input) {
|
|
3923
|
+
if (input.requestedDeviceKey !== void 0 && input.dispatchDeviceKey !== input.requestedDeviceKey) return {
|
|
3924
|
+
kind: "refused",
|
|
3925
|
+
reason: "device-not-servable",
|
|
3926
|
+
detail: `device pool ${input.requestedDeviceKey} cannot run this tree (the capability gate would fall back to ${input.dispatchDeviceKey ?? "the node default"})`
|
|
3927
|
+
};
|
|
3928
|
+
const asked = requestedModels(input.requested);
|
|
3929
|
+
for (const step of input.executing) {
|
|
3930
|
+
const want = asked.get(step.addonId);
|
|
3931
|
+
if (want === void 0 || want.length === 0) return {
|
|
3932
|
+
kind: "refused",
|
|
3933
|
+
reason: "model-not-servable",
|
|
3934
|
+
detail: `${step.addonId}: no model pinned by the caller (would run ${step.modelId}, ${input.format})`
|
|
3935
|
+
};
|
|
3936
|
+
if (want !== step.modelId) return {
|
|
3937
|
+
kind: "refused",
|
|
3938
|
+
reason: "model-not-servable",
|
|
3939
|
+
detail: `${step.addonId}: asked for ${want}, this pool would run ${step.modelId} (${input.format})`
|
|
3940
|
+
};
|
|
3941
|
+
}
|
|
3942
|
+
return { kind: "pinned" };
|
|
3943
|
+
}
|
|
3944
|
+
var PRIMARY_ROOT_STEP = "object-detection";
|
|
3945
|
+
/**
|
|
3946
|
+
* The root model a frame result reports: `object-detection` when it runs (it
|
|
3947
|
+
* is the detector the tracks come from), else the first enabled video root.
|
|
3948
|
+
* `undefined` when no root runs — never a fabricated id.
|
|
3949
|
+
*/
|
|
3950
|
+
function resolveRootModelId(roots) {
|
|
3951
|
+
const enabled = roots.filter((s) => s.enabled && s.slot !== "audio-classifier");
|
|
3952
|
+
return (enabled.find((s) => s.addonId === PRIMARY_ROOT_STEP) ?? enabled[0])?.modelId;
|
|
3953
|
+
}
|
|
3954
|
+
//#endregion
|
|
3161
3955
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
3162
3956
|
/** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
|
|
3163
3957
|
var BACKOFF_SCHEDULE_MS = [
|
|
@@ -3428,6 +4222,7 @@ var KNOWN_POSTPROCESSORS = new Set([
|
|
|
3428
4222
|
"rfdetr",
|
|
3429
4223
|
"yolo-seg",
|
|
3430
4224
|
"scrfd",
|
|
4225
|
+
"yunet",
|
|
3431
4226
|
"arcface",
|
|
3432
4227
|
"clip",
|
|
3433
4228
|
"softmax",
|
|
@@ -3529,9 +4324,12 @@ var DropSampler = class {
|
|
|
3529
4324
|
* accumulates and a summary is released once per interval — carrying the
|
|
3530
4325
|
* number suppressed, so nothing about the RATE is lost even though the lines
|
|
3531
4326
|
* are.
|
|
4327
|
+
*
|
|
4328
|
+
* `channel` separates occasions that share a device — a replay's drops are
|
|
4329
|
+
* sampled apart from the live ones.
|
|
3532
4330
|
*/
|
|
3533
|
-
record(deviceId, macroClass) {
|
|
3534
|
-
const key = `${deviceId}:${macroClass}`;
|
|
4331
|
+
record(deviceId, macroClass, channel = "live") {
|
|
4332
|
+
const key = `${channel}:${deviceId}:${macroClass}`;
|
|
3535
4333
|
const at = this.now();
|
|
3536
4334
|
const prev = this.seen.get(key);
|
|
3537
4335
|
if (prev === void 0) {
|
|
@@ -5317,6 +6115,11 @@ var PipelineExecutor = class {
|
|
|
5317
6115
|
async run(tree, rootInput, fullFrameJpegProvider, imageWidth, imageHeight, deviceId, runOpts, nativeCropProvider, cropZoneBbox, rootInputViewProvider) {
|
|
5318
6116
|
const startMs = Date.now();
|
|
5319
6117
|
const verbosity = runOpts?.traceVerbosity ?? "off";
|
|
6118
|
+
const occasion = runOpts?.occasion ?? "live";
|
|
6119
|
+
const logTags = occasion === "replay" ? {
|
|
6120
|
+
deviceId,
|
|
6121
|
+
occasion
|
|
6122
|
+
} : { deviceId };
|
|
5320
6123
|
const traceBuilder = new ExecutionTraceBuilder(verbosity, deviceId, imageWidth, imageHeight, this.opts.engineRuntime);
|
|
5321
6124
|
const firstLevel = [];
|
|
5322
6125
|
const details = [];
|
|
@@ -5332,9 +6135,9 @@ var PipelineExecutor = class {
|
|
|
5332
6135
|
idGen,
|
|
5333
6136
|
details,
|
|
5334
6137
|
onClassifierClassDropped: (drop) => {
|
|
5335
|
-
const sample = this.classFilterDropSampler.record(deviceId, `${drop.stepId}:${drop.droppedClass}
|
|
6138
|
+
const sample = this.classFilterDropSampler.record(deviceId, `${drop.stepId}:${drop.droppedClass}`, occasion);
|
|
5336
6139
|
if (sample.logExample) this.opts.logger?.info("classifier winner excluded by class filter — label dropped", {
|
|
5337
|
-
tags:
|
|
6140
|
+
tags: logTags,
|
|
5338
6141
|
meta: {
|
|
5339
6142
|
step: drop.stepId,
|
|
5340
6143
|
model: drop.modelId,
|
|
@@ -5345,7 +6148,7 @@ var PipelineExecutor = class {
|
|
|
5345
6148
|
}
|
|
5346
6149
|
});
|
|
5347
6150
|
if (sample.summaryCount !== null) this.opts.logger?.info("classifier class filter: still dropping", {
|
|
5348
|
-
tags:
|
|
6151
|
+
tags: logTags,
|
|
5349
6152
|
meta: {
|
|
5350
6153
|
step: drop.stepId,
|
|
5351
6154
|
droppedClass: drop.droppedClass,
|
|
@@ -5378,7 +6181,7 @@ var PipelineExecutor = class {
|
|
|
5378
6181
|
await this.executeChildren(rootStep.children, mutable, rootView?.jpegProvider ?? fullFrameJpegProvider, imageWidth, imageHeight, traceBuilder, stepTimings, ctx, poolAgg, runOpts?.plane, deviceId, nativeCropProvider);
|
|
5379
6182
|
} catch (err) {
|
|
5380
6183
|
this.opts.logger?.warn("Pipeline child execution failed — keeping parent detection", {
|
|
5381
|
-
tags:
|
|
6184
|
+
tags: logTags,
|
|
5382
6185
|
meta: {
|
|
5383
6186
|
rootStepId: rootStep.stepId,
|
|
5384
6187
|
parentClass: mutable.macroClass,
|
|
@@ -5432,9 +6235,9 @@ var PipelineExecutor = class {
|
|
|
5432
6235
|
const fullFrameBar = rootStep.settings?.["fullFrameGuardMaxConfidence"];
|
|
5433
6236
|
const fullFrameHardArea = rootStep.settings?.["fullFrameGuardHardAreaRatio"];
|
|
5434
6237
|
if (isFullFramePhantomDetection(mutable.bbox, imageWidth, imageHeight, mutable.score, typeof fullFrameBar === "number" ? fullFrameBar : void 0, typeof fullFrameHardArea === "number" ? fullFrameHardArea : void 0)) {
|
|
5435
|
-
const sample = this.dropSampler.record(deviceId, String(mutable.macroClass));
|
|
6238
|
+
const sample = this.dropSampler.record(deviceId, String(mutable.macroClass), occasion);
|
|
5436
6239
|
if (sample.summaryCount !== null) this.opts.logger?.info("full-frame phantom guard: still dropping", {
|
|
5437
|
-
tags:
|
|
6240
|
+
tags: logTags,
|
|
5438
6241
|
meta: {
|
|
5439
6242
|
macroClass: mutable.macroClass,
|
|
5440
6243
|
suppressed: sample.summaryCount,
|
|
@@ -5443,7 +6246,7 @@ var PipelineExecutor = class {
|
|
|
5443
6246
|
}
|
|
5444
6247
|
});
|
|
5445
6248
|
if (sample.logExample) this.opts.logger?.info("full-frame phantom guard: detection dropped", {
|
|
5446
|
-
tags:
|
|
6249
|
+
tags: logTags,
|
|
5447
6250
|
meta: {
|
|
5448
6251
|
macroClass: mutable.macroClass,
|
|
5449
6252
|
score: mutable.score,
|
|
@@ -5469,7 +6272,7 @@ var PipelineExecutor = class {
|
|
|
5469
6272
|
continue;
|
|
5470
6273
|
}
|
|
5471
6274
|
if (isFullFrameSurvivor(mutable.bbox, imageWidth, imageHeight, mutable.score, typeof fullFrameBar === "number" ? fullFrameBar : void 0)) this.opts.logger?.info("full-frame box SURVIVED the guard on score", {
|
|
5472
|
-
tags:
|
|
6275
|
+
tags: logTags,
|
|
5473
6276
|
meta: {
|
|
5474
6277
|
macroClass: mutable.macroClass,
|
|
5475
6278
|
score: mutable.score,
|
|
@@ -5483,7 +6286,7 @@ var PipelineExecutor = class {
|
|
|
5483
6286
|
await this.executeChildren(rootStep.children, mutable, rootView?.jpegProvider ?? fullFrameJpegProvider, imageWidth, imageHeight, traceBuilder, stepTimings, ctx, poolAgg, runOpts?.plane, deviceId, nativeCropProvider);
|
|
5484
6287
|
} catch (err) {
|
|
5485
6288
|
this.opts.logger?.warn("Pipeline child execution failed — keeping parent detection", {
|
|
5486
|
-
tags:
|
|
6289
|
+
tags: logTags,
|
|
5487
6290
|
meta: {
|
|
5488
6291
|
rootStepId: rootStep.stepId,
|
|
5489
6292
|
parentClass: mutable.macroClass,
|
|
@@ -6500,6 +7303,34 @@ function parseWavToAudioChunk(filePath) {
|
|
|
6500
7303
|
function enginesEqual(a, b) {
|
|
6501
7304
|
return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
|
|
6502
7305
|
}
|
|
7306
|
+
/** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
|
|
7307
|
+
var ModelLoadRefusedError = class extends Error {
|
|
7308
|
+
model;
|
|
7309
|
+
state;
|
|
7310
|
+
constructor(message, model, state) {
|
|
7311
|
+
super(message);
|
|
7312
|
+
this.name = "ModelLoadRefusedError";
|
|
7313
|
+
this.model = model;
|
|
7314
|
+
this.state = state;
|
|
7315
|
+
}
|
|
7316
|
+
};
|
|
7317
|
+
/** `step/model` — how the governor and the log lines name a model. */
|
|
7318
|
+
function stepModelKey(step) {
|
|
7319
|
+
return `${step.addonId}/${step.modelId}`;
|
|
7320
|
+
}
|
|
7321
|
+
/** The reporting state a refusal verdict stands for. */
|
|
7322
|
+
function stateOf(verdict) {
|
|
7323
|
+
return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
|
|
7324
|
+
}
|
|
7325
|
+
/**
|
|
7326
|
+
* The reason a condemned pool is charged against its device's restart budget.
|
|
7327
|
+
* A pool recycled because it HUNG (D653) names the hang — `inference device
|
|
7328
|
+
* FAILED … reason: pool worker is not ready` could not tell a GPU compile that
|
|
7329
|
+
* never returned from a crash.
|
|
7330
|
+
*/
|
|
7331
|
+
function deathReasonOf(factory) {
|
|
7332
|
+
return factory.getDeathCause()?.message ?? "pool worker is not ready";
|
|
7333
|
+
}
|
|
6503
7334
|
/** Build a `RuntimeEnv` from the running process + probed hardware. */
|
|
6504
7335
|
function runtimeEnvFromProcess(hardware) {
|
|
6505
7336
|
return {
|
|
@@ -6541,25 +7372,6 @@ var PROBE_READY_TIMEOUT_MS = 12e4;
|
|
|
6541
7372
|
* race bound keeps a wedged transport from wedging `setApi`.
|
|
6542
7373
|
*/
|
|
6543
7374
|
var PROBE_DIRECT_RPC_TIMEOUT_MS = 6e4;
|
|
6544
|
-
/**
|
|
6545
|
-
* Build the onnx-cpu floor pick using `pickBestRuntime` with a null hardware
|
|
6546
|
-
* env. Used wherever the old `detectBestEngine()` sync probe fell back — the
|
|
6547
|
-
* result is identical (onnx / cpu) but is now derived through the shared rules
|
|
6548
|
-
* instead of duplicated inline.
|
|
6549
|
-
*/
|
|
6550
|
-
/**
|
|
6551
|
-
* Is `backend` even POSSIBLE on this node's OS/arch (ignoring gpu detail)?
|
|
6552
|
-
* coreml ⇒ darwin only; openvino ⇒ x64 non-darwin only; onnx ⇒ anywhere. Used to
|
|
6553
|
-
* reject a persisted/global engine choice that the node's PLATFORM fundamentally
|
|
6554
|
-
* cannot run (e.g. the cluster's OpenVINO default landing on a Mac) — distinct
|
|
6555
|
-
* from the gpu-dependent support (linux without a probed Intel iGPU still keeps
|
|
6556
|
-
* openvino as a valid platform choice; the device falls back to cpu).
|
|
6557
|
-
*/
|
|
6558
|
-
function backendPossibleOnPlatform(backend) {
|
|
6559
|
-
if (backend === "coreml") return process.platform === "darwin";
|
|
6560
|
-
if (backend === "openvino") return process.arch === "x64" && process.platform !== "darwin";
|
|
6561
|
-
return true;
|
|
6562
|
-
}
|
|
6563
7375
|
var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
6564
7376
|
modelsDir;
|
|
6565
7377
|
eventBus;
|
|
@@ -6699,22 +7511,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6699
7511
|
*/
|
|
6700
7512
|
needsAutoPick = false;
|
|
6701
7513
|
/**
|
|
6702
|
-
* Warm cache for benchmark engine-override runs.
|
|
6703
|
-
*
|
|
6704
|
-
* Each override rebuild costs a full Python pool spin-up (~300-500ms)
|
|
6705
|
-
* plus the per-model load. Benchmark tabs iterate up to 800× against
|
|
6706
|
-
* the same override (e.g. yolov9s / coreml / all) — paying that spin-up
|
|
6707
|
-
* each iteration dwarfs actual inference time.
|
|
6708
|
-
*
|
|
6709
|
-
* When the same override engine is requested within `OVERRIDE_CACHE_TTL_MS`,
|
|
6710
|
-
* we reuse the prior transient factory. Hitting a different override
|
|
6711
|
-
* or exceeding the TTL disposes the cache and spins a new factory.
|
|
6712
|
-
* The prior "main" factory (`priorFactory`) is still restored so
|
|
6713
|
-
* runtime dispatch (camera frames) keeps its own resident engine.
|
|
6714
|
-
*/
|
|
6715
|
-
overrideCache = null;
|
|
6716
|
-
overrideCacheTimer = null;
|
|
6717
|
-
/**
|
|
6718
7514
|
* Idle-TTL for a per-device inference pool (multi-device C3). A device pool
|
|
6719
7515
|
* with no dispatch for this long is disposed and recreated on next use — the
|
|
6720
7516
|
* memory guardrail that keeps `factoriesByDevice` from accumulating warm pools
|
|
@@ -6747,7 +7543,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6747
7543
|
error: err instanceof Error ? err.message : String(err)
|
|
6748
7544
|
} })
|
|
6749
7545
|
});
|
|
6750
|
-
static OVERRIDE_CACHE_TTL_MS = 6e4;
|
|
6751
7546
|
/**
|
|
6752
7547
|
* RSS bound + periodic memory telemetry for the Python inference pools
|
|
6753
7548
|
* (2026-08-18 OOM audit: the pools, not the recorder, were the host's OOM
|
|
@@ -6884,15 +7679,14 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6884
7679
|
/**
|
|
6885
7680
|
* Lazy-install the pip requirements file matching `engine.backend` into
|
|
6886
7681
|
* the embedded Python before the inference pool spawns. No-op for
|
|
6887
|
-
*
|
|
6888
|
-
*
|
|
7682
|
+
* backends with no requirements file on disk, or before the addon
|
|
7683
|
+
* context has been wired (warm-pool race window).
|
|
6889
7684
|
*
|
|
6890
7685
|
* Idempotent — `installPythonRequirements` short-circuits on a hash
|
|
6891
7686
|
* marker, so back-to-back EngineFactory rebuilds (benchmark overrides,
|
|
6892
7687
|
* tuning respawns) skip the pip subprocess after the first install.
|
|
6893
7688
|
*/
|
|
6894
7689
|
async ensureBackendDeps(engine) {
|
|
6895
|
-
if (engine.runtime !== "python") return;
|
|
6896
7690
|
if (!this.addonCtx) return;
|
|
6897
7691
|
const pythonAddonDir = this.executorOptions.pythonAddonDir;
|
|
6898
7692
|
if (!pythonAddonDir) return;
|
|
@@ -7457,8 +8251,22 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7457
8251
|
* result with a warning logged, same posture as `recomputeConfigIssues`.
|
|
7458
8252
|
*/
|
|
7459
8253
|
async validatePipeline(input) {
|
|
7460
|
-
|
|
8254
|
+
let format = this.currentEngine.format;
|
|
7461
8255
|
try {
|
|
8256
|
+
if (input.deviceKey !== void 0) try {
|
|
8257
|
+
format = resolveDeviceEngine(input.deviceKey).format;
|
|
8258
|
+
} catch (err) {
|
|
8259
|
+
return {
|
|
8260
|
+
ok: false,
|
|
8261
|
+
issues: [{
|
|
8262
|
+
addonId: "*",
|
|
8263
|
+
kind: "unknown-device",
|
|
8264
|
+
detail: errMsg(err)
|
|
8265
|
+
}],
|
|
8266
|
+
substitutions: [],
|
|
8267
|
+
format
|
|
8268
|
+
};
|
|
8269
|
+
}
|
|
7462
8270
|
const steps = input.steps;
|
|
7463
8271
|
const enabledSteps = this.pruneToEnabledSteps(steps);
|
|
7464
8272
|
const substitutions = collectSubstitutions(enabledSteps, format);
|
|
@@ -7783,7 +8591,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7783
8591
|
const nodeId = this.addonCtx?.kernel?.localNodeId ?? "hub";
|
|
7784
8592
|
const sessionId = input.sessionId ?? `run-${Date.now()}-${Math.random().toString(36).slice(2, 10)}`;
|
|
7785
8593
|
const isRuntime = Boolean(input.frame || input.frameRef);
|
|
7786
|
-
const progressToBus = !isRuntime || input.sessionId !== void 0;
|
|
8594
|
+
const progressToBus = !isRuntime && input.replay !== true || input.sessionId !== void 0;
|
|
7787
8595
|
const emit = (message, extra) => {
|
|
7788
8596
|
onProgress?.(message);
|
|
7789
8597
|
if (!progressToBus) return;
|
|
@@ -7925,7 +8733,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7925
8733
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
7926
8734
|
steps: input.steps,
|
|
7927
8735
|
deviceKey: input.deviceKey,
|
|
7928
|
-
engineOverride: input.engine,
|
|
7929
8736
|
deviceId: input.deviceId,
|
|
7930
8737
|
plane: input.plane
|
|
7931
8738
|
});
|
|
@@ -7945,322 +8752,153 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7945
8752
|
}
|
|
7946
8753
|
const enabledSteps = flattenSteps(benchmarkSteps);
|
|
7947
8754
|
emit(`Pipeline: ${enabledSteps.length} step(s) — ${enabledSteps.map((s) => s.addonId).join(" → ")}`);
|
|
7948
|
-
|
|
7949
|
-
const
|
|
7950
|
-
|
|
7951
|
-
|
|
7952
|
-
|
|
7953
|
-
|
|
7954
|
-
|
|
7955
|
-
|
|
7956
|
-
|
|
7957
|
-
|
|
7958
|
-
|
|
7959
|
-
|
|
7960
|
-
const cached = this.overrideCache;
|
|
7961
|
-
if (cached.initPromise) await cached.initPromise;
|
|
7962
|
-
this.log.debug("Benchmark: reusing warm override factory", { meta: { engine: `${runEngine.runtime}/${runEngine.backend}/${runEngine.device ?? "default"}` } });
|
|
7963
|
-
this.currentEngine = runEngine;
|
|
7964
|
-
this.engineFactory = cached.factory;
|
|
7965
|
-
this.executor = cached.executor;
|
|
7966
|
-
} else {
|
|
7967
|
-
if (this.overrideCache) {
|
|
7968
|
-
try {
|
|
7969
|
-
await this.overrideCache.factory.dispose();
|
|
7970
|
-
} catch {}
|
|
7971
|
-
this.overrideCache = null;
|
|
7972
|
-
}
|
|
7973
|
-
this.log.info("Benchmark: engine override", { meta: {
|
|
7974
|
-
from: `${priorEngine.runtime}/${priorEngine.backend}`,
|
|
7975
|
-
to: `${runEngine.runtime}/${runEngine.backend}`
|
|
7976
|
-
} });
|
|
7977
|
-
await this.ensureBackendDeps(runEngine);
|
|
7978
|
-
const newFactory = new EngineFactory({
|
|
7979
|
-
engine: runEngine,
|
|
7980
|
-
modelsDir: this.modelsDir,
|
|
7981
|
-
logger: this.log.child("engine-override"),
|
|
7982
|
-
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
7983
|
-
provisioning: this.executorOptions.provisioning,
|
|
7984
|
-
resolveCustomModel: this.customModelResolver
|
|
7985
|
-
});
|
|
7986
|
-
const initPromise = newFactory.initialize([]);
|
|
7987
|
-
this.overrideCache = {
|
|
7988
|
-
engine: runEngine,
|
|
7989
|
-
factory: newFactory,
|
|
7990
|
-
executor: null,
|
|
7991
|
-
lastUsedMs: Date.now(),
|
|
7992
|
-
initPromise
|
|
7993
|
-
};
|
|
7994
|
-
this.scheduleOverrideCacheEviction();
|
|
7995
|
-
try {
|
|
7996
|
-
await initPromise;
|
|
7997
|
-
this.overrideCache.initPromise = null;
|
|
7998
|
-
} catch (err) {
|
|
7999
|
-
this.overrideCache = null;
|
|
8000
|
-
if (this.overrideCacheTimer) {
|
|
8001
|
-
clearTimeout(this.overrideCacheTimer);
|
|
8002
|
-
this.overrideCacheTimer = null;
|
|
8003
|
-
}
|
|
8004
|
-
throw err;
|
|
8005
|
-
}
|
|
8006
|
-
this.currentEngine = runEngine;
|
|
8007
|
-
this.engineFactory = newFactory;
|
|
8008
|
-
this.executor = null;
|
|
8009
|
-
}
|
|
8010
|
-
try {
|
|
8011
|
-
const deviceLabel = runEngine.device ?? "default";
|
|
8012
|
-
emit(`Engine: ${runEngine.runtime}/${runEngine.backend}/${deviceLabel}${input.engine ? " (override)" : ""}`, {
|
|
8013
|
-
step: "engine",
|
|
8014
|
-
engine: {
|
|
8015
|
-
runtime: runEngine.runtime,
|
|
8016
|
-
backend: runEngine.backend,
|
|
8017
|
-
device: deviceLabel
|
|
8018
|
-
}
|
|
8019
|
-
});
|
|
8020
|
-
const runtimeStr = `${runEngine.runtime}+${runEngine.backend}`;
|
|
8021
|
-
const executor = this.executor ?? new PipelineExecutor({
|
|
8022
|
-
engineRuntime: runtimeStr,
|
|
8023
|
-
logger: this.log
|
|
8755
|
+
const rootModelId = resolveRootModelId(benchmarkSteps);
|
|
8756
|
+
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
8757
|
+
...r,
|
|
8758
|
+
rootModelId
|
|
8759
|
+
};
|
|
8760
|
+
if (input.replay === true) {
|
|
8761
|
+
const verdict = checkReplayPin({
|
|
8762
|
+
requested: input.steps,
|
|
8763
|
+
executing: input.plane === "frame" ? collectFramePlaneSteps(benchmarkSteps, isDetailPlaneStep) : enabledSteps,
|
|
8764
|
+
requestedDeviceKey: input.deviceKey,
|
|
8765
|
+
dispatchDeviceKey,
|
|
8766
|
+
format: dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format : this.currentEngine.format
|
|
8024
8767
|
});
|
|
8025
|
-
|
|
8026
|
-
|
|
8027
|
-
|
|
8028
|
-
|
|
8029
|
-
|
|
8030
|
-
|
|
8031
|
-
|
|
8032
|
-
|
|
8033
|
-
} });
|
|
8034
|
-
for (const s of needed) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8035
|
-
step: s.addonId,
|
|
8036
|
-
addonId: s.addonId,
|
|
8037
|
-
modelId: s.modelId
|
|
8768
|
+
if (verdict.kind === "refused") {
|
|
8769
|
+
this.log.warn("replay refused — the node cannot run exactly what was pinned", {
|
|
8770
|
+
...input.deviceId !== void 0 ? { tags: { deviceId: input.deviceId } } : {},
|
|
8771
|
+
meta: {
|
|
8772
|
+
reason: verdict.reason,
|
|
8773
|
+
detail: verdict.detail,
|
|
8774
|
+
deviceKey: input.deviceKey
|
|
8775
|
+
}
|
|
8038
8776
|
});
|
|
8039
|
-
|
|
8040
|
-
emit(`All models loaded`);
|
|
8041
|
-
}
|
|
8042
|
-
emit("Running inference...");
|
|
8043
|
-
const tree = buildExecutableTree(benchmarkSteps, (stepId) => dispatchFactory.getEngine(stepId), this.customModelResolver);
|
|
8044
|
-
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8045
|
-
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8046
|
-
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
8047
|
-
const effectiveTree = Object.keys(deviceOverrides).length > 0 ? applyDeviceOverridesToTree(tree, "object-detection", deviceOverrides) : tree;
|
|
8048
|
-
const nativeCropProvider = this.buildNativeCropProviderFromRef(input.nativeCropRef) ?? this.buildNativeFaceCropProvider(input.frameHandle);
|
|
8049
|
-
let cropZoneBbox;
|
|
8050
|
-
if (isRuntime && effectiveDeviceId > 0 && effectiveTree.roots.some((r) => r.definition.extractMode === "crop-zone")) {
|
|
8051
|
-
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8052
|
-
const proxy = this.deviceProxies.get(effectiveDeviceId);
|
|
8053
|
-
if (proxy) cropZoneBbox = resolvePackageCropBbox(proxy.state.zones.value?.zones ?? [], proxy.state.zoneRules.value?.package ?? [], imageWidth, imageHeight) ?? void 0;
|
|
8054
|
-
}
|
|
8055
|
-
let frameViewResolver;
|
|
8056
|
-
let rootInputViewProvider;
|
|
8057
|
-
if (runtimeFrameRef) {
|
|
8058
|
-
frameViewResolver = createRootFrameViewResolver(localFrameRegistry, runtimeFrameRef);
|
|
8059
|
-
rootInputViewProvider = async (root, crop) => {
|
|
8060
|
-
const view = await frameViewResolver(root, crop ? {
|
|
8061
|
-
left: crop[0],
|
|
8062
|
-
top: crop[1],
|
|
8063
|
-
width: crop[2] - crop[0],
|
|
8064
|
-
height: crop[3] - crop[1]
|
|
8065
|
-
} : void 0);
|
|
8066
|
-
const data = Buffer.from(view.data.buffer, view.data.byteOffset, view.data.byteLength);
|
|
8067
|
-
return {
|
|
8068
|
-
input: {
|
|
8069
|
-
kind: "jpeg",
|
|
8070
|
-
data
|
|
8071
|
-
},
|
|
8072
|
-
jpegProvider: async () => data,
|
|
8073
|
-
width: view.width,
|
|
8074
|
-
height: view.height,
|
|
8075
|
-
geometry: view.geometry
|
|
8076
|
-
};
|
|
8077
|
-
};
|
|
8078
|
-
}
|
|
8079
|
-
let executionSucceeded = false;
|
|
8080
|
-
let execution;
|
|
8081
|
-
try {
|
|
8082
|
-
execution = await executor.run(effectiveTree, rootInput, jpegProvider, imageWidth, imageHeight, effectiveDeviceId, {
|
|
8083
|
-
traceVerbosity: isRuntime ? this.eventBus ? "summary" : "off" : "full",
|
|
8084
|
-
plane: input.plane
|
|
8085
|
-
}, nativeCropProvider, cropZoneBbox, rootInputViewProvider);
|
|
8086
|
-
executionSucceeded = true;
|
|
8087
|
-
} finally {
|
|
8088
|
-
frameViewResolver?.release(executionSucceeded ? "success" : "error");
|
|
8089
|
-
}
|
|
8090
|
-
const { result, trace } = execution;
|
|
8091
|
-
if (isRuntime) {
|
|
8092
|
-
if (trace && this.eventBus) this.eventBus.emit(createEvent(EventCategory.PipelineTrace, {
|
|
8093
|
-
type: "device",
|
|
8094
|
-
id: trace.deviceId,
|
|
8095
|
-
nodeId: "hub"
|
|
8096
|
-
}, trace));
|
|
8097
|
-
if (effectiveDeviceId > 0) {
|
|
8098
|
-
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8099
|
-
return this.gateDetectionsByZoneRules(effectiveDeviceId, result);
|
|
8100
|
-
}
|
|
8101
|
-
return result;
|
|
8102
|
-
}
|
|
8103
|
-
for (const t of result.debug?.stepTimings ?? []) emit(`${t.source}${t.modelId ? ` (${t.modelId})` : ""} → ${t.ms}ms`, {
|
|
8104
|
-
step: t.source,
|
|
8105
|
-
addonId: t.source,
|
|
8106
|
-
modelId: t.modelId ?? void 0,
|
|
8107
|
-
ms: t.ms
|
|
8108
|
-
});
|
|
8109
|
-
emit(`Done — ${result.detections.length} detection(s) in ${result.debug?.totalInferenceMs ?? 0}ms`, { ms: result.debug?.totalInferenceMs });
|
|
8110
|
-
const wallMs = performance.now() - wallT0;
|
|
8111
|
-
const inferMs = result.debug?.totalInferenceMs ?? 0;
|
|
8112
|
-
const overheadMs = Math.max(0, wallMs - decodeMs - setupMs - inferMs);
|
|
8113
|
-
return {
|
|
8114
|
-
...result,
|
|
8115
|
-
debug: {
|
|
8116
|
-
...result.debug,
|
|
8117
|
-
decodeMs: Math.round(decodeMs * 100) / 100,
|
|
8118
|
-
setupMs: Math.round(setupMs * 100) / 100,
|
|
8119
|
-
wallMs: Math.round(wallMs * 100) / 100,
|
|
8120
|
-
overheadMs: Math.round(overheadMs * 100) / 100
|
|
8121
|
-
}
|
|
8122
|
-
};
|
|
8123
|
-
} finally {
|
|
8124
|
-
if (differs && this.overrideCache) {
|
|
8125
|
-
this.overrideCache.executor = this.executor;
|
|
8126
|
-
this.overrideCache.lastUsedMs = Date.now();
|
|
8127
|
-
this.scheduleOverrideCacheEviction();
|
|
8777
|
+
throw new Error(`${verdict.reason}: ${verdict.detail}`);
|
|
8128
8778
|
}
|
|
8129
|
-
if (differs) {
|
|
8130
|
-
this.currentEngine = priorEngine;
|
|
8131
|
-
this.engineFactory = priorFactory;
|
|
8132
|
-
this.executor = priorExecutor;
|
|
8133
|
-
}
|
|
8134
|
-
}
|
|
8135
|
-
}
|
|
8136
|
-
/**
|
|
8137
|
-
* Apply a benchmark engine override (mirroring the body of `runPipeline`
|
|
8138
|
-
* lines 863-953). Swaps `currentEngine` / `engineFactory` / `executor`
|
|
8139
|
-
* to the override's warm factory and returns a restore function the
|
|
8140
|
-
* caller MUST invoke in `finally`. When `override` is undefined or
|
|
8141
|
-
* matches the current engine, this is a no-op and the restore function
|
|
8142
|
-
* does nothing — same shape so callers don't need a conditional path.
|
|
8143
|
-
*
|
|
8144
|
-
* Used by both `runPipeline` (single-frame benchmark) and
|
|
8145
|
-
* `runPipelineBatch` (batched fast path). Without this, the batch
|
|
8146
|
-
* path silently ignored `input.engine` and ran the user's override
|
|
8147
|
-
* against the addon's saved engine — making the override matrix in
|
|
8148
|
-
* the bench UI a no-op for batchSize > 1.
|
|
8149
|
-
*/
|
|
8150
|
-
async applyEngineOverride(override) {
|
|
8151
|
-
const priorEngine = this.currentEngine;
|
|
8152
|
-
const priorFactory = this.engineFactory;
|
|
8153
|
-
const priorExecutor = this.executor;
|
|
8154
|
-
if (!override) return () => {};
|
|
8155
|
-
if (!backendPossibleOnPlatform(override.backend)) {
|
|
8156
|
-
this.log.warn("Engine override impossible on this platform — running on own engine", { meta: {
|
|
8157
|
-
override: `${override.runtime}/${override.backend}/${override.device ?? "default"}`,
|
|
8158
|
-
platform: process.platform,
|
|
8159
|
-
arch: process.arch,
|
|
8160
|
-
engine: `${priorEngine.runtime}/${priorEngine.backend}`
|
|
8161
|
-
} });
|
|
8162
|
-
return () => {};
|
|
8163
8779
|
}
|
|
8164
|
-
const
|
|
8165
|
-
|
|
8166
|
-
|
|
8167
|
-
|
|
8168
|
-
|
|
8169
|
-
|
|
8170
|
-
|
|
8171
|
-
|
|
8172
|
-
|
|
8173
|
-
|
|
8174
|
-
this.log.debug("Benchmark: reusing warm override factory", { meta: { engine: `${runEngine.runtime}/${runEngine.backend}/${runEngine.device ?? "default"}` } });
|
|
8175
|
-
this.currentEngine = runEngine;
|
|
8176
|
-
this.engineFactory = cached.factory;
|
|
8177
|
-
this.executor = cached.executor;
|
|
8178
|
-
} else {
|
|
8179
|
-
if (this.overrideCache) {
|
|
8180
|
-
try {
|
|
8181
|
-
await this.overrideCache.factory.dispose();
|
|
8182
|
-
} catch {}
|
|
8183
|
-
this.overrideCache = null;
|
|
8780
|
+
const appliesCameraGates = isRuntime || input.replay === true;
|
|
8781
|
+
await this.ensureEngineFactory();
|
|
8782
|
+
const runEngine = this.currentEngine;
|
|
8783
|
+
const deviceLabel = runEngine.device ?? "default";
|
|
8784
|
+
emit(`Engine: ${runEngine.runtime}/${runEngine.backend}/${deviceLabel}`, {
|
|
8785
|
+
step: "engine",
|
|
8786
|
+
engine: {
|
|
8787
|
+
runtime: runEngine.runtime,
|
|
8788
|
+
backend: runEngine.backend,
|
|
8789
|
+
device: deviceLabel
|
|
8184
8790
|
}
|
|
8185
|
-
|
|
8186
|
-
|
|
8187
|
-
|
|
8791
|
+
});
|
|
8792
|
+
const runtimeStr = `${runEngine.runtime}+${runEngine.backend}`;
|
|
8793
|
+
const executor = this.executor ?? new PipelineExecutor({
|
|
8794
|
+
engineRuntime: runtimeStr,
|
|
8795
|
+
logger: this.log
|
|
8796
|
+
});
|
|
8797
|
+
this.executor = executor;
|
|
8798
|
+
const dispatchEngine = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey) : runEngine;
|
|
8799
|
+
const dispatchFactory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8800
|
+
const needed = enabledSteps.filter((s) => dispatchFactory.needsPoolUpdate(s));
|
|
8801
|
+
if (needed.length > 0) {
|
|
8802
|
+
this.log.info("Benchmark: models to load", { meta: {
|
|
8803
|
+
count: needed.length,
|
|
8804
|
+
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8188
8805
|
} });
|
|
8189
|
-
|
|
8190
|
-
|
|
8191
|
-
|
|
8192
|
-
|
|
8193
|
-
logger: this.log.child("engine-override"),
|
|
8194
|
-
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8195
|
-
provisioning: this.executorOptions.provisioning,
|
|
8196
|
-
resolveCustomModel: this.customModelResolver
|
|
8806
|
+
for (const s of needed) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8807
|
+
step: s.addonId,
|
|
8808
|
+
addonId: s.addonId,
|
|
8809
|
+
modelId: s.modelId
|
|
8197
8810
|
});
|
|
8198
|
-
|
|
8199
|
-
|
|
8200
|
-
|
|
8201
|
-
|
|
8202
|
-
|
|
8203
|
-
|
|
8204
|
-
|
|
8811
|
+
await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
|
|
8812
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
8813
|
+
deviceKey: deviceKeyOf(dispatchEngine)
|
|
8814
|
+
});
|
|
8815
|
+
emit(`All models loaded`);
|
|
8816
|
+
}
|
|
8817
|
+
emit("Running inference...");
|
|
8818
|
+
const tree = buildExecutableTree(benchmarkSteps, (stepId) => dispatchFactory.getEngine(stepId), this.customModelResolver);
|
|
8819
|
+
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8820
|
+
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8821
|
+
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
8822
|
+
const effectiveTree = Object.keys(deviceOverrides).length > 0 ? applyDeviceOverridesToTree(tree, "object-detection", deviceOverrides) : tree;
|
|
8823
|
+
const nativeCropProvider = this.buildNativeCropProviderFromRef(input.nativeCropRef) ?? this.buildNativeFaceCropProvider(input.frameHandle);
|
|
8824
|
+
let cropZoneBbox;
|
|
8825
|
+
if (appliesCameraGates && effectiveDeviceId > 0 && effectiveTree.roots.some((r) => r.definition.extractMode === "crop-zone")) {
|
|
8826
|
+
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8827
|
+
const proxy = this.deviceProxies.get(effectiveDeviceId);
|
|
8828
|
+
if (proxy) cropZoneBbox = resolvePackageCropBbox(proxy.state.zones.value?.zones ?? [], proxy.state.zoneRules.value?.package ?? [], imageWidth, imageHeight) ?? void 0;
|
|
8829
|
+
}
|
|
8830
|
+
let frameViewResolver;
|
|
8831
|
+
let rootInputViewProvider;
|
|
8832
|
+
if (runtimeFrameRef) {
|
|
8833
|
+
frameViewResolver = createRootFrameViewResolver(localFrameRegistry, runtimeFrameRef);
|
|
8834
|
+
rootInputViewProvider = async (root, crop) => {
|
|
8835
|
+
const view = await frameViewResolver(root, crop ? {
|
|
8836
|
+
left: crop[0],
|
|
8837
|
+
top: crop[1],
|
|
8838
|
+
width: crop[2] - crop[0],
|
|
8839
|
+
height: crop[3] - crop[1]
|
|
8840
|
+
} : void 0);
|
|
8841
|
+
const data = Buffer.from(view.data.buffer, view.data.byteOffset, view.data.byteLength);
|
|
8842
|
+
return {
|
|
8843
|
+
input: {
|
|
8844
|
+
kind: "jpeg",
|
|
8845
|
+
data
|
|
8846
|
+
},
|
|
8847
|
+
jpegProvider: async () => data,
|
|
8848
|
+
width: view.width,
|
|
8849
|
+
height: view.height,
|
|
8850
|
+
geometry: view.geometry
|
|
8851
|
+
};
|
|
8205
8852
|
};
|
|
8206
|
-
|
|
8207
|
-
|
|
8208
|
-
|
|
8209
|
-
|
|
8210
|
-
|
|
8211
|
-
this.
|
|
8212
|
-
|
|
8213
|
-
|
|
8214
|
-
|
|
8215
|
-
|
|
8216
|
-
|
|
8853
|
+
}
|
|
8854
|
+
let executionSucceeded = false;
|
|
8855
|
+
let execution;
|
|
8856
|
+
try {
|
|
8857
|
+
execution = await executor.run(effectiveTree, rootInput, jpegProvider, imageWidth, imageHeight, effectiveDeviceId, {
|
|
8858
|
+
traceVerbosity: isRuntime ? this.eventBus ? "summary" : "off" : "full",
|
|
8859
|
+
plane: input.plane,
|
|
8860
|
+
occasion: input.replay === true ? "replay" : "live"
|
|
8861
|
+
}, nativeCropProvider, cropZoneBbox, rootInputViewProvider);
|
|
8862
|
+
executionSucceeded = true;
|
|
8863
|
+
} finally {
|
|
8864
|
+
frameViewResolver?.release(executionSucceeded ? "success" : "error");
|
|
8865
|
+
}
|
|
8866
|
+
const { result, trace } = execution;
|
|
8867
|
+
if (isRuntime) {
|
|
8868
|
+
if (trace && this.eventBus) this.eventBus.emit(createEvent(EventCategory.PipelineTrace, {
|
|
8869
|
+
type: "device",
|
|
8870
|
+
id: trace.deviceId,
|
|
8871
|
+
nodeId: "hub"
|
|
8872
|
+
}, trace));
|
|
8873
|
+
if (effectiveDeviceId > 0) {
|
|
8874
|
+
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8875
|
+
return stampRoot(this.gateDetectionsByZoneRules(effectiveDeviceId, result));
|
|
8217
8876
|
}
|
|
8218
|
-
|
|
8219
|
-
this.engineFactory = newFactory;
|
|
8220
|
-
this.executor = null;
|
|
8877
|
+
return stampRoot(result);
|
|
8221
8878
|
}
|
|
8222
|
-
|
|
8223
|
-
|
|
8224
|
-
|
|
8225
|
-
|
|
8226
|
-
|
|
8879
|
+
for (const t of result.debug?.stepTimings ?? []) emit(`${t.source}${t.modelId ? ` (${t.modelId})` : ""} → ${t.ms}ms`, {
|
|
8880
|
+
step: t.source,
|
|
8881
|
+
addonId: t.source,
|
|
8882
|
+
modelId: t.modelId ?? void 0,
|
|
8883
|
+
ms: t.ms
|
|
8884
|
+
});
|
|
8885
|
+
emit(`Done — ${result.detections.length} detection(s) in ${result.debug?.totalInferenceMs ?? 0}ms`, { ms: result.debug?.totalInferenceMs });
|
|
8886
|
+
const wallMs = performance.now() - wallT0;
|
|
8887
|
+
const inferMs = result.debug?.totalInferenceMs ?? 0;
|
|
8888
|
+
const overheadMs = Math.max(0, wallMs - decodeMs - setupMs - inferMs);
|
|
8889
|
+
const gated = input.replay === true && effectiveDeviceId > 0 ? await this.ensureDeviceProxy(effectiveDeviceId).then(() => this.gateDetectionsByZoneRules(effectiveDeviceId, result)) : result;
|
|
8890
|
+
return {
|
|
8891
|
+
...stampRoot(gated),
|
|
8892
|
+
debug: {
|
|
8893
|
+
...gated.debug,
|
|
8894
|
+
decodeMs: Math.round(decodeMs * 100) / 100,
|
|
8895
|
+
setupMs: Math.round(setupMs * 100) / 100,
|
|
8896
|
+
wallMs: Math.round(wallMs * 100) / 100,
|
|
8897
|
+
overheadMs: Math.round(overheadMs * 100) / 100
|
|
8227
8898
|
}
|
|
8228
|
-
this.currentEngine = priorEngine;
|
|
8229
|
-
this.engineFactory = priorFactory;
|
|
8230
|
-
this.executor = priorExecutor;
|
|
8231
8899
|
};
|
|
8232
8900
|
}
|
|
8233
8901
|
/**
|
|
8234
|
-
* Schedule eviction of the override cache after `OVERRIDE_CACHE_TTL_MS`
|
|
8235
|
-
* of idleness. Rescheduling resets the timer so an active benchmark
|
|
8236
|
-
* keeps its warm factory alive until iterations stop arriving.
|
|
8237
|
-
*/
|
|
8238
|
-
scheduleOverrideCacheEviction() {
|
|
8239
|
-
if (this.overrideCacheTimer) clearTimeout(this.overrideCacheTimer);
|
|
8240
|
-
this.overrideCacheTimer = setTimeout(() => {
|
|
8241
|
-
this.evictOverrideCache("idle TTL");
|
|
8242
|
-
}, DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS);
|
|
8243
|
-
}
|
|
8244
|
-
async evictOverrideCache(reason) {
|
|
8245
|
-
if (!this.overrideCache) return;
|
|
8246
|
-
const entry = this.overrideCache;
|
|
8247
|
-
this.overrideCache = null;
|
|
8248
|
-
if (this.overrideCacheTimer) {
|
|
8249
|
-
clearTimeout(this.overrideCacheTimer);
|
|
8250
|
-
this.overrideCacheTimer = null;
|
|
8251
|
-
}
|
|
8252
|
-
try {
|
|
8253
|
-
await entry.factory.dispose();
|
|
8254
|
-
this.log.info("Override cache evicted", { meta: {
|
|
8255
|
-
reason,
|
|
8256
|
-
engine: `${entry.engine.runtime}/${entry.engine.backend}/${entry.engine.device ?? "default"}`,
|
|
8257
|
-
ageMs: Date.now() - entry.lastUsedMs
|
|
8258
|
-
} });
|
|
8259
|
-
} catch (err) {
|
|
8260
|
-
this.log.warn("Override cache dispose failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
8261
|
-
}
|
|
8262
|
-
}
|
|
8263
|
-
/**
|
|
8264
8902
|
* Convert PipelineStepInput[] → PipelineDefaultStep[] by enriching from
|
|
8265
8903
|
* StepDefinitions. This is the LIVE per-camera dispatch path — it runs
|
|
8266
8904
|
* once per decoded frame (up to detectionFps × N cameras/sec) — so it
|
|
@@ -8276,13 +8914,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8276
8914
|
* `resolveStepModels` (see its doc comment); a second writer here would
|
|
8277
8915
|
* corrupt it.
|
|
8278
8916
|
*
|
|
8279
|
-
* `format` defaults to `this.currentEngine.format` (the
|
|
8280
|
-
*
|
|
8281
|
-
* per-run engine override (`input.engine`) MUST pass the override's format
|
|
8282
|
-
* explicitly — `this.currentEngine` only gets swapped to the override
|
|
8283
|
-
* LATER in `runPipeline`/`runPipelineBatch`, so relying on the default here
|
|
8284
|
-
* would resolve models against the node's persisted format instead of the
|
|
8285
|
-
* format actually being benchmarked.
|
|
8917
|
+
* `format` defaults to `this.currentEngine.format` (the node-default
|
|
8918
|
+
* pool); a `deviceKey` dispatch passes that device's format explicitly.
|
|
8286
8919
|
*/
|
|
8287
8920
|
inputStepsToPipelineSteps(steps, format = this.currentEngine.format, engine = {
|
|
8288
8921
|
backend: this.currentEngine.backend,
|
|
@@ -8364,8 +8997,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8364
8997
|
* devices — so the whole-tree gate stays.
|
|
8365
8998
|
*
|
|
8366
8999
|
* Fallback is strictly NODE-LOCAL and ordered: the node's own selected
|
|
8367
|
-
* engine
|
|
8368
|
-
* fallback — frames never migrate to another node from here (that is the
|
|
9000
|
+
* engine is the one and only fallback — frames never migrate to another node from here (that is the
|
|
8369
9001
|
* orchestrator's tier). The fallback is logged ONCE per distinct
|
|
8370
9002
|
* (deviceKey, steps, format) signature, with `tags: { deviceId }` when the
|
|
8371
9003
|
* dispatch is camera-scoped.
|
|
@@ -8376,12 +9008,9 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8376
9008
|
* configuration must stay loud, not vanish into a silent fallback chain.
|
|
8377
9009
|
*/
|
|
8378
9010
|
resolveStepsForDispatch(args) {
|
|
8379
|
-
const { deviceKey,
|
|
8380
|
-
const nodeFormat =
|
|
8381
|
-
const nodeEngine =
|
|
8382
|
-
backend: engineOverride.backend,
|
|
8383
|
-
device: engineOverride.device ?? null
|
|
8384
|
-
} : {
|
|
9011
|
+
const { deviceKey, deviceId } = args;
|
|
9012
|
+
const nodeFormat = this.currentEngine.format;
|
|
9013
|
+
const nodeEngine = {
|
|
8385
9014
|
backend: this.currentEngine.backend,
|
|
8386
9015
|
device: this.currentEngine.device ?? null
|
|
8387
9016
|
};
|
|
@@ -8479,53 +9108,193 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8479
9108
|
throw new Error(message);
|
|
8480
9109
|
}
|
|
8481
9110
|
/**
|
|
8482
|
-
*
|
|
8483
|
-
*
|
|
8484
|
-
*
|
|
8485
|
-
*
|
|
8486
|
-
*
|
|
8487
|
-
*
|
|
8488
|
-
*
|
|
8489
|
-
*/
|
|
8490
|
-
modelLoadInFlight =
|
|
9111
|
+
* In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
|
|
9112
|
+
* needing the same models on the same pool share ONE load and its outcome;
|
|
9113
|
+
* a dispatch needing a different set is never handed another set's failure
|
|
9114
|
+
* — it queues behind the pool's current load ({@link modelLoadTail}) and
|
|
9115
|
+
* then decides for itself. Keyed by pool, not node-wide: one GPU compile
|
|
9116
|
+
* that never returned held EVERY load on the node — the NPU's and the CPU's
|
|
9117
|
+
* too — behind it.
|
|
9118
|
+
*/
|
|
9119
|
+
modelLoadInFlight = /* @__PURE__ */ new Map();
|
|
9120
|
+
/** The last load issued on each pool — loads into one pool run one at a time. */
|
|
9121
|
+
modelLoadTail = /* @__PURE__ */ new Map();
|
|
9122
|
+
/** Stands in for "the node-default pool" before it exists. */
|
|
9123
|
+
nullPoolKey = {};
|
|
9124
|
+
/** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
|
|
9125
|
+
loadGovernor = new ModelLoadGovernor();
|
|
8491
9126
|
/** Ensure all models needed by steps are downloaded and loaded in the engine pool.
|
|
8492
9127
|
* `factory`/`format` default to the node's engine; a per-device dispatch passes
|
|
8493
|
-
* that device's factory + format (Phase 2 multi-device).
|
|
8494
|
-
|
|
8495
|
-
|
|
8496
|
-
const
|
|
9128
|
+
* that device's factory + format (Phase 2 multi-device). `context` names the
|
|
9129
|
+
* camera and the accelerator on the line a rejected load writes. */
|
|
9130
|
+
async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
|
|
9131
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9132
|
+
if (needed.length === 0) return;
|
|
9133
|
+
const pool = factory ?? this.nullPoolKey;
|
|
9134
|
+
const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
|
|
9135
|
+
const models = needed.map(stepModelKey);
|
|
9136
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9137
|
+
if (refusal !== null) {
|
|
9138
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
|
|
9139
|
+
throw refusal;
|
|
9140
|
+
}
|
|
9141
|
+
const setKey = [...models].sort().join(",");
|
|
9142
|
+
let perPool = this.modelLoadInFlight.get(pool);
|
|
9143
|
+
if (perPool === void 0) {
|
|
9144
|
+
perPool = /* @__PURE__ */ new Map();
|
|
9145
|
+
this.modelLoadInFlight.set(pool, perPool);
|
|
9146
|
+
}
|
|
9147
|
+
const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
|
|
9148
|
+
try {
|
|
9149
|
+
await flight.work;
|
|
9150
|
+
} catch (err) {
|
|
9151
|
+
if (err instanceof ModelLoadRefusedError) {
|
|
9152
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
|
|
9153
|
+
throw err;
|
|
9154
|
+
}
|
|
9155
|
+
const failed = err instanceof StepVariantLoadError ? `${err.stepId}/${err.modelId}` : models[0] ?? "";
|
|
9156
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, errMsg(err));
|
|
9157
|
+
throw err;
|
|
9158
|
+
}
|
|
9159
|
+
}
|
|
9160
|
+
/**
|
|
9161
|
+
* Why `models` may not be loaded on `pool` right now — a back-off after a
|
|
9162
|
+
* failure, or a refusal by name after repeated compile timeouts — or `null`.
|
|
9163
|
+
*/
|
|
9164
|
+
loadRefusal(pool, deviceKey, models) {
|
|
9165
|
+
const verdict = this.loadGovernor.check(pool, deviceKey, models);
|
|
9166
|
+
if (verdict.kind === "go") return null;
|
|
9167
|
+
return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
|
|
9168
|
+
}
|
|
9169
|
+
/**
|
|
9170
|
+
* A compile that outlived its soft bound came back on `factory`'s pool
|
|
9171
|
+
* (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
|
|
9172
|
+
* again, so that model's back-off is lifted at once rather than waited out;
|
|
9173
|
+
* no other model's is. A FAILURE is one more failure of that model, and its
|
|
9174
|
+
* back-off grows.
|
|
9175
|
+
*/
|
|
9176
|
+
noteLateCompile(factory, deviceKey, event) {
|
|
9177
|
+
const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
|
|
9178
|
+
if (event.ok) {
|
|
9179
|
+
this.log.info("model loadable again after late compile", { meta: {
|
|
9180
|
+
deviceKey,
|
|
9181
|
+
model: event.model,
|
|
9182
|
+
backoffLifted: models
|
|
9183
|
+
} });
|
|
9184
|
+
return;
|
|
9185
|
+
}
|
|
9186
|
+
this.log.warn("late compile failed — the model stays backed off", { meta: {
|
|
9187
|
+
deviceKey,
|
|
9188
|
+
model: event.model,
|
|
9189
|
+
backoffExtended: models,
|
|
9190
|
+
error: event.error ?? null
|
|
9191
|
+
} });
|
|
9192
|
+
}
|
|
9193
|
+
/** The enabled steps the factory's pool does not reflect yet. */
|
|
9194
|
+
stepsNeedingLoad(steps, factory) {
|
|
8497
9195
|
const needed = [];
|
|
8498
|
-
for (const step of
|
|
9196
|
+
for (const step of flattenSteps(steps)) {
|
|
8499
9197
|
if (!step.enabled) continue;
|
|
8500
9198
|
if (factory !== null && !factory.needsPoolUpdate(step)) continue;
|
|
8501
9199
|
needed.push(step);
|
|
8502
9200
|
}
|
|
8503
|
-
|
|
8504
|
-
|
|
8505
|
-
|
|
8506
|
-
|
|
8507
|
-
|
|
8508
|
-
|
|
8509
|
-
|
|
8510
|
-
|
|
8511
|
-
|
|
8512
|
-
|
|
8513
|
-
|
|
8514
|
-
|
|
8515
|
-
|
|
8516
|
-
|
|
9201
|
+
return needed;
|
|
9202
|
+
}
|
|
9203
|
+
/**
|
|
9204
|
+
* Issue ONE load for a model set on a pool, queued behind the pool's
|
|
9205
|
+
* previous load, and record its outcome with the governor. The set is
|
|
9206
|
+
* re-derived once the queue reaches it: the load ahead may have brought
|
|
9207
|
+
* some of it in.
|
|
9208
|
+
*/
|
|
9209
|
+
issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
|
|
9210
|
+
const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
|
|
9211
|
+
const flight = {
|
|
9212
|
+
work: Promise.resolve(),
|
|
9213
|
+
generation: null
|
|
9214
|
+
};
|
|
9215
|
+
flight.work = (async () => {
|
|
9216
|
+
await previous;
|
|
9217
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9218
|
+
if (needed.length === 0) return;
|
|
9219
|
+
const models = needed.map(stepModelKey);
|
|
9220
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9221
|
+
if (refusal !== null) throw refusal;
|
|
9222
|
+
try {
|
|
9223
|
+
await this.downloadNeededModels(needed, format);
|
|
9224
|
+
this.log.info("Loading additional models for benchmark", { meta: {
|
|
9225
|
+
count: needed.length,
|
|
9226
|
+
models,
|
|
9227
|
+
deviceKey
|
|
9228
|
+
} });
|
|
9229
|
+
await factory.loadAdditional(needed);
|
|
9230
|
+
} catch (err) {
|
|
9231
|
+
flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
|
|
9232
|
+
throw err;
|
|
8517
9233
|
}
|
|
8518
|
-
this.
|
|
8519
|
-
count: needed.length,
|
|
8520
|
-
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8521
|
-
} });
|
|
8522
|
-
await factory.loadAdditional(needed);
|
|
9234
|
+
this.loadGovernor.recordSuccess(pool, deviceKey, models);
|
|
8523
9235
|
})();
|
|
8524
|
-
|
|
8525
|
-
|
|
8526
|
-
|
|
8527
|
-
|
|
8528
|
-
if (
|
|
9236
|
+
perPool.set(setKey, flight);
|
|
9237
|
+
const tail = flight.work.catch(() => void 0);
|
|
9238
|
+
this.modelLoadTail.set(pool, tail);
|
|
9239
|
+
tail.then(() => {
|
|
9240
|
+
if (perPool.get(setKey) === flight) perPool.delete(setKey);
|
|
9241
|
+
if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
|
|
9242
|
+
});
|
|
9243
|
+
return flight;
|
|
9244
|
+
}
|
|
9245
|
+
/** Charge a failed load to the model that failed; say so once if it is now refused. */
|
|
9246
|
+
recordModelLoadFailure(pool, deviceKey, models, err) {
|
|
9247
|
+
const failedModels = err instanceof StepVariantLoadError ? [`${err.stepId}/${err.modelId}`] : models;
|
|
9248
|
+
const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
|
|
9249
|
+
const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
|
|
9250
|
+
let generation = 0;
|
|
9251
|
+
for (const model of failedModels) {
|
|
9252
|
+
const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
|
|
9253
|
+
generation = outcome.generation;
|
|
9254
|
+
if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
|
|
9255
|
+
deviceKey,
|
|
9256
|
+
model,
|
|
9257
|
+
compileTimeouts: outcome.compileTimeouts,
|
|
9258
|
+
error: errMsg(err),
|
|
9259
|
+
device: "keeps serving every other model",
|
|
9260
|
+
rearm: "pipelineExecutor.rearmInferenceDevice"
|
|
9261
|
+
} });
|
|
9262
|
+
}
|
|
9263
|
+
return generation;
|
|
9264
|
+
}
|
|
9265
|
+
/**
|
|
9266
|
+
* One WARN per camera per state change of the model that cost it its frame.
|
|
9267
|
+
* The ERROR for the failed load itself is written once, where the load
|
|
9268
|
+
* failed (`Step variant load failed`); this line answers "which cameras
|
|
9269
|
+
* did it cost?", tagged so it can be counted per camera.
|
|
9270
|
+
*/
|
|
9271
|
+
reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
|
|
9272
|
+
if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
|
|
9273
|
+
this.log.warn("model load for dispatch failed", {
|
|
9274
|
+
...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
|
|
9275
|
+
meta: {
|
|
9276
|
+
deviceKey,
|
|
9277
|
+
models,
|
|
9278
|
+
model,
|
|
9279
|
+
state,
|
|
9280
|
+
error
|
|
9281
|
+
}
|
|
9282
|
+
});
|
|
9283
|
+
}
|
|
9284
|
+
/** Download any model of `needed` missing in `format` (the device's). */
|
|
9285
|
+
async downloadNeededModels(needed, format) {
|
|
9286
|
+
for (const step of needed) {
|
|
9287
|
+
const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
|
|
9288
|
+
if (modelEntry && !isModelDownloaded(this.modelsDir, modelEntry, format)) {
|
|
9289
|
+
if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
|
|
9290
|
+
if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
|
|
9291
|
+
this.log.info("Downloading model for step", { meta: {
|
|
9292
|
+
modelId: step.modelId,
|
|
9293
|
+
format,
|
|
9294
|
+
step: step.addonId
|
|
9295
|
+
} });
|
|
9296
|
+
await this.downloadWithRetry(modelEntry, format, 3);
|
|
9297
|
+
}
|
|
8529
9298
|
}
|
|
8530
9299
|
}
|
|
8531
9300
|
/** Download a model with retry + exponential backoff */
|
|
@@ -8626,12 +9395,11 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8626
9395
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
8627
9396
|
steps: input.steps,
|
|
8628
9397
|
deviceKey: input.deviceKey,
|
|
8629
|
-
engineOverride: input.engine,
|
|
8630
9398
|
deviceId: input.deviceId,
|
|
8631
9399
|
plane: void 0
|
|
8632
9400
|
});
|
|
8633
9401
|
const dispatchDeviceKey = dispatchResolution.deviceKey;
|
|
8634
|
-
const resolveFormat = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format :
|
|
9402
|
+
const resolveFormat = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format : this.currentEngine.format;
|
|
8635
9403
|
const benchmarkSteps = dispatchResolution.steps;
|
|
8636
9404
|
const enabledSteps = flattenEnabledVideoSteps(benchmarkSteps);
|
|
8637
9405
|
if (enabledSteps.length === 0) throw new Error("runPipelineBatch: no enabled steps");
|
|
@@ -8652,53 +9420,48 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8652
9420
|
const uniformDims = input.frames.every((f) => f.width === firstFrame.width && f.height === firstFrame.height);
|
|
8653
9421
|
const singleRoot = enabledSteps.length === 1;
|
|
8654
9422
|
await this.ensureEngineFactory();
|
|
8655
|
-
const
|
|
8656
|
-
|
|
8657
|
-
|
|
8658
|
-
|
|
8659
|
-
|
|
8660
|
-
|
|
8661
|
-
|
|
8662
|
-
|
|
8663
|
-
|
|
8664
|
-
|
|
8665
|
-
|
|
8666
|
-
|
|
8667
|
-
|
|
8668
|
-
|
|
8669
|
-
|
|
8670
|
-
|
|
8671
|
-
|
|
8672
|
-
|
|
8673
|
-
|
|
8674
|
-
|
|
8675
|
-
|
|
8676
|
-
|
|
8677
|
-
|
|
8678
|
-
|
|
8679
|
-
|
|
8680
|
-
|
|
8681
|
-
|
|
8682
|
-
|
|
8683
|
-
|
|
8684
|
-
|
|
8685
|
-
|
|
8686
|
-
|
|
8687
|
-
|
|
8688
|
-
|
|
8689
|
-
|
|
8690
|
-
|
|
8691
|
-
|
|
8692
|
-
|
|
8693
|
-
|
|
8694
|
-
|
|
8695
|
-
|
|
8696
|
-
|
|
8697
|
-
const perFrameMs = totalMs / rawResults.length;
|
|
8698
|
-
return { results: rawResults.map((raw, i) => assembleBatchedFrameResult(raw, input.frames[i], rootStep.addonId, input.deviceId ?? 0, perFrameMs, perFrameMs)) };
|
|
8699
|
-
} finally {
|
|
8700
|
-
restoreEngine();
|
|
8701
|
-
}
|
|
9423
|
+
const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
9424
|
+
if (!factory) throw new Error("runPipelineBatch: factory not initialised");
|
|
9425
|
+
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
|
|
9426
|
+
const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
|
|
9427
|
+
this.log.info("runPipelineBatch path decision", { meta: {
|
|
9428
|
+
phase: "batch",
|
|
9429
|
+
canFastPath,
|
|
9430
|
+
singleRoot,
|
|
9431
|
+
allRaw,
|
|
9432
|
+
uniformDims,
|
|
9433
|
+
supportsBatch: factory.supportsBatch()
|
|
9434
|
+
} });
|
|
9435
|
+
if (!canFastPath) return { results: await Promise.all(input.frames.map((frame) => this.runPipeline({
|
|
9436
|
+
steps: input.steps,
|
|
9437
|
+
frame,
|
|
9438
|
+
deviceId: input.deviceId,
|
|
9439
|
+
sessionId: input.sessionId,
|
|
9440
|
+
deviceKey: dispatchDeviceKey
|
|
9441
|
+
}))) };
|
|
9442
|
+
const items = input.frames.map((frame) => ({
|
|
9443
|
+
raw: Buffer.from(frame.data),
|
|
9444
|
+
width: frame.width,
|
|
9445
|
+
height: frame.height,
|
|
9446
|
+
format: frame.format
|
|
9447
|
+
}));
|
|
9448
|
+
this.log.info("runPipelineBatch fast path dispatch", { meta: {
|
|
9449
|
+
phase: "batch",
|
|
9450
|
+
stepId: rootStep.addonId,
|
|
9451
|
+
modelId: rootStep.modelId,
|
|
9452
|
+
items: items.length,
|
|
9453
|
+
totalRawBytes: items.reduce((s, it) => s + it.raw.length, 0)
|
|
9454
|
+
} });
|
|
9455
|
+
const start = performance.now();
|
|
9456
|
+
const rawResults = await factory.batchInferRaw(rootStep.addonId, items, rootStep.modelId, input.frameId ?? 0);
|
|
9457
|
+
const totalMs = performance.now() - start;
|
|
9458
|
+
this.log.info("runPipelineBatch fast path returned", { meta: {
|
|
9459
|
+
phase: "batch",
|
|
9460
|
+
resultCount: rawResults.length,
|
|
9461
|
+
ms: Math.round(totalMs)
|
|
9462
|
+
} });
|
|
9463
|
+
const perFrameMs = totalMs / rawResults.length;
|
|
9464
|
+
return { results: rawResults.map((raw, i) => assembleBatchedFrameResult(raw, input.frames[i], rootStep.addonId, input.deviceId ?? 0, perFrameMs, perFrameMs)) };
|
|
8702
9465
|
}
|
|
8703
9466
|
async listReferenceImages() {
|
|
8704
9467
|
const dir = this.resolveRefImagesDir();
|
|
@@ -8803,7 +9566,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8803
9566
|
this.poolMemoryGuard = null;
|
|
8804
9567
|
this.inferenceTimeoutGuard?.stop();
|
|
8805
9568
|
this.inferenceTimeoutGuard = null;
|
|
8806
|
-
await this.evictOverrideCache("shutdown");
|
|
8807
9569
|
if (this.engineFactory) {
|
|
8808
9570
|
await this.engineFactory.dispose();
|
|
8809
9571
|
this.engineFactory = null;
|
|
@@ -8885,13 +9647,22 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8885
9647
|
* own runtime's row, so this is not "the node's tuning" — the per-pool truth
|
|
8886
9648
|
* is on each pool's `Python inference pool` log line and in `listLoadedEngines`.
|
|
8887
9649
|
*/
|
|
8888
|
-
|
|
8889
|
-
|
|
9650
|
+
/**
|
|
9651
|
+
* One pool's provisioning — the device pool named by `deviceKey`, else the
|
|
9652
|
+
* node default. Resolved with the SAME override `resolveDeviceFactory` and
|
|
9653
|
+
* `ensureEngineFactory` build the pool with, so the answer describes the
|
|
9654
|
+
* pool that ran the work (D646).
|
|
9655
|
+
*/
|
|
9656
|
+
async getEffectiveTuning(input = {}) {
|
|
9657
|
+
const p = resolvePoolProvisioning(input.deviceKey !== void 0 ? resolveDeviceEngine(input.deviceKey) : this.currentEngine, this.executorOptions.provisioning);
|
|
8890
9658
|
return {
|
|
9659
|
+
deviceKey: input.deviceKey ?? null,
|
|
9660
|
+
runtime: p.runtime,
|
|
8891
9661
|
batchMode: p.batchMode,
|
|
8892
9662
|
windowMs: p.windowMs,
|
|
8893
9663
|
maxBatchSize: p.maxBatchSize,
|
|
8894
|
-
concurrency: p.concurrency
|
|
9664
|
+
concurrency: p.concurrency,
|
|
9665
|
+
numWorkers: p.numWorkers
|
|
8895
9666
|
};
|
|
8896
9667
|
}
|
|
8897
9668
|
/**
|
|
@@ -8906,7 +9677,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8906
9677
|
if (this.engineFactory.isReady()) return;
|
|
8907
9678
|
const dead = this.engineFactory;
|
|
8908
9679
|
this.engineFactory = null;
|
|
8909
|
-
this.
|
|
9680
|
+
this.noteFactoryDeath(defaultDeviceKey, dead);
|
|
8910
9681
|
await dead.dispose().catch(() => void 0);
|
|
8911
9682
|
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8912
9683
|
}
|
|
@@ -8924,7 +9695,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8924
9695
|
logger: this.log.child("engine"),
|
|
8925
9696
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8926
9697
|
provisioning: this.executorOptions.provisioning,
|
|
8927
|
-
resolveCustomModel: this.customModelResolver
|
|
9698
|
+
resolveCustomModel: this.customModelResolver,
|
|
9699
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
|
|
8928
9700
|
});
|
|
8929
9701
|
try {
|
|
8930
9702
|
await factory.initialize([]);
|
|
@@ -9044,15 +9816,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9044
9816
|
* dispose — which rejects whatever was still in flight on it with a reason
|
|
9045
9817
|
* rather than letting those requests sit until their own deadlines.
|
|
9046
9818
|
*/
|
|
9047
|
-
async condemnDeviceFactory(deviceKey, factory
|
|
9819
|
+
async condemnDeviceFactory(deviceKey, factory) {
|
|
9048
9820
|
if (this.factoriesByDevice.get(deviceKey) === factory) {
|
|
9049
9821
|
this.factoriesByDevice.delete(deviceKey);
|
|
9050
9822
|
this.deviceReaper.cancel(deviceKey);
|
|
9051
|
-
this.
|
|
9823
|
+
this.noteFactoryDeath(deviceKey, factory);
|
|
9052
9824
|
}
|
|
9053
9825
|
await factory.dispose().catch(() => void 0);
|
|
9054
9826
|
}
|
|
9055
9827
|
/**
|
|
9828
|
+
* Charge one dead pool to its device — unless it died of a compile that was
|
|
9829
|
+
* still running at its hard bound (D653, fix round 1). That death is not a
|
|
9830
|
+
* crash of the DEVICE: the model that hung was already counted when its
|
|
9831
|
+
* load answered `compile-timeout`, and the same model timing out again on
|
|
9832
|
+
* the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
|
|
9833
|
+
* the respawns. A crash stays a crash.
|
|
9834
|
+
*/
|
|
9835
|
+
noteFactoryDeath(deviceKey, factory) {
|
|
9836
|
+
const cause = factory.getDeathCause();
|
|
9837
|
+
if (cause !== null && !cause.chargesDeviceBudget) {
|
|
9838
|
+
this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
|
|
9839
|
+
deviceKey,
|
|
9840
|
+
reason: cause.reason,
|
|
9841
|
+
cause: cause.message
|
|
9842
|
+
} });
|
|
9843
|
+
return;
|
|
9844
|
+
}
|
|
9845
|
+
this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
|
|
9846
|
+
}
|
|
9847
|
+
/**
|
|
9056
9848
|
* Every inference device this node currently refuses, and why.
|
|
9057
9849
|
*
|
|
9058
9850
|
* The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
|
|
@@ -9087,7 +9879,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9087
9879
|
state: "backoff",
|
|
9088
9880
|
since: Date.now(),
|
|
9089
9881
|
deaths: 0,
|
|
9090
|
-
lastError:
|
|
9882
|
+
lastError: deathReasonOf(factory)
|
|
9091
9883
|
});
|
|
9092
9884
|
}
|
|
9093
9885
|
return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
|
|
@@ -9104,10 +9896,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9104
9896
|
* no-op, reported as such.
|
|
9105
9897
|
*/
|
|
9106
9898
|
async rearmInferenceDevice(input) {
|
|
9107
|
-
const
|
|
9899
|
+
const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
|
|
9900
|
+
const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
|
|
9901
|
+
const rearmed = rearmedDevice || rearmedModels > 0;
|
|
9108
9902
|
this.log.info("inference device re-armed by operator", { meta: {
|
|
9109
9903
|
deviceKey: input.deviceKey,
|
|
9110
|
-
rearmed
|
|
9904
|
+
rearmed,
|
|
9905
|
+
rearmedModels
|
|
9111
9906
|
} });
|
|
9112
9907
|
return { rearmed };
|
|
9113
9908
|
}
|
|
@@ -9124,7 +9919,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9124
9919
|
this.deviceReaper.touch(deviceKey);
|
|
9125
9920
|
return existing;
|
|
9126
9921
|
}
|
|
9127
|
-
await this.condemnDeviceFactory(deviceKey, existing
|
|
9922
|
+
await this.condemnDeviceFactory(deviceKey, existing);
|
|
9128
9923
|
this.refuseIfDeviceUnusable(deviceKey);
|
|
9129
9924
|
}
|
|
9130
9925
|
const inflight = this.deviceFactoryInflight.get(deviceKey);
|
|
@@ -9137,7 +9932,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9137
9932
|
logger: this.log.child(`engine:${deviceKey}`),
|
|
9138
9933
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
9139
9934
|
provisioning: this.executorOptions.provisioning,
|
|
9140
|
-
resolveCustomModel: this.customModelResolver
|
|
9935
|
+
resolveCustomModel: this.customModelResolver,
|
|
9936
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
|
|
9141
9937
|
});
|
|
9142
9938
|
try {
|
|
9143
9939
|
await factory.initialize([]);
|
|
@@ -9279,9 +10075,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9279
10075
|
}
|
|
9280
10076
|
async listLoadedEngines() {
|
|
9281
10077
|
const out = [];
|
|
9282
|
-
|
|
9283
|
-
const runtimeFactoryIsOverride = this.engineFactory !== null && this.engineFactory === overrideFactory;
|
|
9284
|
-
if (this.engineFactory && !runtimeFactoryIsOverride) {
|
|
10078
|
+
if (this.engineFactory) {
|
|
9285
10079
|
const eng = this.currentEngine;
|
|
9286
10080
|
const engineKey = `${eng.runtime}/${eng.backend}/${eng.device ?? "default"}`;
|
|
9287
10081
|
const loaded = this.engineFactory.listLoaded();
|
|
@@ -9296,22 +10090,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9296
10090
|
idleTtlMs: null
|
|
9297
10091
|
});
|
|
9298
10092
|
}
|
|
9299
|
-
if (this.overrideCache) {
|
|
9300
|
-
const eng = this.overrideCache.engine;
|
|
9301
|
-
const engineKey = `${eng.runtime}/${eng.backend}/${eng.device ?? "default"} (warm)`;
|
|
9302
|
-
const loaded = this.overrideCache.factory.listLoaded();
|
|
9303
|
-
const idleMs = Math.max(0, Date.now() - this.overrideCache.lastUsedMs);
|
|
9304
|
-
if (idleMs <= DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS) out.push({
|
|
9305
|
-
engineKey,
|
|
9306
|
-
engine: eng,
|
|
9307
|
-
modelsLoaded: loaded.map((l) => `${l.stepId}/${l.modelId}`),
|
|
9308
|
-
inUseByCameras: [],
|
|
9309
|
-
kind: "warm-override",
|
|
9310
|
-
poolPid: this.overrideCache.factory.getPoolPid(),
|
|
9311
|
-
idleMs,
|
|
9312
|
-
idleTtlMs: DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS
|
|
9313
|
-
});
|
|
9314
|
-
}
|
|
9315
10093
|
for (const [deviceKey, factory] of this.factoriesByDevice) {
|
|
9316
10094
|
const eng = resolveDeviceEngine(deviceKey);
|
|
9317
10095
|
const loaded = factory.listLoaded();
|
|
@@ -9340,14 +10118,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9340
10118
|
return { success: true };
|
|
9341
10119
|
}
|
|
9342
10120
|
async killEngine(input) {
|
|
9343
|
-
if (this.overrideCache && enginesEqual(this.overrideCache.engine, input.engine)) {
|
|
9344
|
-
await this.evictOverrideCache("killEngine");
|
|
9345
|
-
this.log.info("Override engine killed", { meta: {
|
|
9346
|
-
...input.engine,
|
|
9347
|
-
force: input.force ?? false
|
|
9348
|
-
} });
|
|
9349
|
-
return { success: true };
|
|
9350
|
-
}
|
|
9351
10121
|
if (!this.engineFactory) return {
|
|
9352
10122
|
success: false,
|
|
9353
10123
|
reason: "not loaded"
|
|
@@ -9865,7 +10635,7 @@ var DetectionPipelineAddon = class extends BaseAddon {
|
|
|
9865
10635
|
/**
|
|
9866
10636
|
* Embedded Python path resolved once at boot via
|
|
9867
10637
|
* `ctx.deps.ensurePython()`. Empty string means the download failed
|
|
9868
|
-
*
|
|
10638
|
+
* — the provider's
|
|
9869
10639
|
* `ensureBackendDeps` and EngineFactory's `initPythonPool` raise a
|
|
9870
10640
|
* clear error in that case.
|
|
9871
10641
|
*/
|