@camstack/addon-pipeline 1.2.294 → 1.2.296
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +241 -0
- package/dist/audio-analyzer/index.js +2 -2
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/{default-detection-model-0dPKRKUD.mjs → default-detection-model-Co578D8C.mjs} +181 -99
- package/dist/{default-detection-model-D24AJOTn.js → default-detection-model-D1daTtqT.js} +181 -99
- package/dist/detection-pipeline/index.js +1301 -531
- package/dist/detection-pipeline/index.mjs +1301 -531
- package/dist/{dist-CJR259Xf.js → dist-8up-f2TX.js} +3688 -2698
- package/dist/{dist-RXbmRAwP.mjs → dist-CCd0Q3nr.mjs} +3676 -2698
- package/dist/motion-wasm/index.js +1 -1
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/{node-atmRSHPk.mjs → node-DgMSXSWP.mjs} +1 -1
- package/dist/{node-DWg9zbY1.js → node-lpQgHes9.js} +1 -1
- package/dist/pipeline-runner/index.js +975 -270
- package/dist/pipeline-runner/index.mjs +975 -270
- package/dist/{process-memory-BJUXvTjd.js → process-memory-CX_92V_r.js} +1 -1
- package/dist/{process-memory-BgFOHFnx.mjs → process-memory-DFC_O5zE.mjs} +1 -1
- package/dist/recorder/index.js +14 -6
- package/dist/recorder/index.mjs +14 -6
- package/dist/{segment-demux-js-C_fPJub3.js → segment-demux-js-DzBx6NN2.js} +1 -1
- package/dist/{segment-demux-js-G7wFpHzn.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
- package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
- package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
- package/dist/stream-broker/_stub.js +2 -2
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-6IyM-BIn.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-BBYHWS3M.mjs → hostInit-BPtppL3W.mjs} +2 -2
- package/dist/stream-broker/index.js +4 -4
- package/dist/stream-broker/index.mjs +4 -4
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
- package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
- package/package.json +3 -2
- package/python/inference_pool.py +422 -64
- package/python/postprocessors/__init__.py +2 -0
- package/python/postprocessors/ssd.py +73 -17
- package/python/postprocessors/test_ssd.py +205 -0
- package/python/postprocessors/test_yunet.py +292 -0
- package/python/postprocessors/testdata/ssdlite_mobiledet_outputs.json +1 -0
- package/python/postprocessors/yunet.py +275 -0
- package/python/test_inference_pool_compile_off_loop.py +414 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-6IHzlLJ_.mjs +0 -26
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DoyA71_q.mjs +0 -26
|
@@ -3,11 +3,11 @@ Object.defineProperties(exports, {
|
|
|
3
3
|
[Symbol.toStringTag]: { value: "Module" }
|
|
4
4
|
});
|
|
5
5
|
const require_chunk = require("../chunk-emK7D4bc.js");
|
|
6
|
-
const require_dist = require("../dist-
|
|
7
|
-
const require_node = require("../node-
|
|
8
|
-
const require_default_detection_model = require("../default-detection-model-
|
|
6
|
+
const require_dist = require("../dist-8up-f2TX.js");
|
|
7
|
+
const require_node = require("../node-lpQgHes9.js");
|
|
8
|
+
const require_default_detection_model = require("../default-detection-model-D1daTtqT.js");
|
|
9
9
|
const require_lazy_sharp = require("../lazy-sharp-BuwKlQLR.js");
|
|
10
|
-
const require_process_memory = require("../process-memory-
|
|
10
|
+
const require_process_memory = require("../process-memory-CX_92V_r.js");
|
|
11
11
|
let node_os = require("node:os");
|
|
12
12
|
node_os = require_chunk.__toESM(node_os);
|
|
13
13
|
let node_fs = require("node:fs");
|
|
@@ -422,6 +422,12 @@ var ClusterModelSource = class {
|
|
|
422
422
|
const next = require_dist.pickClusterStepModels(view);
|
|
423
423
|
const nextSettings = require_dist.pickClusterStepSettings(view);
|
|
424
424
|
this.lastReadAtMs = this.now();
|
|
425
|
+
const invalid = require_dist.pickInvalidClusterStepModels(view);
|
|
426
|
+
if (invalid.length > 0) this.logger.warn("cluster step model row holds a value that is not a model id — running the catalog default for that step; no re-embed pass will run on it", { meta: {
|
|
427
|
+
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
428
|
+
invalid,
|
|
429
|
+
inUse: Object.fromEntries(invalid.map((e) => [e.stepId, next[e.stepId]]))
|
|
430
|
+
} });
|
|
425
431
|
const changed = Object.keys(next).filter((stepId) => next[stepId] !== this.models[stepId]);
|
|
426
432
|
const settingsChanged = Object.keys(nextSettings).some((stepId) => {
|
|
427
433
|
const from = this.settings[stepId] ?? {};
|
|
@@ -569,6 +575,98 @@ var DeviceOverrideMirror = class {
|
|
|
569
575
|
}
|
|
570
576
|
};
|
|
571
577
|
//#endregion
|
|
578
|
+
//#region src/detection-pipeline/engine/pool-worker-health.ts
|
|
579
|
+
/**
|
|
580
|
+
* The one table of load bounds (D653, fix round 1).
|
|
581
|
+
*
|
|
582
|
+
* The soft bound decides how soon the operator HEARS about a slow compile; the
|
|
583
|
+
* hard bound decides when a compile is given up for hung. A single 120 s bound
|
|
584
|
+
* that also killed the compile was wrong twice: a slow but finite compile was
|
|
585
|
+
* killed before it could write its cache, so every respawn faced the same cold
|
|
586
|
+
* compile and three of them walked the device to `failed`; and on CoreML the
|
|
587
|
+
* bound was shorter than the cross-process compile-cache lock wait alone.
|
|
588
|
+
*/
|
|
589
|
+
var POOL_MODEL_LOAD_BOUNDS = {
|
|
590
|
+
openvino: {
|
|
591
|
+
softMs: 12e4,
|
|
592
|
+
hardMs: 6e5,
|
|
593
|
+
why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
|
|
594
|
+
gilMayBeHeldDuringCompile: false,
|
|
595
|
+
gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
|
|
596
|
+
},
|
|
597
|
+
coreml: {
|
|
598
|
+
softMs: 3e5,
|
|
599
|
+
hardMs: 9e5,
|
|
600
|
+
why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
|
|
601
|
+
gilMayBeHeldDuringCompile: true,
|
|
602
|
+
gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
|
|
603
|
+
},
|
|
604
|
+
onnxruntime: {
|
|
605
|
+
softMs: 12e4,
|
|
606
|
+
hardMs: 6e5,
|
|
607
|
+
why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
|
|
608
|
+
gilMayBeHeldDuringCompile: false,
|
|
609
|
+
gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
|
|
610
|
+
},
|
|
611
|
+
edgetpu: {
|
|
612
|
+
softMs: 12e4,
|
|
613
|
+
hardMs: 6e5,
|
|
614
|
+
why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
|
|
615
|
+
gilMayBeHeldDuringCompile: false,
|
|
616
|
+
gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
|
|
617
|
+
}
|
|
618
|
+
};
|
|
619
|
+
/**
|
|
620
|
+
* The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
|
|
621
|
+
* so the worker's NAMED answer always lands first.
|
|
622
|
+
*/
|
|
623
|
+
var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
|
|
624
|
+
function chargesDeviceBudget(reason) {
|
|
625
|
+
return reason !== "compile-hung";
|
|
626
|
+
}
|
|
627
|
+
/**
|
|
628
|
+
* Tracks live inference outcomes on one worker and says when its executor has
|
|
629
|
+
* produced nothing for too long. Consulted on demand — no timer.
|
|
630
|
+
*
|
|
631
|
+
* A SHED (`dropped`) does not count as a result: the Python loop answers sheds
|
|
632
|
+
* itself, so a worker whose executor threads are hung keeps shedding happily.
|
|
633
|
+
*/
|
|
634
|
+
var InferStallDetector = class {
|
|
635
|
+
unanswered = 0;
|
|
636
|
+
lastResultAt;
|
|
637
|
+
/**
|
|
638
|
+
* When the current streak of unanswered requests began — the first one after
|
|
639
|
+
* a result. IDLE time is not silence: quiet cameras send nothing, and a
|
|
640
|
+
* worker that was asked nothing has failed nothing. Measuring from the last
|
|
641
|
+
* result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
|
|
642
|
+
* cameras) poison a healthy worker and charge the device.
|
|
643
|
+
*/
|
|
644
|
+
streakStartedAt = null;
|
|
645
|
+
constructor(now) {
|
|
646
|
+
this.lastResultAt = now;
|
|
647
|
+
}
|
|
648
|
+
/** A reply that carried a real result (or a real error) from the executor. */
|
|
649
|
+
noteResult(now) {
|
|
650
|
+
this.unanswered = 0;
|
|
651
|
+
this.lastResultAt = now;
|
|
652
|
+
this.streakStartedAt = null;
|
|
653
|
+
}
|
|
654
|
+
/**
|
|
655
|
+
* A live request that ended without a result (deadline expiry or a shed).
|
|
656
|
+
* Returns the stall when the worker has crossed both thresholds.
|
|
657
|
+
*/
|
|
658
|
+
noteUnanswered(now) {
|
|
659
|
+
this.unanswered += 1;
|
|
660
|
+
if (this.streakStartedAt === null) this.streakStartedAt = now;
|
|
661
|
+
const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
|
|
662
|
+
if (this.unanswered >= 20 && silentMs >= 3e4) return {
|
|
663
|
+
unanswered: this.unanswered,
|
|
664
|
+
silentMs
|
|
665
|
+
};
|
|
666
|
+
return null;
|
|
667
|
+
}
|
|
668
|
+
};
|
|
669
|
+
//#endregion
|
|
572
670
|
//#region src/detection-pipeline/engine/shared-inference-pool.ts
|
|
573
671
|
/**
|
|
574
672
|
* SharedInferencePool — TypeScript wrapper for inference_pool.py.
|
|
@@ -610,6 +708,27 @@ var RAW_FMT_CODE = {
|
|
|
610
708
|
gray: 2
|
|
611
709
|
};
|
|
612
710
|
/**
|
|
711
|
+
* A load the worker answered with a machine-readable failure class. The
|
|
712
|
+
* provider tells a `compile-timeout` (counted against THAT MODEL on the
|
|
713
|
+
* device) from an ordinary failure by `reason`, never by parsing text.
|
|
714
|
+
*/
|
|
715
|
+
var PoolModelLoadError = class extends Error {
|
|
716
|
+
reason;
|
|
717
|
+
/** The model's file stem, as the worker named it. */
|
|
718
|
+
model;
|
|
719
|
+
constructor(message, reason, model) {
|
|
720
|
+
super(message);
|
|
721
|
+
this.name = "PoolModelLoadError";
|
|
722
|
+
this.reason = reason;
|
|
723
|
+
this.model = model;
|
|
724
|
+
}
|
|
725
|
+
};
|
|
726
|
+
/**
|
|
727
|
+
* Request id the worker uses for UNSOLICITED events — a reply to no request
|
|
728
|
+
* (`compile-finished-late`). Never allocated to a request.
|
|
729
|
+
*/
|
|
730
|
+
var WORKER_EVENT_REQ_ID = 4294967295;
|
|
731
|
+
/**
|
|
613
732
|
* Per-inference-request reply timeout (ms). Turns a wedged request (worker
|
|
614
733
|
* alive but no reply) into a rejection so every caller settles — runtime frame
|
|
615
734
|
* dispatch drops the frame; the benchmark full-tree run rejects the wedged
|
|
@@ -618,6 +737,22 @@ var RAW_FMT_CODE = {
|
|
|
618
737
|
* large frame) never trips; override via env for constrained hardware.
|
|
619
738
|
*/
|
|
620
739
|
var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
|
|
740
|
+
/** Commands that compile a model — the only ones that get the load deadline. */
|
|
741
|
+
var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
|
|
742
|
+
/**
|
|
743
|
+
* Deadline of the liveness probe (`mem_stats`) sent after a command times out.
|
|
744
|
+
* The worker answers `mem_stats` on its event loop, never behind a compile
|
|
745
|
+
* (D653), so a healthy worker replies in milliseconds even mid-compile; ten
|
|
746
|
+
* seconds only has to outlast a GC pause or a burst of inference replies.
|
|
747
|
+
*/
|
|
748
|
+
var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
|
|
749
|
+
/**
|
|
750
|
+
* How long a poisoned worker may keep its in-flight LIVE requests before it is
|
|
751
|
+
* killed anyway. A poisoned worker takes no new work, and every live request
|
|
752
|
+
* carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
|
|
753
|
+
* by then; the second of slack keeps the backstop from racing the last reply.
|
|
754
|
+
*/
|
|
755
|
+
var POOL_POISON_DRAIN_SLACK_MS = 1e3;
|
|
621
756
|
/**
|
|
622
757
|
* Max inference requests outstanding to ONE worker before new ones are SHED.
|
|
623
758
|
*
|
|
@@ -805,6 +940,27 @@ var PoolWorker = class {
|
|
|
805
940
|
wireVersion = 1;
|
|
806
941
|
nextRequestId = 1;
|
|
807
942
|
ready = false;
|
|
943
|
+
/** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
|
|
944
|
+
* a poisoned worker is recycled, not revived. */
|
|
945
|
+
poisonVerdict = null;
|
|
946
|
+
/** The kill of a poisoned worker has been started. */
|
|
947
|
+
recycling = false;
|
|
948
|
+
/** Kills a poisoned worker whose drain did not finish in time. */
|
|
949
|
+
recycleBackstop = null;
|
|
950
|
+
/** A liveness probe is in flight — one at a time, never a probe storm. */
|
|
951
|
+
probeInFlight = false;
|
|
952
|
+
/** The child has exited (any cause). */
|
|
953
|
+
exited = false;
|
|
954
|
+
/** A compile past its soft bound, still running in the worker (D653). */
|
|
955
|
+
abandonedCompile = null;
|
|
956
|
+
/** Says when live inference has produced nothing for too long (D653). */
|
|
957
|
+
inferStall = new InferStallDetector(Date.now());
|
|
958
|
+
/**
|
|
959
|
+
* A load's HOST deadline expired with no reply since the last reply of any
|
|
960
|
+
* kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
|
|
961
|
+
* is over for this worker until something — anything — comes back.
|
|
962
|
+
*/
|
|
963
|
+
loadDeadlineMissed = false;
|
|
808
964
|
log;
|
|
809
965
|
opts;
|
|
810
966
|
constructor(opts) {
|
|
@@ -829,6 +985,24 @@ var PoolWorker = class {
|
|
|
829
985
|
isReady() {
|
|
830
986
|
return this.ready;
|
|
831
987
|
}
|
|
988
|
+
/** Why this worker was recycled, typed for the provider's budget, or `null`. */
|
|
989
|
+
getDeathCause() {
|
|
990
|
+
const v = this.poisonVerdict;
|
|
991
|
+
const message = this.getPoisonDescription();
|
|
992
|
+
if (v === null || message === null) return null;
|
|
993
|
+
return {
|
|
994
|
+
reason: v.reason,
|
|
995
|
+
chargesDeviceBudget: chargesDeviceBudget(v.reason),
|
|
996
|
+
message
|
|
997
|
+
};
|
|
998
|
+
}
|
|
999
|
+
/** Why this live worker was declared unusable, or `null` (D653). */
|
|
1000
|
+
getPoisonDescription() {
|
|
1001
|
+
const v = this.poisonVerdict;
|
|
1002
|
+
if (v === null) return null;
|
|
1003
|
+
const target = v.command.model !== null ? ` of ${v.command.model}` : "";
|
|
1004
|
+
return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
|
|
1005
|
+
}
|
|
832
1006
|
async initialize(initialModels) {
|
|
833
1007
|
this.process = (0, node_child_process.spawn)(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
|
|
834
1008
|
"pipe",
|
|
@@ -852,7 +1026,11 @@ var PoolWorker = class {
|
|
|
852
1026
|
const spawnedProcess = this.process;
|
|
853
1027
|
spawnedProcess.on("exit", (code, signal) => {
|
|
854
1028
|
this.ready = false;
|
|
1029
|
+
this.exited = true;
|
|
1030
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1031
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
855
1032
|
if (this.process !== spawnedProcess) return;
|
|
1033
|
+
const poisonedBy = this.getPoisonDescription();
|
|
856
1034
|
this.log.error("Worker process exited", { meta: {
|
|
857
1035
|
worker: this.opts.workerLabel,
|
|
858
1036
|
pid: spawnedProcess.pid ?? null,
|
|
@@ -860,9 +1038,10 @@ var PoolWorker = class {
|
|
|
860
1038
|
device: this.opts.device ?? "default",
|
|
861
1039
|
code,
|
|
862
1040
|
signal,
|
|
863
|
-
inFlight: this.pending.size
|
|
1041
|
+
inFlight: this.pending.size,
|
|
1042
|
+
...poisonedBy !== null ? { poisonedBy } : {}
|
|
864
1043
|
} });
|
|
865
|
-
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
|
|
1044
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
|
|
866
1045
|
});
|
|
867
1046
|
this.process.stdout.on("data", (chunk) => {
|
|
868
1047
|
const t0 = Date.now();
|
|
@@ -876,6 +1055,7 @@ var PoolWorker = class {
|
|
|
876
1055
|
runtime: this.opts.poolRuntime,
|
|
877
1056
|
concurrency: this.opts.concurrency,
|
|
878
1057
|
protocolVersion: 2,
|
|
1058
|
+
modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
|
|
879
1059
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
880
1060
|
};
|
|
881
1061
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -898,6 +1078,7 @@ var PoolWorker = class {
|
|
|
898
1078
|
clearTimeout(timeout);
|
|
899
1079
|
if (result["status"] === "ready") {
|
|
900
1080
|
this.ready = true;
|
|
1081
|
+
this.inferStall.noteResult(Date.now());
|
|
901
1082
|
const loadedCount = result["models"];
|
|
902
1083
|
const startupMs = result["startupMs"];
|
|
903
1084
|
const workers = result["workers"] ?? 1;
|
|
@@ -920,7 +1101,7 @@ var PoolWorker = class {
|
|
|
920
1101
|
async infer(modelByte, jpeg, deviceId) {
|
|
921
1102
|
this.ensureReady();
|
|
922
1103
|
const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
|
|
923
|
-
return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
|
|
1104
|
+
return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
|
|
924
1105
|
}
|
|
925
1106
|
async inferRaw(modelByte, raw, width, height, format, deviceId) {
|
|
926
1107
|
this.ensureReady();
|
|
@@ -973,12 +1154,18 @@ var PoolWorker = class {
|
|
|
973
1154
|
const payload = Buffer.allocUnsafe(5);
|
|
974
1155
|
payload[0] = modelByte;
|
|
975
1156
|
payload.writeUInt32LE(frameId, 1);
|
|
976
|
-
return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
|
|
1157
|
+
return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
|
|
977
1158
|
}
|
|
978
1159
|
async sendCommand(cmd) {
|
|
979
1160
|
this.ensureReady();
|
|
1161
|
+
const command = describeCommand(cmd);
|
|
980
1162
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
981
|
-
|
|
1163
|
+
const sentAt = Date.now();
|
|
1164
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
|
|
1165
|
+
const raw = await this.dispatch(MSG_COMMAND, payload, { command });
|
|
1166
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
|
|
1167
|
+
this.inspectCommandReply(raw, command, sentAt);
|
|
1168
|
+
return raw;
|
|
982
1169
|
}
|
|
983
1170
|
/** Command whose reply shape is NOT the load/unload/replace/status envelope
|
|
984
1171
|
* (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
|
|
@@ -986,13 +1173,15 @@ var PoolWorker = class {
|
|
|
986
1173
|
async sendRawCommand(cmd) {
|
|
987
1174
|
this.ensureReady();
|
|
988
1175
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
989
|
-
return this.dispatch(MSG_COMMAND, payload);
|
|
1176
|
+
return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
|
|
990
1177
|
}
|
|
991
1178
|
async dispose() {
|
|
992
1179
|
const proc = this.process;
|
|
993
1180
|
if (!proc) return;
|
|
994
1181
|
this.process = null;
|
|
995
1182
|
this.ready = false;
|
|
1183
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1184
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
996
1185
|
await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
|
|
997
1186
|
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
|
|
998
1187
|
}
|
|
@@ -1035,8 +1224,11 @@ var PoolWorker = class {
|
|
|
1035
1224
|
}
|
|
1036
1225
|
/** Live inference gets seconds; commands and model loads keep the long
|
|
1037
1226
|
* timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
|
|
1038
|
-
deadlineFor(msgType) {
|
|
1039
|
-
|
|
1227
|
+
deadlineFor(msgType, opts = {}) {
|
|
1228
|
+
if (opts.timeoutMs !== void 0) return opts.timeoutMs;
|
|
1229
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
|
|
1230
|
+
if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
|
|
1231
|
+
return POOL_INFER_TIMEOUT_MS;
|
|
1040
1232
|
}
|
|
1041
1233
|
/**
|
|
1042
1234
|
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
@@ -1053,16 +1245,18 @@ var PoolWorker = class {
|
|
|
1053
1245
|
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1054
1246
|
return stamp;
|
|
1055
1247
|
}
|
|
1056
|
-
dispatch(msgType, payload,
|
|
1057
|
-
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1248
|
+
dispatch(msgType, payload, opts = {}) {
|
|
1249
|
+
const shed = this.shedIfSaturated(msgType, opts.deviceId);
|
|
1058
1250
|
if (shed) return Promise.resolve(shed);
|
|
1059
1251
|
const reqId = this.allocRequestId();
|
|
1060
1252
|
return new Promise((resolve, reject) => {
|
|
1061
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1253
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
|
|
1062
1254
|
this.pending.set(reqId, {
|
|
1063
1255
|
resolve,
|
|
1064
1256
|
reject,
|
|
1065
|
-
timer
|
|
1257
|
+
timer,
|
|
1258
|
+
...opts.command !== void 0 ? { command: opts.command } : {},
|
|
1259
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1066
1260
|
});
|
|
1067
1261
|
try {
|
|
1068
1262
|
const stamp = this.deadlineStamp(msgType);
|
|
@@ -1093,8 +1287,9 @@ var PoolWorker = class {
|
|
|
1093
1287
|
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1094
1288
|
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1095
1289
|
*/
|
|
1096
|
-
armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1097
|
-
const
|
|
1290
|
+
armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
|
|
1291
|
+
const { deviceId, command } = opts;
|
|
1292
|
+
const timeoutMs = this.deadlineFor(msgType, opts);
|
|
1098
1293
|
const timer = setTimeout(() => {
|
|
1099
1294
|
if (this.pending.delete(reqId)) {
|
|
1100
1295
|
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
@@ -1118,6 +1313,7 @@ var PoolWorker = class {
|
|
|
1118
1313
|
dropped: true,
|
|
1119
1314
|
shedReason: "deadline-expired"
|
|
1120
1315
|
});
|
|
1316
|
+
this.noteInferUnanswered();
|
|
1121
1317
|
return;
|
|
1122
1318
|
}
|
|
1123
1319
|
this.timedOutCount++;
|
|
@@ -1130,10 +1326,20 @@ var PoolWorker = class {
|
|
|
1130
1326
|
device: this.opts.device ?? "default",
|
|
1131
1327
|
inFlight: this.pending.size,
|
|
1132
1328
|
reqId,
|
|
1133
|
-
timeoutMs
|
|
1329
|
+
timeoutMs,
|
|
1330
|
+
...command !== void 0 ? {
|
|
1331
|
+
command: command.cmd,
|
|
1332
|
+
modelIndex: command.modelIndex,
|
|
1333
|
+
model: command.model
|
|
1334
|
+
} : {}
|
|
1134
1335
|
}
|
|
1135
1336
|
});
|
|
1136
|
-
|
|
1337
|
+
const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
|
|
1338
|
+
if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
|
|
1339
|
+
this.loadDeadlineMissed = true;
|
|
1340
|
+
reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
|
|
1341
|
+
} else reject(new Error(message));
|
|
1342
|
+
if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
|
|
1137
1343
|
}
|
|
1138
1344
|
}, timeoutMs);
|
|
1139
1345
|
timer.unref?.();
|
|
@@ -1144,11 +1350,12 @@ var PoolWorker = class {
|
|
|
1144
1350
|
if (shed) return Promise.resolve(shed);
|
|
1145
1351
|
const reqId = this.allocRequestId();
|
|
1146
1352
|
return new Promise((resolve, reject) => {
|
|
1147
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1353
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
|
|
1148
1354
|
this.pending.set(reqId, {
|
|
1149
1355
|
resolve,
|
|
1150
1356
|
reject,
|
|
1151
|
-
timer
|
|
1357
|
+
timer,
|
|
1358
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1152
1359
|
});
|
|
1153
1360
|
try {
|
|
1154
1361
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
@@ -1170,10 +1377,10 @@ var PoolWorker = class {
|
|
|
1170
1377
|
}
|
|
1171
1378
|
allocRequestId() {
|
|
1172
1379
|
let id = this.nextRequestId;
|
|
1173
|
-
this.nextRequestId = id >=
|
|
1380
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1174
1381
|
while (this.pending.has(id)) {
|
|
1175
1382
|
id = this.nextRequestId;
|
|
1176
|
-
this.nextRequestId = id >=
|
|
1383
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1177
1384
|
}
|
|
1178
1385
|
return id;
|
|
1179
1386
|
}
|
|
@@ -1188,6 +1395,8 @@ var PoolWorker = class {
|
|
|
1188
1395
|
this.process.stdin.write(payload);
|
|
1189
1396
|
}
|
|
1190
1397
|
ensureReady() {
|
|
1398
|
+
const poisonedBy = this.getPoisonDescription();
|
|
1399
|
+
if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
|
|
1191
1400
|
if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
|
|
1192
1401
|
}
|
|
1193
1402
|
/** Time spent in `Buffer.concat` since the last report. */
|
|
@@ -1227,23 +1436,273 @@ var PoolWorker = class {
|
|
|
1227
1436
|
const reqId = this.receiveBuffer.readUInt32LE(4);
|
|
1228
1437
|
const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
|
|
1229
1438
|
this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
|
|
1439
|
+
if (reqId === WORKER_EVENT_REQ_ID) {
|
|
1440
|
+
this.handleWorkerEvent(jsonBytes);
|
|
1441
|
+
continue;
|
|
1442
|
+
}
|
|
1230
1443
|
const entry = this.pending.get(reqId);
|
|
1231
1444
|
if (!entry) {
|
|
1232
1445
|
this.log.warn("Response for unknown request id", { meta: {
|
|
1233
1446
|
worker: this.opts.workerLabel,
|
|
1234
1447
|
reqId
|
|
1235
1448
|
} });
|
|
1449
|
+
this.noteLateReply(jsonBytes);
|
|
1236
1450
|
continue;
|
|
1237
1451
|
}
|
|
1238
1452
|
this.pending.delete(reqId);
|
|
1239
1453
|
if (entry.timer) clearTimeout(entry.timer);
|
|
1240
1454
|
try {
|
|
1241
1455
|
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1456
|
+
if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
|
|
1457
|
+
else this.inferStall.noteResult(Date.now());
|
|
1242
1458
|
entry.resolve(parsed);
|
|
1243
1459
|
} catch (err) {
|
|
1244
1460
|
entry.reject(err instanceof Error ? err : new Error(String(err)));
|
|
1245
1461
|
}
|
|
1246
1462
|
}
|
|
1463
|
+
this.noteAnyReply();
|
|
1464
|
+
if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
|
|
1465
|
+
}
|
|
1466
|
+
/**
|
|
1467
|
+
* Something came back from the worker: its loop was alive just now, so the
|
|
1468
|
+
* missed-deadline verdict is lifted.
|
|
1469
|
+
*/
|
|
1470
|
+
noteAnyReply() {
|
|
1471
|
+
this.loadDeadlineMissed = false;
|
|
1472
|
+
}
|
|
1473
|
+
/**
|
|
1474
|
+
* The invariant the load deadline depends on (D653 § 2): loads into one
|
|
1475
|
+
* pool are serialised, so at most ONE load is outstanding per worker. The
|
|
1476
|
+
* worker runs model commands in order and starts a load's soft bound only
|
|
1477
|
+
* when it begins, while the host's deadline for it runs from the SEND. A
|
|
1478
|
+
* load queued behind another could therefore see its host deadline fire
|
|
1479
|
+
* before the worker's named answer. Nothing in the provider issues that; if
|
|
1480
|
+
* anything ever does, this says so.
|
|
1481
|
+
*/
|
|
1482
|
+
warnIfLoadQueued(next) {
|
|
1483
|
+
const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
|
|
1484
|
+
if (ahead.length === 0) return;
|
|
1485
|
+
this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
|
|
1486
|
+
worker: this.opts.workerLabel,
|
|
1487
|
+
pid: this.getPid(),
|
|
1488
|
+
runtime: this.opts.poolRuntime,
|
|
1489
|
+
device: this.opts.device ?? "default",
|
|
1490
|
+
model: next.model,
|
|
1491
|
+
queuedBehind: ahead.map((c) => c.model)
|
|
1492
|
+
} });
|
|
1493
|
+
}
|
|
1494
|
+
hasPendingLoad() {
|
|
1495
|
+
for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
|
|
1496
|
+
return false;
|
|
1497
|
+
}
|
|
1498
|
+
/**
|
|
1499
|
+
* A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
|
|
1500
|
+
* or that an earlier one still has (`worker-poisoned`). The worker is not
|
|
1501
|
+
* recycled for it (fix round 1): the compile keeps running so a slow but
|
|
1502
|
+
* finite one writes its cache, the loaded models keep serving, and the worker
|
|
1503
|
+
* starts no new load. Only a compile still running at the HARD bound costs
|
|
1504
|
+
* the process.
|
|
1505
|
+
*/
|
|
1506
|
+
inspectCommandReply(raw, command, sentAt) {
|
|
1507
|
+
const reason = raw["reason"];
|
|
1508
|
+
if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
|
|
1509
|
+
if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
|
|
1510
|
+
const model = raw["modelId"];
|
|
1511
|
+
const abandoned = {
|
|
1512
|
+
...command,
|
|
1513
|
+
model: typeof model === "string" ? model : command.model
|
|
1514
|
+
};
|
|
1515
|
+
const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
|
|
1516
|
+
const reported = raw["compileElapsedMs"];
|
|
1517
|
+
const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
|
|
1518
|
+
const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
|
|
1519
|
+
hardTimer.unref?.();
|
|
1520
|
+
this.abandonedCompile = {
|
|
1521
|
+
command: abandoned,
|
|
1522
|
+
since: Date.now(),
|
|
1523
|
+
hardTimer
|
|
1524
|
+
};
|
|
1525
|
+
this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
|
|
1526
|
+
worker: this.opts.workerLabel,
|
|
1527
|
+
pid: this.getPid(),
|
|
1528
|
+
runtime: this.opts.poolRuntime,
|
|
1529
|
+
device: this.opts.device ?? "default",
|
|
1530
|
+
model: abandoned.model,
|
|
1531
|
+
modelIndex: abandoned.modelIndex,
|
|
1532
|
+
softMs: bounds.softMs,
|
|
1533
|
+
hardMs: bounds.hardMs,
|
|
1534
|
+
loads: "refused until it returns"
|
|
1535
|
+
} });
|
|
1536
|
+
}
|
|
1537
|
+
/** An unsolicited worker event (`compile-finished-late`). */
|
|
1538
|
+
handleWorkerEvent(jsonBytes) {
|
|
1539
|
+
let event;
|
|
1540
|
+
try {
|
|
1541
|
+
event = JSON.parse(jsonBytes.toString("utf8"));
|
|
1542
|
+
} catch {
|
|
1543
|
+
return;
|
|
1544
|
+
}
|
|
1545
|
+
if (event["event"] !== "compile-finished-late") return;
|
|
1546
|
+
const abandoned = this.abandonedCompile;
|
|
1547
|
+
if (abandoned !== null) clearTimeout(abandoned.hardTimer);
|
|
1548
|
+
this.abandonedCompile = null;
|
|
1549
|
+
this.inferStall.noteResult(Date.now());
|
|
1550
|
+
const ok = event["ok"] === true;
|
|
1551
|
+
const meta = {
|
|
1552
|
+
worker: this.opts.workerLabel,
|
|
1553
|
+
pid: this.getPid(),
|
|
1554
|
+
runtime: this.opts.poolRuntime,
|
|
1555
|
+
device: this.opts.device ?? "default",
|
|
1556
|
+
model: typeof event["modelId"] === "string" ? event["modelId"] : null,
|
|
1557
|
+
elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
|
|
1558
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1559
|
+
};
|
|
1560
|
+
if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
|
|
1561
|
+
else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
|
|
1562
|
+
this.opts.onCompileFinishedLate?.({
|
|
1563
|
+
model: meta.model,
|
|
1564
|
+
ok,
|
|
1565
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1566
|
+
});
|
|
1567
|
+
}
|
|
1568
|
+
/**
|
|
1569
|
+
* The GIL grace (D653 round 2): may a silent loop be a compile holding the
|
|
1570
|
+
* GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
|
|
1571
|
+
* on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
|
|
1572
|
+
*
|
|
1573
|
+
* NOT while a compile runs past its soft bound: the worker REPLIED
|
|
1574
|
+
* `compile-timeout`, so its loop is proven alive, and an inference hang
|
|
1575
|
+
* behind that compile — the 2026-09-26 incident exactly — must be seen on
|
|
1576
|
+
* the normal rule, not ~600-900 s later at the hard bound.
|
|
1577
|
+
*/
|
|
1578
|
+
gilGraceActive() {
|
|
1579
|
+
if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
|
|
1580
|
+
if (this.loadDeadlineMissed) return false;
|
|
1581
|
+
return this.hasPendingLoad();
|
|
1582
|
+
}
|
|
1583
|
+
/**
|
|
1584
|
+
* The reason a silent worker is recycled with: a freeze that began with a
|
|
1585
|
+
* load missing its deadline is that compile's doing (`compile-hung`, charged
|
|
1586
|
+
* to the model, not the device); anything else is the device's.
|
|
1587
|
+
*/
|
|
1588
|
+
silenceReason(fallback) {
|
|
1589
|
+
return this.loadDeadlineMissed ? "compile-hung" : fallback;
|
|
1590
|
+
}
|
|
1591
|
+
/** A live request ended with no result: judge the executor (D653, item 4). */
|
|
1592
|
+
noteInferUnanswered() {
|
|
1593
|
+
const stall = this.inferStall.noteUnanswered(Date.now());
|
|
1594
|
+
if (stall === null || this.gilGraceActive()) return;
|
|
1595
|
+
this.poison(this.silenceReason("infer-unresponsive"), {
|
|
1596
|
+
cmd: "infer",
|
|
1597
|
+
modelIndex: null,
|
|
1598
|
+
model: null
|
|
1599
|
+
}, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
|
|
1600
|
+
}
|
|
1601
|
+
/** A reply whose request was already abandoned: a real result still proves the executor runs. */
|
|
1602
|
+
noteLateReply(jsonBytes) {
|
|
1603
|
+
try {
|
|
1604
|
+
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1605
|
+
if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
|
|
1606
|
+
} catch {}
|
|
1607
|
+
}
|
|
1608
|
+
/**
|
|
1609
|
+
* Ask a worker whose command just timed out whether its loop still answers.
|
|
1610
|
+
* `mem_stats` is served on the loop, never behind a compile, so a worker
|
|
1611
|
+
* that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
|
|
1612
|
+
* slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
|
|
1613
|
+
* `mem_stats` lost while the process looked alive.
|
|
1614
|
+
*/
|
|
1615
|
+
probeLiveness(timedOut) {
|
|
1616
|
+
if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
|
|
1617
|
+
this.probeInFlight = true;
|
|
1618
|
+
this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
|
|
1619
|
+
command: {
|
|
1620
|
+
cmd: "mem_stats",
|
|
1621
|
+
modelIndex: null,
|
|
1622
|
+
model: null
|
|
1623
|
+
},
|
|
1624
|
+
timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
|
|
1625
|
+
probe: true
|
|
1626
|
+
}).then(() => {
|
|
1627
|
+
this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
|
|
1628
|
+
worker: this.opts.workerLabel,
|
|
1629
|
+
pid: this.getPid(),
|
|
1630
|
+
runtime: this.opts.poolRuntime,
|
|
1631
|
+
device: this.opts.device ?? "default",
|
|
1632
|
+
command: timedOut.cmd,
|
|
1633
|
+
modelIndex: timedOut.modelIndex,
|
|
1634
|
+
model: timedOut.model
|
|
1635
|
+
} });
|
|
1636
|
+
}, (err) => {
|
|
1637
|
+
if (this.gilGraceActive()) {
|
|
1638
|
+
this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
|
|
1639
|
+
worker: this.opts.workerLabel,
|
|
1640
|
+
pid: this.getPid(),
|
|
1641
|
+
runtime: this.opts.poolRuntime,
|
|
1642
|
+
device: this.opts.device ?? "default",
|
|
1643
|
+
command: timedOut.cmd,
|
|
1644
|
+
model: timedOut.model
|
|
1645
|
+
} });
|
|
1646
|
+
return;
|
|
1647
|
+
}
|
|
1648
|
+
this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1649
|
+
}).finally(() => {
|
|
1650
|
+
this.probeInFlight = false;
|
|
1651
|
+
});
|
|
1652
|
+
}
|
|
1653
|
+
/**
|
|
1654
|
+
* Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
|
|
1655
|
+
* and recycle it once it has drained.
|
|
1656
|
+
*
|
|
1657
|
+
* `ready = false` is what hands the pool back to the provider: its next
|
|
1658
|
+
* dispatch finds the factory not ready and condemns it under the per-device
|
|
1659
|
+
* restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
|
|
1660
|
+
* excludes) — the same path a crashed worker takes, except that a
|
|
1661
|
+
* `compile-hung` death is not charged to the device (see PoolPoisonReason).
|
|
1662
|
+
* An unresponsive worker is killed at once (it will answer nothing it
|
|
1663
|
+
* holds); a compile-hung one keeps serving its loaded models until its
|
|
1664
|
+
* in-flight requests are answered, bounded by the live deadline.
|
|
1665
|
+
*/
|
|
1666
|
+
poison(reason, command, detail) {
|
|
1667
|
+
if (this.poisonVerdict !== null || this.exited || this.process === null) return;
|
|
1668
|
+
this.poisonVerdict = {
|
|
1669
|
+
reason,
|
|
1670
|
+
command,
|
|
1671
|
+
detail,
|
|
1672
|
+
pid: this.getPid()
|
|
1673
|
+
};
|
|
1674
|
+
this.ready = false;
|
|
1675
|
+
const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
|
|
1676
|
+
this.log.error("pool worker POISONED — recycling it", { meta: {
|
|
1677
|
+
worker: this.opts.workerLabel,
|
|
1678
|
+
pid: this.getPid(),
|
|
1679
|
+
runtime: this.opts.poolRuntime,
|
|
1680
|
+
device: this.opts.device ?? "default",
|
|
1681
|
+
reason,
|
|
1682
|
+
command: command.cmd,
|
|
1683
|
+
modelIndex: command.modelIndex,
|
|
1684
|
+
model: command.model,
|
|
1685
|
+
detail,
|
|
1686
|
+
inFlight: this.pending.size,
|
|
1687
|
+
inFlightCommands
|
|
1688
|
+
} });
|
|
1689
|
+
if (reason !== "compile-hung" || this.pending.size === 0) {
|
|
1690
|
+
this.recycle();
|
|
1691
|
+
return;
|
|
1692
|
+
}
|
|
1693
|
+
this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
|
|
1694
|
+
this.recycleBackstop.unref?.();
|
|
1695
|
+
}
|
|
1696
|
+
/**
|
|
1697
|
+
* Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
|
|
1698
|
+
* reported and rejects whatever it still held — a deliberate dispose would
|
|
1699
|
+
* silence both, and this is not one.
|
|
1700
|
+
*/
|
|
1701
|
+
recycle() {
|
|
1702
|
+
if (this.recycling || this.exited || this.process === null) return;
|
|
1703
|
+
this.recycling = true;
|
|
1704
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1705
|
+
terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
|
|
1247
1706
|
}
|
|
1248
1707
|
rejectAll(err) {
|
|
1249
1708
|
const entries = [...this.pending.values()];
|
|
@@ -1290,8 +1749,10 @@ var SharedInferencePool = class {
|
|
|
1290
1749
|
this.tuning = options.tuning ?? null;
|
|
1291
1750
|
this.numWorkers = Math.max(1, options.numWorkers ?? 1);
|
|
1292
1751
|
this.device = options.device;
|
|
1752
|
+
this.onCompileFinishedLate = options.onCompileFinishedLate;
|
|
1293
1753
|
}
|
|
1294
1754
|
device;
|
|
1755
|
+
onCompileFinishedLate;
|
|
1295
1756
|
/** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
|
|
1296
1757
|
/** Summed backlog and shed count across the pool's workers. A rising
|
|
1297
1758
|
* `inFlight` with a rising `shed` is a worker falling behind; a rising
|
|
@@ -1330,7 +1791,8 @@ var SharedInferencePool = class {
|
|
|
1330
1791
|
tuning: this.tuning,
|
|
1331
1792
|
logger: this.log,
|
|
1332
1793
|
workerLabel: `w${i}`,
|
|
1333
|
-
...this.device ? { device: this.device } : {}
|
|
1794
|
+
...this.device ? { device: this.device } : {},
|
|
1795
|
+
...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
|
|
1334
1796
|
}));
|
|
1335
1797
|
const t0 = performance.now();
|
|
1336
1798
|
const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
|
|
@@ -1425,7 +1887,7 @@ var SharedInferencePool = class {
|
|
|
1425
1887
|
index,
|
|
1426
1888
|
config: serializeModelConfig(config)
|
|
1427
1889
|
})));
|
|
1428
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
1890
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1429
1891
|
if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
|
|
1430
1892
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1431
1893
|
}
|
|
@@ -1461,7 +1923,7 @@ var SharedInferencePool = class {
|
|
|
1461
1923
|
index,
|
|
1462
1924
|
config: serializeModelConfig(config)
|
|
1463
1925
|
})));
|
|
1464
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
1926
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1465
1927
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1466
1928
|
}
|
|
1467
1929
|
/**
|
|
@@ -1511,9 +1973,35 @@ var SharedInferencePool = class {
|
|
|
1511
1973
|
allocateIndex() {
|
|
1512
1974
|
return this.nextFreeIndex++;
|
|
1513
1975
|
}
|
|
1976
|
+
/**
|
|
1977
|
+
* Give back an index whose load FAILED, so the retry reuses it. Only the most
|
|
1978
|
+
* recent allocation can be returned. Loads into one pool are serialised by
|
|
1979
|
+
* the provider, so a failed load is normally the latest one; anything else
|
|
1980
|
+
* is left allocated rather than risk handing out a live slot twice.
|
|
1981
|
+
*/
|
|
1982
|
+
releaseIndex(index) {
|
|
1983
|
+
if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
|
|
1984
|
+
}
|
|
1514
1985
|
isReady() {
|
|
1515
1986
|
return this.workers.length > 0 && this.workers.every((w) => w.isReady());
|
|
1516
1987
|
}
|
|
1988
|
+
/**
|
|
1989
|
+
* Why a worker of this pool was declared unusable while alive (D653), or
|
|
1990
|
+
* `null`. The provider charges the restart budget with THIS — or, for a
|
|
1991
|
+
* `compile-hung` death, does not charge the device at all — so the
|
|
1992
|
+
* `inference device FAILED` line names the cause instead of "pool worker is
|
|
1993
|
+
* not ready".
|
|
1994
|
+
*/
|
|
1995
|
+
getDeathCause() {
|
|
1996
|
+
for (const w of this.workers) {
|
|
1997
|
+
const cause = w.getDeathCause();
|
|
1998
|
+
if (cause !== null) return {
|
|
1999
|
+
...cause,
|
|
2000
|
+
message: `worker ${cause.message}`
|
|
2001
|
+
};
|
|
2002
|
+
}
|
|
2003
|
+
return null;
|
|
2004
|
+
}
|
|
1517
2005
|
async dispose() {
|
|
1518
2006
|
await Promise.all(this.workers.map((w) => w.dispose()));
|
|
1519
2007
|
this.workers.length = 0;
|
|
@@ -1578,6 +2066,23 @@ var SharedInferencePool = class {
|
|
|
1578
2066
|
return found;
|
|
1579
2067
|
}
|
|
1580
2068
|
};
|
|
2069
|
+
/** A failed command's reply as one message: its reason class first, when it has one. */
|
|
2070
|
+
function describeCommandFailure(resp) {
|
|
2071
|
+
const error = resp.error ?? "unknown";
|
|
2072
|
+
return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
|
|
2073
|
+
}
|
|
2074
|
+
/** Name a command for the lines that must say which one hung (D653). */
|
|
2075
|
+
function describeCommand(cmd) {
|
|
2076
|
+
const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
|
|
2077
|
+
const index = cmd["index"];
|
|
2078
|
+
const config = cmd["config"];
|
|
2079
|
+
const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
|
|
2080
|
+
return {
|
|
2081
|
+
cmd: name,
|
|
2082
|
+
modelIndex: typeof index === "number" ? index : null,
|
|
2083
|
+
model: typeof modelPath === "string" && modelPath.length > 0 ? node_path.basename(modelPath, node_path.extname(modelPath)) : null
|
|
2084
|
+
};
|
|
2085
|
+
}
|
|
1581
2086
|
function serializeModelConfig(config) {
|
|
1582
2087
|
const result = {
|
|
1583
2088
|
path: config.path,
|
|
@@ -1703,12 +2208,13 @@ function resolveEffectivePostprocessorForEntry(definition, entry) {
|
|
|
1703
2208
|
//#region src/detection-pipeline/registry/model-knob-applicability.ts
|
|
1704
2209
|
/** Decodes that run NMS in the pipeline (and hence consult `nmsIouThreshold`).
|
|
1705
2210
|
* Mirrors the Python postprocessors that read the knob — yolo.py,
|
|
1706
|
-
* yolo_seg.py, rfdetr.py, scrfd.py. */
|
|
2211
|
+
* yolo_seg.py, rfdetr.py, scrfd.py, yunet.py. */
|
|
1707
2212
|
var NMS_CONSUMING_POSTPROCESSORS = new Set([
|
|
1708
2213
|
"yolo",
|
|
1709
2214
|
"yolo-seg",
|
|
1710
2215
|
"rfdetr",
|
|
1711
|
-
"scrfd"
|
|
2216
|
+
"scrfd",
|
|
2217
|
+
"yunet"
|
|
1712
2218
|
]);
|
|
1713
2219
|
/**
|
|
1714
2220
|
* Does this decode consult `nmsIouThreshold`?
|
|
@@ -1864,6 +2370,28 @@ function poolDecodeSettingsEqual(a, b) {
|
|
|
1864
2370
|
}
|
|
1865
2371
|
//#endregion
|
|
1866
2372
|
//#region src/detection-pipeline/engine/pipeline-model-manager.ts
|
|
2373
|
+
/**
|
|
2374
|
+
* A (step, model) load the pool rejected — carrying the pool's failure class
|
|
2375
|
+
* (`compile-timeout`, …) so the provider can count a timeout against THAT
|
|
2376
|
+
* model without parsing a message (D653).
|
|
2377
|
+
*/
|
|
2378
|
+
var StepVariantLoadError = class extends Error {
|
|
2379
|
+
stepId;
|
|
2380
|
+
modelId;
|
|
2381
|
+
poolIndex;
|
|
2382
|
+
reason;
|
|
2383
|
+
/** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
|
|
2384
|
+
poolModel;
|
|
2385
|
+
constructor(stepId, modelId, poolIndex, cause) {
|
|
2386
|
+
super(cause instanceof Error ? cause.message : String(cause));
|
|
2387
|
+
this.name = "StepVariantLoadError";
|
|
2388
|
+
this.stepId = stepId;
|
|
2389
|
+
this.modelId = modelId;
|
|
2390
|
+
this.poolIndex = poolIndex;
|
|
2391
|
+
this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
|
|
2392
|
+
this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
|
|
2393
|
+
}
|
|
2394
|
+
};
|
|
1867
2395
|
var PipelineModelManager = class {
|
|
1868
2396
|
pool;
|
|
1869
2397
|
source;
|
|
@@ -1875,11 +2403,20 @@ var PipelineModelManager = class {
|
|
|
1875
2403
|
lruClock = 0;
|
|
1876
2404
|
log;
|
|
1877
2405
|
maxModelsPerStep;
|
|
2406
|
+
deviceKey;
|
|
2407
|
+
/**
|
|
2408
|
+
* Loads issued and not yet answered, keyed `stepId::modelId`. A second
|
|
2409
|
+
* caller for the same pair JOINS the pending load instead of issuing its own
|
|
2410
|
+
* (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
|
|
2411
|
+
* compile, each allocating a pool index of its own.
|
|
2412
|
+
*/
|
|
2413
|
+
pendingLoads = /* @__PURE__ */ new Map();
|
|
1878
2414
|
constructor(pool, source, logger, options) {
|
|
1879
2415
|
this.pool = pool;
|
|
1880
2416
|
this.source = source;
|
|
1881
2417
|
this.log = logger;
|
|
1882
2418
|
this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
|
|
2419
|
+
this.deviceKey = options?.deviceKey ?? null;
|
|
1883
2420
|
}
|
|
1884
2421
|
/**
|
|
1885
2422
|
* Apply a new pipeline configuration — driven by the runtime config
|
|
@@ -2029,6 +2566,23 @@ var PipelineModelManager = class {
|
|
|
2029
2566
|
await this.reconcileDecode(existing, settings);
|
|
2030
2567
|
return existing;
|
|
2031
2568
|
}
|
|
2569
|
+
const key = `${stepId}::${modelId}`;
|
|
2570
|
+
const pending = this.pendingLoads.get(key);
|
|
2571
|
+
if (pending) {
|
|
2572
|
+
const joined = await pending;
|
|
2573
|
+
await this.reconcileDecode(joined, settings);
|
|
2574
|
+
return joined;
|
|
2575
|
+
}
|
|
2576
|
+
const load = this.issueLoad(perStep, stepId, modelId, settings);
|
|
2577
|
+
this.pendingLoads.set(key, load);
|
|
2578
|
+
try {
|
|
2579
|
+
return await load;
|
|
2580
|
+
} finally {
|
|
2581
|
+
this.pendingLoads.delete(key);
|
|
2582
|
+
}
|
|
2583
|
+
}
|
|
2584
|
+
/** Evict if at capacity, then send ONE load command and record the result. */
|
|
2585
|
+
async issueLoad(perStep, stepId, modelId, settings) {
|
|
2032
2586
|
while (perStep.size >= this.maxModelsPerStep) {
|
|
2033
2587
|
const evicted = this.pickEvictionTarget(stepId);
|
|
2034
2588
|
if (!evicted) break;
|
|
@@ -2040,16 +2594,31 @@ var PipelineModelManager = class {
|
|
|
2040
2594
|
cap: this.maxModelsPerStep
|
|
2041
2595
|
} });
|
|
2042
2596
|
}
|
|
2043
|
-
const index = this.pool.allocateIndex();
|
|
2044
2597
|
const config = this.source.buildConfig(stepId, modelId, settings);
|
|
2045
2598
|
const decode = poolDecodeSettingsOf(config);
|
|
2599
|
+
const index = this.pool.allocateIndex();
|
|
2046
2600
|
this.log.info("Loading step variant", { meta: {
|
|
2047
2601
|
step: stepId,
|
|
2048
2602
|
modelId,
|
|
2049
2603
|
poolIndex: index,
|
|
2604
|
+
deviceKey: this.deviceKey,
|
|
2050
2605
|
...decode
|
|
2051
2606
|
} });
|
|
2052
|
-
|
|
2607
|
+
let loadMs;
|
|
2608
|
+
try {
|
|
2609
|
+
({loadMs} = await this.pool.loadModel(index, config));
|
|
2610
|
+
} catch (err) {
|
|
2611
|
+
this.pool.releaseIndex(index);
|
|
2612
|
+
this.log.error("Step variant load failed", { meta: {
|
|
2613
|
+
step: stepId,
|
|
2614
|
+
modelId,
|
|
2615
|
+
poolIndex: index,
|
|
2616
|
+
deviceKey: this.deviceKey,
|
|
2617
|
+
reason: err instanceof PoolModelLoadError ? err.reason : null,
|
|
2618
|
+
error: err instanceof Error ? err.message : String(err)
|
|
2619
|
+
} });
|
|
2620
|
+
throw new StepVariantLoadError(stepId, modelId, index, err);
|
|
2621
|
+
}
|
|
2053
2622
|
this.log.info("Step variant loaded", { meta: {
|
|
2054
2623
|
step: stepId,
|
|
2055
2624
|
modelId,
|
|
@@ -2419,6 +2988,15 @@ var EngineFactory = class {
|
|
|
2419
2988
|
isReady() {
|
|
2420
2989
|
return this.pool?.isReady() ?? false;
|
|
2421
2990
|
}
|
|
2991
|
+
/**
|
|
2992
|
+
* Why this factory's pool stopped being usable while its process was alive —
|
|
2993
|
+
* a compile that never returned, or a worker that answered nothing (D653) —
|
|
2994
|
+
* or `null`. A pool that merely crashed says nothing here; its `Worker
|
|
2995
|
+
* process exited` line already names the signal.
|
|
2996
|
+
*/
|
|
2997
|
+
getDeathCause() {
|
|
2998
|
+
return this.pool?.getDeathCause() ?? null;
|
|
2999
|
+
}
|
|
2422
3000
|
/** Native pid of the underlying Python pool, if any. */
|
|
2423
3001
|
getPoolPid() {
|
|
2424
3002
|
return this.pool?.getPid() ?? null;
|
|
@@ -2545,12 +3123,13 @@ var EngineFactory = class {
|
|
|
2545
3123
|
concurrency,
|
|
2546
3124
|
tuning: resolvedTuning,
|
|
2547
3125
|
numWorkers,
|
|
2548
|
-
...this.opts.engine.device ? { device: this.opts.engine.device } : {}
|
|
3126
|
+
...this.opts.engine.device ? { device: this.opts.engine.device } : {},
|
|
3127
|
+
...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
|
|
2549
3128
|
});
|
|
2550
3129
|
this.poolManager = new PipelineModelManager(this.pool, {
|
|
2551
3130
|
buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
|
|
2552
3131
|
resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
|
|
2553
|
-
}, this.log.child("model-mgr"));
|
|
3132
|
+
}, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
|
|
2554
3133
|
await this.pool.initialize([]);
|
|
2555
3134
|
await this.poolManager.applyConfig(steps);
|
|
2556
3135
|
}
|
|
@@ -2666,6 +3245,175 @@ function buildPoolModelConfigForStep(inputs) {
|
|
|
2666
3245
|
};
|
|
2667
3246
|
}
|
|
2668
3247
|
//#endregion
|
|
3248
|
+
//#region src/detection-pipeline/engine/model-load-governor.ts
|
|
3249
|
+
/**
|
|
3250
|
+
* What the dispatch path may ask of a pool's model loads, and what it must say
|
|
3251
|
+
* when it may not (D653, fix round 1).
|
|
3252
|
+
*
|
|
3253
|
+
* Three pieces of state, consulted ON DEMAND by the dispatch that needs a
|
|
3254
|
+
* model — there is no timer anywhere in here:
|
|
3255
|
+
*
|
|
3256
|
+
* 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
|
|
3257
|
+
* failed load, so without it every frame of every camera re-issued the
|
|
3258
|
+
* failing load and wrote two ERRORs. After a failure the model is not
|
|
3259
|
+
* re-issued on that pool for an interval that doubles per consecutive
|
|
3260
|
+
* failure up to a cap; a success clears it.
|
|
3261
|
+
* 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
|
|
3262
|
+
* bound is not a crash of the device. The SAME model timing out
|
|
3263
|
+
* {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
|
|
3264
|
+
* since a slow compile now runs on and writes its cache, only a compile
|
|
3265
|
+
* that never finishes can do — refuses THAT MODEL there, by name. The
|
|
3266
|
+
* device keeps serving every other model.
|
|
3267
|
+
* 3. Which camera has already been TOLD about the current state of a model,
|
|
3268
|
+
* so each camera gets one line per state change, never one per frame.
|
|
3269
|
+
*/
|
|
3270
|
+
/** The first back-off after a failed load. */
|
|
3271
|
+
var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
|
|
3272
|
+
/** The back-off doubles per consecutive failure up to this. */
|
|
3273
|
+
var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
|
|
3274
|
+
/** Timeouts older than this no longer count toward a refusal. */
|
|
3275
|
+
var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
|
|
3276
|
+
var ModelLoadGovernor = class {
|
|
3277
|
+
/** Keyed by the pool OBJECT, so a respawned pool starts clean. */
|
|
3278
|
+
backoff = /* @__PURE__ */ new WeakMap();
|
|
3279
|
+
timeouts = /* @__PURE__ */ new Map();
|
|
3280
|
+
reported = /* @__PURE__ */ new Map();
|
|
3281
|
+
generation = 0;
|
|
3282
|
+
/** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
|
|
3283
|
+
check(pool, deviceKey, models, now = Date.now()) {
|
|
3284
|
+
for (const model of models) {
|
|
3285
|
+
const t = this.timeouts.get(timeoutKey(deviceKey, model));
|
|
3286
|
+
if (t?.refused === true) return {
|
|
3287
|
+
kind: "refused",
|
|
3288
|
+
model,
|
|
3289
|
+
timeouts: t.at.length,
|
|
3290
|
+
error: t.lastError
|
|
3291
|
+
};
|
|
3292
|
+
}
|
|
3293
|
+
const perPool = this.backoff.get(pool);
|
|
3294
|
+
if (perPool === void 0) return { kind: "go" };
|
|
3295
|
+
for (const model of models) {
|
|
3296
|
+
const b = perPool.get(model);
|
|
3297
|
+
if (b !== void 0 && now < b.retryAtMs) return {
|
|
3298
|
+
kind: "backoff",
|
|
3299
|
+
model,
|
|
3300
|
+
retryInMs: b.retryAtMs - now,
|
|
3301
|
+
failures: b.failures,
|
|
3302
|
+
error: b.error,
|
|
3303
|
+
generation: b.generation
|
|
3304
|
+
};
|
|
3305
|
+
}
|
|
3306
|
+
return { kind: "go" };
|
|
3307
|
+
}
|
|
3308
|
+
/** One load of `model` on `pool` failed. */
|
|
3309
|
+
recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
|
|
3310
|
+
let perPool = this.backoff.get(pool);
|
|
3311
|
+
if (perPool === void 0) {
|
|
3312
|
+
perPool = /* @__PURE__ */ new Map();
|
|
3313
|
+
this.backoff.set(pool, perPool);
|
|
3314
|
+
}
|
|
3315
|
+
const previousEntry = perPool.get(model);
|
|
3316
|
+
const failures = (previousEntry?.failures ?? 0) + 1;
|
|
3317
|
+
const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
|
|
3318
|
+
this.generation += 1;
|
|
3319
|
+
const generation = this.generation;
|
|
3320
|
+
perPool.set(model, {
|
|
3321
|
+
failures,
|
|
3322
|
+
retryAtMs: now + retryInMs,
|
|
3323
|
+
error,
|
|
3324
|
+
generation,
|
|
3325
|
+
poolModel: poolModel ?? previousEntry?.poolModel ?? null,
|
|
3326
|
+
reason
|
|
3327
|
+
});
|
|
3328
|
+
if (!compileTimeout) return {
|
|
3329
|
+
generation,
|
|
3330
|
+
retryInMs,
|
|
3331
|
+
refusedNow: false,
|
|
3332
|
+
compileTimeouts: 0
|
|
3333
|
+
};
|
|
3334
|
+
const key = timeoutKey(deviceKey, model);
|
|
3335
|
+
const previous = this.timeouts.get(key);
|
|
3336
|
+
const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
|
|
3337
|
+
const refused = at.length >= 2;
|
|
3338
|
+
this.timeouts.set(key, {
|
|
3339
|
+
at,
|
|
3340
|
+
refused,
|
|
3341
|
+
lastError: error
|
|
3342
|
+
});
|
|
3343
|
+
return {
|
|
3344
|
+
generation,
|
|
3345
|
+
retryInMs,
|
|
3346
|
+
refusedNow: refused && previous?.refused !== true,
|
|
3347
|
+
compileTimeouts: at.length
|
|
3348
|
+
};
|
|
3349
|
+
}
|
|
3350
|
+
/**
|
|
3351
|
+
* A compile that outlived its bound came back on `pool` (D653 round 3).
|
|
3352
|
+
*
|
|
3353
|
+
* `ok`: THAT model's cache is written and the worker loads again, so its
|
|
3354
|
+
* back-off is lifted — and only its: another model's failure on the same
|
|
3355
|
+
* pool is not the compile that just finished. Not ok: it is one more
|
|
3356
|
+
* failure of that model, and its back-off grows. Refusals by name are never
|
|
3357
|
+
* lifted here; a model refused for timing out twice stays refused until the
|
|
3358
|
+
* operator re-arms. Returns the governor keys it touched.
|
|
3359
|
+
*/
|
|
3360
|
+
settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
|
|
3361
|
+
const perPool = this.backoff.get(pool);
|
|
3362
|
+
if (perPool === void 0) return [];
|
|
3363
|
+
const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
|
|
3364
|
+
for (const key of refusedMeanwhile) perPool.delete(key);
|
|
3365
|
+
const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
|
|
3366
|
+
for (const key of own) if (ok) perPool.delete(key);
|
|
3367
|
+
else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
|
|
3368
|
+
return [...refusedMeanwhile, ...own];
|
|
3369
|
+
}
|
|
3370
|
+
/** `models` loaded on `pool`: forget their failures there. */
|
|
3371
|
+
recordSuccess(pool, deviceKey, models) {
|
|
3372
|
+
const perPool = this.backoff.get(pool);
|
|
3373
|
+
for (const model of models) {
|
|
3374
|
+
perPool?.delete(model);
|
|
3375
|
+
this.timeouts.delete(timeoutKey(deviceKey, model));
|
|
3376
|
+
}
|
|
3377
|
+
}
|
|
3378
|
+
/**
|
|
3379
|
+
* Has `camera` been told about this `state` of `model` on `deviceKey` yet?
|
|
3380
|
+
* Returns `true` exactly once per state change.
|
|
3381
|
+
*/
|
|
3382
|
+
shouldReport(camera, deviceKey, model, state) {
|
|
3383
|
+
const key = `${camera ?? "-"}|${deviceKey}|${model}`;
|
|
3384
|
+
if (this.reported.get(key) === state) return false;
|
|
3385
|
+
this.reported.set(key, state);
|
|
3386
|
+
return true;
|
|
3387
|
+
}
|
|
3388
|
+
/** Every model refused on some device. */
|
|
3389
|
+
refused() {
|
|
3390
|
+
const out = [];
|
|
3391
|
+
for (const [key, t] of this.timeouts) {
|
|
3392
|
+
if (!t.refused) continue;
|
|
3393
|
+
const [deviceKey = "", model = ""] = key.split("\0");
|
|
3394
|
+
out.push({
|
|
3395
|
+
deviceKey,
|
|
3396
|
+
model,
|
|
3397
|
+
timeouts: t.at.length,
|
|
3398
|
+
lastError: t.lastError
|
|
3399
|
+
});
|
|
3400
|
+
}
|
|
3401
|
+
return out;
|
|
3402
|
+
}
|
|
3403
|
+
/** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
|
|
3404
|
+
rearm(deviceKey) {
|
|
3405
|
+
let cleared = 0;
|
|
3406
|
+
for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
|
|
3407
|
+
this.timeouts.delete(key);
|
|
3408
|
+
cleared += 1;
|
|
3409
|
+
}
|
|
3410
|
+
return cleared;
|
|
3411
|
+
}
|
|
3412
|
+
};
|
|
3413
|
+
function timeoutKey(deviceKey, model) {
|
|
3414
|
+
return `${deviceKey}\u0000${model}`;
|
|
3415
|
+
}
|
|
3416
|
+
//#endregion
|
|
2669
3417
|
//#region src/detection-pipeline/engine/idle-pool-reaper.ts
|
|
2670
3418
|
var IdlePoolReaper = class {
|
|
2671
3419
|
lastUsed = /* @__PURE__ */ new Map();
|
|
@@ -3165,6 +3913,52 @@ function createInferenceTimeoutGuard(deps) {
|
|
|
3165
3913
|
};
|
|
3166
3914
|
}
|
|
3167
3915
|
//#endregion
|
|
3916
|
+
//#region src/detection-pipeline/model-pin.ts
|
|
3917
|
+
function requestedModels(steps) {
|
|
3918
|
+
const out = /* @__PURE__ */ new Map();
|
|
3919
|
+
const walk = (nodes) => {
|
|
3920
|
+
for (const step of nodes) {
|
|
3921
|
+
if (!step.enabled) continue;
|
|
3922
|
+
out.set(step.addonId, step.modelId);
|
|
3923
|
+
if (step.children !== void 0) walk(step.children);
|
|
3924
|
+
}
|
|
3925
|
+
};
|
|
3926
|
+
walk(steps);
|
|
3927
|
+
return out;
|
|
3928
|
+
}
|
|
3929
|
+
function checkReplayPin(input) {
|
|
3930
|
+
if (input.requestedDeviceKey !== void 0 && input.dispatchDeviceKey !== input.requestedDeviceKey) return {
|
|
3931
|
+
kind: "refused",
|
|
3932
|
+
reason: "device-not-servable",
|
|
3933
|
+
detail: `device pool ${input.requestedDeviceKey} cannot run this tree (the capability gate would fall back to ${input.dispatchDeviceKey ?? "the node default"})`
|
|
3934
|
+
};
|
|
3935
|
+
const asked = requestedModels(input.requested);
|
|
3936
|
+
for (const step of input.executing) {
|
|
3937
|
+
const want = asked.get(step.addonId);
|
|
3938
|
+
if (want === void 0 || want.length === 0) return {
|
|
3939
|
+
kind: "refused",
|
|
3940
|
+
reason: "model-not-servable",
|
|
3941
|
+
detail: `${step.addonId}: no model pinned by the caller (would run ${step.modelId}, ${input.format})`
|
|
3942
|
+
};
|
|
3943
|
+
if (want !== step.modelId) return {
|
|
3944
|
+
kind: "refused",
|
|
3945
|
+
reason: "model-not-servable",
|
|
3946
|
+
detail: `${step.addonId}: asked for ${want}, this pool would run ${step.modelId} (${input.format})`
|
|
3947
|
+
};
|
|
3948
|
+
}
|
|
3949
|
+
return { kind: "pinned" };
|
|
3950
|
+
}
|
|
3951
|
+
var PRIMARY_ROOT_STEP = "object-detection";
|
|
3952
|
+
/**
|
|
3953
|
+
* The root model a frame result reports: `object-detection` when it runs (it
|
|
3954
|
+
* is the detector the tracks come from), else the first enabled video root.
|
|
3955
|
+
* `undefined` when no root runs — never a fabricated id.
|
|
3956
|
+
*/
|
|
3957
|
+
function resolveRootModelId(roots) {
|
|
3958
|
+
const enabled = roots.filter((s) => s.enabled && s.slot !== "audio-classifier");
|
|
3959
|
+
return (enabled.find((s) => s.addonId === PRIMARY_ROOT_STEP) ?? enabled[0])?.modelId;
|
|
3960
|
+
}
|
|
3961
|
+
//#endregion
|
|
3168
3962
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
3169
3963
|
/** Incremental backoff growing to a ~5 min cap; retries indefinitely at cap. */
|
|
3170
3964
|
var BACKOFF_SCHEDULE_MS = [
|
|
@@ -3435,6 +4229,7 @@ var KNOWN_POSTPROCESSORS = new Set([
|
|
|
3435
4229
|
"rfdetr",
|
|
3436
4230
|
"yolo-seg",
|
|
3437
4231
|
"scrfd",
|
|
4232
|
+
"yunet",
|
|
3438
4233
|
"arcface",
|
|
3439
4234
|
"clip",
|
|
3440
4235
|
"softmax",
|
|
@@ -3536,9 +4331,12 @@ var DropSampler = class {
|
|
|
3536
4331
|
* accumulates and a summary is released once per interval — carrying the
|
|
3537
4332
|
* number suppressed, so nothing about the RATE is lost even though the lines
|
|
3538
4333
|
* are.
|
|
4334
|
+
*
|
|
4335
|
+
* `channel` separates occasions that share a device — a replay's drops are
|
|
4336
|
+
* sampled apart from the live ones.
|
|
3539
4337
|
*/
|
|
3540
|
-
record(deviceId, macroClass) {
|
|
3541
|
-
const key = `${deviceId}:${macroClass}`;
|
|
4338
|
+
record(deviceId, macroClass, channel = "live") {
|
|
4339
|
+
const key = `${channel}:${deviceId}:${macroClass}`;
|
|
3542
4340
|
const at = this.now();
|
|
3543
4341
|
const prev = this.seen.get(key);
|
|
3544
4342
|
if (prev === void 0) {
|
|
@@ -5324,6 +6122,11 @@ var PipelineExecutor = class {
|
|
|
5324
6122
|
async run(tree, rootInput, fullFrameJpegProvider, imageWidth, imageHeight, deviceId, runOpts, nativeCropProvider, cropZoneBbox, rootInputViewProvider) {
|
|
5325
6123
|
const startMs = Date.now();
|
|
5326
6124
|
const verbosity = runOpts?.traceVerbosity ?? "off";
|
|
6125
|
+
const occasion = runOpts?.occasion ?? "live";
|
|
6126
|
+
const logTags = occasion === "replay" ? {
|
|
6127
|
+
deviceId,
|
|
6128
|
+
occasion
|
|
6129
|
+
} : { deviceId };
|
|
5327
6130
|
const traceBuilder = new ExecutionTraceBuilder(verbosity, deviceId, imageWidth, imageHeight, this.opts.engineRuntime);
|
|
5328
6131
|
const firstLevel = [];
|
|
5329
6132
|
const details = [];
|
|
@@ -5339,9 +6142,9 @@ var PipelineExecutor = class {
|
|
|
5339
6142
|
idGen,
|
|
5340
6143
|
details,
|
|
5341
6144
|
onClassifierClassDropped: (drop) => {
|
|
5342
|
-
const sample = this.classFilterDropSampler.record(deviceId, `${drop.stepId}:${drop.droppedClass}
|
|
6145
|
+
const sample = this.classFilterDropSampler.record(deviceId, `${drop.stepId}:${drop.droppedClass}`, occasion);
|
|
5343
6146
|
if (sample.logExample) this.opts.logger?.info("classifier winner excluded by class filter — label dropped", {
|
|
5344
|
-
tags:
|
|
6147
|
+
tags: logTags,
|
|
5345
6148
|
meta: {
|
|
5346
6149
|
step: drop.stepId,
|
|
5347
6150
|
model: drop.modelId,
|
|
@@ -5352,7 +6155,7 @@ var PipelineExecutor = class {
|
|
|
5352
6155
|
}
|
|
5353
6156
|
});
|
|
5354
6157
|
if (sample.summaryCount !== null) this.opts.logger?.info("classifier class filter: still dropping", {
|
|
5355
|
-
tags:
|
|
6158
|
+
tags: logTags,
|
|
5356
6159
|
meta: {
|
|
5357
6160
|
step: drop.stepId,
|
|
5358
6161
|
droppedClass: drop.droppedClass,
|
|
@@ -5385,7 +6188,7 @@ var PipelineExecutor = class {
|
|
|
5385
6188
|
await this.executeChildren(rootStep.children, mutable, rootView?.jpegProvider ?? fullFrameJpegProvider, imageWidth, imageHeight, traceBuilder, stepTimings, ctx, poolAgg, runOpts?.plane, deviceId, nativeCropProvider);
|
|
5386
6189
|
} catch (err) {
|
|
5387
6190
|
this.opts.logger?.warn("Pipeline child execution failed — keeping parent detection", {
|
|
5388
|
-
tags:
|
|
6191
|
+
tags: logTags,
|
|
5389
6192
|
meta: {
|
|
5390
6193
|
rootStepId: rootStep.stepId,
|
|
5391
6194
|
parentClass: mutable.macroClass,
|
|
@@ -5439,9 +6242,9 @@ var PipelineExecutor = class {
|
|
|
5439
6242
|
const fullFrameBar = rootStep.settings?.["fullFrameGuardMaxConfidence"];
|
|
5440
6243
|
const fullFrameHardArea = rootStep.settings?.["fullFrameGuardHardAreaRatio"];
|
|
5441
6244
|
if (isFullFramePhantomDetection(mutable.bbox, imageWidth, imageHeight, mutable.score, typeof fullFrameBar === "number" ? fullFrameBar : void 0, typeof fullFrameHardArea === "number" ? fullFrameHardArea : void 0)) {
|
|
5442
|
-
const sample = this.dropSampler.record(deviceId, String(mutable.macroClass));
|
|
6245
|
+
const sample = this.dropSampler.record(deviceId, String(mutable.macroClass), occasion);
|
|
5443
6246
|
if (sample.summaryCount !== null) this.opts.logger?.info("full-frame phantom guard: still dropping", {
|
|
5444
|
-
tags:
|
|
6247
|
+
tags: logTags,
|
|
5445
6248
|
meta: {
|
|
5446
6249
|
macroClass: mutable.macroClass,
|
|
5447
6250
|
suppressed: sample.summaryCount,
|
|
@@ -5450,7 +6253,7 @@ var PipelineExecutor = class {
|
|
|
5450
6253
|
}
|
|
5451
6254
|
});
|
|
5452
6255
|
if (sample.logExample) this.opts.logger?.info("full-frame phantom guard: detection dropped", {
|
|
5453
|
-
tags:
|
|
6256
|
+
tags: logTags,
|
|
5454
6257
|
meta: {
|
|
5455
6258
|
macroClass: mutable.macroClass,
|
|
5456
6259
|
score: mutable.score,
|
|
@@ -5476,7 +6279,7 @@ var PipelineExecutor = class {
|
|
|
5476
6279
|
continue;
|
|
5477
6280
|
}
|
|
5478
6281
|
if (isFullFrameSurvivor(mutable.bbox, imageWidth, imageHeight, mutable.score, typeof fullFrameBar === "number" ? fullFrameBar : void 0)) this.opts.logger?.info("full-frame box SURVIVED the guard on score", {
|
|
5479
|
-
tags:
|
|
6282
|
+
tags: logTags,
|
|
5480
6283
|
meta: {
|
|
5481
6284
|
macroClass: mutable.macroClass,
|
|
5482
6285
|
score: mutable.score,
|
|
@@ -5490,7 +6293,7 @@ var PipelineExecutor = class {
|
|
|
5490
6293
|
await this.executeChildren(rootStep.children, mutable, rootView?.jpegProvider ?? fullFrameJpegProvider, imageWidth, imageHeight, traceBuilder, stepTimings, ctx, poolAgg, runOpts?.plane, deviceId, nativeCropProvider);
|
|
5491
6294
|
} catch (err) {
|
|
5492
6295
|
this.opts.logger?.warn("Pipeline child execution failed — keeping parent detection", {
|
|
5493
|
-
tags:
|
|
6296
|
+
tags: logTags,
|
|
5494
6297
|
meta: {
|
|
5495
6298
|
rootStepId: rootStep.stepId,
|
|
5496
6299
|
parentClass: mutable.macroClass,
|
|
@@ -6507,6 +7310,34 @@ function parseWavToAudioChunk(filePath) {
|
|
|
6507
7310
|
function enginesEqual(a, b) {
|
|
6508
7311
|
return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
|
|
6509
7312
|
}
|
|
7313
|
+
/** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
|
|
7314
|
+
var ModelLoadRefusedError = class extends Error {
|
|
7315
|
+
model;
|
|
7316
|
+
state;
|
|
7317
|
+
constructor(message, model, state) {
|
|
7318
|
+
super(message);
|
|
7319
|
+
this.name = "ModelLoadRefusedError";
|
|
7320
|
+
this.model = model;
|
|
7321
|
+
this.state = state;
|
|
7322
|
+
}
|
|
7323
|
+
};
|
|
7324
|
+
/** `step/model` — how the governor and the log lines name a model. */
|
|
7325
|
+
function stepModelKey(step) {
|
|
7326
|
+
return `${step.addonId}/${step.modelId}`;
|
|
7327
|
+
}
|
|
7328
|
+
/** The reporting state a refusal verdict stands for. */
|
|
7329
|
+
function stateOf(verdict) {
|
|
7330
|
+
return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
|
|
7331
|
+
}
|
|
7332
|
+
/**
|
|
7333
|
+
* The reason a condemned pool is charged against its device's restart budget.
|
|
7334
|
+
* A pool recycled because it HUNG (D653) names the hang — `inference device
|
|
7335
|
+
* FAILED … reason: pool worker is not ready` could not tell a GPU compile that
|
|
7336
|
+
* never returned from a crash.
|
|
7337
|
+
*/
|
|
7338
|
+
function deathReasonOf(factory) {
|
|
7339
|
+
return factory.getDeathCause()?.message ?? "pool worker is not ready";
|
|
7340
|
+
}
|
|
6510
7341
|
/** Build a `RuntimeEnv` from the running process + probed hardware. */
|
|
6511
7342
|
function runtimeEnvFromProcess(hardware) {
|
|
6512
7343
|
return {
|
|
@@ -6548,25 +7379,6 @@ var PROBE_READY_TIMEOUT_MS = 12e4;
|
|
|
6548
7379
|
* race bound keeps a wedged transport from wedging `setApi`.
|
|
6549
7380
|
*/
|
|
6550
7381
|
var PROBE_DIRECT_RPC_TIMEOUT_MS = 6e4;
|
|
6551
|
-
/**
|
|
6552
|
-
* Build the onnx-cpu floor pick using `pickBestRuntime` with a null hardware
|
|
6553
|
-
* env. Used wherever the old `detectBestEngine()` sync probe fell back — the
|
|
6554
|
-
* result is identical (onnx / cpu) but is now derived through the shared rules
|
|
6555
|
-
* instead of duplicated inline.
|
|
6556
|
-
*/
|
|
6557
|
-
/**
|
|
6558
|
-
* Is `backend` even POSSIBLE on this node's OS/arch (ignoring gpu detail)?
|
|
6559
|
-
* coreml ⇒ darwin only; openvino ⇒ x64 non-darwin only; onnx ⇒ anywhere. Used to
|
|
6560
|
-
* reject a persisted/global engine choice that the node's PLATFORM fundamentally
|
|
6561
|
-
* cannot run (e.g. the cluster's OpenVINO default landing on a Mac) — distinct
|
|
6562
|
-
* from the gpu-dependent support (linux without a probed Intel iGPU still keeps
|
|
6563
|
-
* openvino as a valid platform choice; the device falls back to cpu).
|
|
6564
|
-
*/
|
|
6565
|
-
function backendPossibleOnPlatform(backend) {
|
|
6566
|
-
if (backend === "coreml") return process.platform === "darwin";
|
|
6567
|
-
if (backend === "openvino") return process.arch === "x64" && process.platform !== "darwin";
|
|
6568
|
-
return true;
|
|
6569
|
-
}
|
|
6570
7382
|
var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
6571
7383
|
modelsDir;
|
|
6572
7384
|
eventBus;
|
|
@@ -6706,22 +7518,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6706
7518
|
*/
|
|
6707
7519
|
needsAutoPick = false;
|
|
6708
7520
|
/**
|
|
6709
|
-
* Warm cache for benchmark engine-override runs.
|
|
6710
|
-
*
|
|
6711
|
-
* Each override rebuild costs a full Python pool spin-up (~300-500ms)
|
|
6712
|
-
* plus the per-model load. Benchmark tabs iterate up to 800× against
|
|
6713
|
-
* the same override (e.g. yolov9s / coreml / all) — paying that spin-up
|
|
6714
|
-
* each iteration dwarfs actual inference time.
|
|
6715
|
-
*
|
|
6716
|
-
* When the same override engine is requested within `OVERRIDE_CACHE_TTL_MS`,
|
|
6717
|
-
* we reuse the prior transient factory. Hitting a different override
|
|
6718
|
-
* or exceeding the TTL disposes the cache and spins a new factory.
|
|
6719
|
-
* The prior "main" factory (`priorFactory`) is still restored so
|
|
6720
|
-
* runtime dispatch (camera frames) keeps its own resident engine.
|
|
6721
|
-
*/
|
|
6722
|
-
overrideCache = null;
|
|
6723
|
-
overrideCacheTimer = null;
|
|
6724
|
-
/**
|
|
6725
7521
|
* Idle-TTL for a per-device inference pool (multi-device C3). A device pool
|
|
6726
7522
|
* with no dispatch for this long is disposed and recreated on next use — the
|
|
6727
7523
|
* memory guardrail that keeps `factoriesByDevice` from accumulating warm pools
|
|
@@ -6754,7 +7550,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6754
7550
|
error: err instanceof Error ? err.message : String(err)
|
|
6755
7551
|
} })
|
|
6756
7552
|
});
|
|
6757
|
-
static OVERRIDE_CACHE_TTL_MS = 6e4;
|
|
6758
7553
|
/**
|
|
6759
7554
|
* RSS bound + periodic memory telemetry for the Python inference pools
|
|
6760
7555
|
* (2026-08-18 OOM audit: the pools, not the recorder, were the host's OOM
|
|
@@ -6891,15 +7686,14 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6891
7686
|
/**
|
|
6892
7687
|
* Lazy-install the pip requirements file matching `engine.backend` into
|
|
6893
7688
|
* the embedded Python before the inference pool spawns. No-op for
|
|
6894
|
-
*
|
|
6895
|
-
*
|
|
7689
|
+
* backends with no requirements file on disk, or before the addon
|
|
7690
|
+
* context has been wired (warm-pool race window).
|
|
6896
7691
|
*
|
|
6897
7692
|
* Idempotent — `installPythonRequirements` short-circuits on a hash
|
|
6898
7693
|
* marker, so back-to-back EngineFactory rebuilds (benchmark overrides,
|
|
6899
7694
|
* tuning respawns) skip the pip subprocess after the first install.
|
|
6900
7695
|
*/
|
|
6901
7696
|
async ensureBackendDeps(engine) {
|
|
6902
|
-
if (engine.runtime !== "python") return;
|
|
6903
7697
|
if (!this.addonCtx) return;
|
|
6904
7698
|
const pythonAddonDir = this.executorOptions.pythonAddonDir;
|
|
6905
7699
|
if (!pythonAddonDir) return;
|
|
@@ -7464,8 +8258,22 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7464
8258
|
* result with a warning logged, same posture as `recomputeConfigIssues`.
|
|
7465
8259
|
*/
|
|
7466
8260
|
async validatePipeline(input) {
|
|
7467
|
-
|
|
8261
|
+
let format = this.currentEngine.format;
|
|
7468
8262
|
try {
|
|
8263
|
+
if (input.deviceKey !== void 0) try {
|
|
8264
|
+
format = resolveDeviceEngine(input.deviceKey).format;
|
|
8265
|
+
} catch (err) {
|
|
8266
|
+
return {
|
|
8267
|
+
ok: false,
|
|
8268
|
+
issues: [{
|
|
8269
|
+
addonId: "*",
|
|
8270
|
+
kind: "unknown-device",
|
|
8271
|
+
detail: require_dist.errMsg(err)
|
|
8272
|
+
}],
|
|
8273
|
+
substitutions: [],
|
|
8274
|
+
format
|
|
8275
|
+
};
|
|
8276
|
+
}
|
|
7469
8277
|
const steps = input.steps;
|
|
7470
8278
|
const enabledSteps = this.pruneToEnabledSteps(steps);
|
|
7471
8279
|
const substitutions = collectSubstitutions(enabledSteps, format);
|
|
@@ -7790,7 +8598,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7790
8598
|
const nodeId = this.addonCtx?.kernel?.localNodeId ?? "hub";
|
|
7791
8599
|
const sessionId = input.sessionId ?? `run-${Date.now()}-${Math.random().toString(36).slice(2, 10)}`;
|
|
7792
8600
|
const isRuntime = Boolean(input.frame || input.frameRef);
|
|
7793
|
-
const progressToBus = !isRuntime || input.sessionId !== void 0;
|
|
8601
|
+
const progressToBus = !isRuntime && input.replay !== true || input.sessionId !== void 0;
|
|
7794
8602
|
const emit = (message, extra) => {
|
|
7795
8603
|
onProgress?.(message);
|
|
7796
8604
|
if (!progressToBus) return;
|
|
@@ -7932,7 +8740,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7932
8740
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
7933
8741
|
steps: input.steps,
|
|
7934
8742
|
deviceKey: input.deviceKey,
|
|
7935
|
-
engineOverride: input.engine,
|
|
7936
8743
|
deviceId: input.deviceId,
|
|
7937
8744
|
plane: input.plane
|
|
7938
8745
|
});
|
|
@@ -7952,322 +8759,153 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7952
8759
|
}
|
|
7953
8760
|
const enabledSteps = flattenSteps(benchmarkSteps);
|
|
7954
8761
|
emit(`Pipeline: ${enabledSteps.length} step(s) — ${enabledSteps.map((s) => s.addonId).join(" → ")}`);
|
|
7955
|
-
|
|
7956
|
-
const
|
|
7957
|
-
|
|
7958
|
-
|
|
7959
|
-
|
|
7960
|
-
|
|
7961
|
-
|
|
7962
|
-
|
|
7963
|
-
|
|
7964
|
-
|
|
7965
|
-
|
|
7966
|
-
|
|
7967
|
-
const cached = this.overrideCache;
|
|
7968
|
-
if (cached.initPromise) await cached.initPromise;
|
|
7969
|
-
this.log.debug("Benchmark: reusing warm override factory", { meta: { engine: `${runEngine.runtime}/${runEngine.backend}/${runEngine.device ?? "default"}` } });
|
|
7970
|
-
this.currentEngine = runEngine;
|
|
7971
|
-
this.engineFactory = cached.factory;
|
|
7972
|
-
this.executor = cached.executor;
|
|
7973
|
-
} else {
|
|
7974
|
-
if (this.overrideCache) {
|
|
7975
|
-
try {
|
|
7976
|
-
await this.overrideCache.factory.dispose();
|
|
7977
|
-
} catch {}
|
|
7978
|
-
this.overrideCache = null;
|
|
7979
|
-
}
|
|
7980
|
-
this.log.info("Benchmark: engine override", { meta: {
|
|
7981
|
-
from: `${priorEngine.runtime}/${priorEngine.backend}`,
|
|
7982
|
-
to: `${runEngine.runtime}/${runEngine.backend}`
|
|
7983
|
-
} });
|
|
7984
|
-
await this.ensureBackendDeps(runEngine);
|
|
7985
|
-
const newFactory = new EngineFactory({
|
|
7986
|
-
engine: runEngine,
|
|
7987
|
-
modelsDir: this.modelsDir,
|
|
7988
|
-
logger: this.log.child("engine-override"),
|
|
7989
|
-
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
7990
|
-
provisioning: this.executorOptions.provisioning,
|
|
7991
|
-
resolveCustomModel: this.customModelResolver
|
|
7992
|
-
});
|
|
7993
|
-
const initPromise = newFactory.initialize([]);
|
|
7994
|
-
this.overrideCache = {
|
|
7995
|
-
engine: runEngine,
|
|
7996
|
-
factory: newFactory,
|
|
7997
|
-
executor: null,
|
|
7998
|
-
lastUsedMs: Date.now(),
|
|
7999
|
-
initPromise
|
|
8000
|
-
};
|
|
8001
|
-
this.scheduleOverrideCacheEviction();
|
|
8002
|
-
try {
|
|
8003
|
-
await initPromise;
|
|
8004
|
-
this.overrideCache.initPromise = null;
|
|
8005
|
-
} catch (err) {
|
|
8006
|
-
this.overrideCache = null;
|
|
8007
|
-
if (this.overrideCacheTimer) {
|
|
8008
|
-
clearTimeout(this.overrideCacheTimer);
|
|
8009
|
-
this.overrideCacheTimer = null;
|
|
8010
|
-
}
|
|
8011
|
-
throw err;
|
|
8012
|
-
}
|
|
8013
|
-
this.currentEngine = runEngine;
|
|
8014
|
-
this.engineFactory = newFactory;
|
|
8015
|
-
this.executor = null;
|
|
8016
|
-
}
|
|
8017
|
-
try {
|
|
8018
|
-
const deviceLabel = runEngine.device ?? "default";
|
|
8019
|
-
emit(`Engine: ${runEngine.runtime}/${runEngine.backend}/${deviceLabel}${input.engine ? " (override)" : ""}`, {
|
|
8020
|
-
step: "engine",
|
|
8021
|
-
engine: {
|
|
8022
|
-
runtime: runEngine.runtime,
|
|
8023
|
-
backend: runEngine.backend,
|
|
8024
|
-
device: deviceLabel
|
|
8025
|
-
}
|
|
8026
|
-
});
|
|
8027
|
-
const runtimeStr = `${runEngine.runtime}+${runEngine.backend}`;
|
|
8028
|
-
const executor = this.executor ?? new PipelineExecutor({
|
|
8029
|
-
engineRuntime: runtimeStr,
|
|
8030
|
-
logger: this.log
|
|
8762
|
+
const rootModelId = resolveRootModelId(benchmarkSteps);
|
|
8763
|
+
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
8764
|
+
...r,
|
|
8765
|
+
rootModelId
|
|
8766
|
+
};
|
|
8767
|
+
if (input.replay === true) {
|
|
8768
|
+
const verdict = checkReplayPin({
|
|
8769
|
+
requested: input.steps,
|
|
8770
|
+
executing: input.plane === "frame" ? collectFramePlaneSteps(benchmarkSteps, isDetailPlaneStep) : enabledSteps,
|
|
8771
|
+
requestedDeviceKey: input.deviceKey,
|
|
8772
|
+
dispatchDeviceKey,
|
|
8773
|
+
format: dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format : this.currentEngine.format
|
|
8031
8774
|
});
|
|
8032
|
-
|
|
8033
|
-
|
|
8034
|
-
|
|
8035
|
-
|
|
8036
|
-
|
|
8037
|
-
|
|
8038
|
-
|
|
8039
|
-
|
|
8040
|
-
} });
|
|
8041
|
-
for (const s of needed) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8042
|
-
step: s.addonId,
|
|
8043
|
-
addonId: s.addonId,
|
|
8044
|
-
modelId: s.modelId
|
|
8775
|
+
if (verdict.kind === "refused") {
|
|
8776
|
+
this.log.warn("replay refused — the node cannot run exactly what was pinned", {
|
|
8777
|
+
...input.deviceId !== void 0 ? { tags: { deviceId: input.deviceId } } : {},
|
|
8778
|
+
meta: {
|
|
8779
|
+
reason: verdict.reason,
|
|
8780
|
+
detail: verdict.detail,
|
|
8781
|
+
deviceKey: input.deviceKey
|
|
8782
|
+
}
|
|
8045
8783
|
});
|
|
8046
|
-
|
|
8047
|
-
emit(`All models loaded`);
|
|
8048
|
-
}
|
|
8049
|
-
emit("Running inference...");
|
|
8050
|
-
const tree = buildExecutableTree(benchmarkSteps, (stepId) => dispatchFactory.getEngine(stepId), this.customModelResolver);
|
|
8051
|
-
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8052
|
-
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8053
|
-
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
8054
|
-
const effectiveTree = Object.keys(deviceOverrides).length > 0 ? applyDeviceOverridesToTree(tree, "object-detection", deviceOverrides) : tree;
|
|
8055
|
-
const nativeCropProvider = this.buildNativeCropProviderFromRef(input.nativeCropRef) ?? this.buildNativeFaceCropProvider(input.frameHandle);
|
|
8056
|
-
let cropZoneBbox;
|
|
8057
|
-
if (isRuntime && effectiveDeviceId > 0 && effectiveTree.roots.some((r) => r.definition.extractMode === "crop-zone")) {
|
|
8058
|
-
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8059
|
-
const proxy = this.deviceProxies.get(effectiveDeviceId);
|
|
8060
|
-
if (proxy) cropZoneBbox = resolvePackageCropBbox(proxy.state.zones.value?.zones ?? [], proxy.state.zoneRules.value?.package ?? [], imageWidth, imageHeight) ?? void 0;
|
|
8061
|
-
}
|
|
8062
|
-
let frameViewResolver;
|
|
8063
|
-
let rootInputViewProvider;
|
|
8064
|
-
if (runtimeFrameRef) {
|
|
8065
|
-
frameViewResolver = createRootFrameViewResolver(require_default_detection_model.localFrameRegistry, runtimeFrameRef);
|
|
8066
|
-
rootInputViewProvider = async (root, crop) => {
|
|
8067
|
-
const view = await frameViewResolver(root, crop ? {
|
|
8068
|
-
left: crop[0],
|
|
8069
|
-
top: crop[1],
|
|
8070
|
-
width: crop[2] - crop[0],
|
|
8071
|
-
height: crop[3] - crop[1]
|
|
8072
|
-
} : void 0);
|
|
8073
|
-
const data = Buffer.from(view.data.buffer, view.data.byteOffset, view.data.byteLength);
|
|
8074
|
-
return {
|
|
8075
|
-
input: {
|
|
8076
|
-
kind: "jpeg",
|
|
8077
|
-
data
|
|
8078
|
-
},
|
|
8079
|
-
jpegProvider: async () => data,
|
|
8080
|
-
width: view.width,
|
|
8081
|
-
height: view.height,
|
|
8082
|
-
geometry: view.geometry
|
|
8083
|
-
};
|
|
8084
|
-
};
|
|
8085
|
-
}
|
|
8086
|
-
let executionSucceeded = false;
|
|
8087
|
-
let execution;
|
|
8088
|
-
try {
|
|
8089
|
-
execution = await executor.run(effectiveTree, rootInput, jpegProvider, imageWidth, imageHeight, effectiveDeviceId, {
|
|
8090
|
-
traceVerbosity: isRuntime ? this.eventBus ? "summary" : "off" : "full",
|
|
8091
|
-
plane: input.plane
|
|
8092
|
-
}, nativeCropProvider, cropZoneBbox, rootInputViewProvider);
|
|
8093
|
-
executionSucceeded = true;
|
|
8094
|
-
} finally {
|
|
8095
|
-
frameViewResolver?.release(executionSucceeded ? "success" : "error");
|
|
8096
|
-
}
|
|
8097
|
-
const { result, trace } = execution;
|
|
8098
|
-
if (isRuntime) {
|
|
8099
|
-
if (trace && this.eventBus) this.eventBus.emit(require_dist.createEvent(require_dist.EventCategory.PipelineTrace, {
|
|
8100
|
-
type: "device",
|
|
8101
|
-
id: trace.deviceId,
|
|
8102
|
-
nodeId: "hub"
|
|
8103
|
-
}, trace));
|
|
8104
|
-
if (effectiveDeviceId > 0) {
|
|
8105
|
-
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8106
|
-
return this.gateDetectionsByZoneRules(effectiveDeviceId, result);
|
|
8107
|
-
}
|
|
8108
|
-
return result;
|
|
8109
|
-
}
|
|
8110
|
-
for (const t of result.debug?.stepTimings ?? []) emit(`${t.source}${t.modelId ? ` (${t.modelId})` : ""} → ${t.ms}ms`, {
|
|
8111
|
-
step: t.source,
|
|
8112
|
-
addonId: t.source,
|
|
8113
|
-
modelId: t.modelId ?? void 0,
|
|
8114
|
-
ms: t.ms
|
|
8115
|
-
});
|
|
8116
|
-
emit(`Done — ${result.detections.length} detection(s) in ${result.debug?.totalInferenceMs ?? 0}ms`, { ms: result.debug?.totalInferenceMs });
|
|
8117
|
-
const wallMs = performance.now() - wallT0;
|
|
8118
|
-
const inferMs = result.debug?.totalInferenceMs ?? 0;
|
|
8119
|
-
const overheadMs = Math.max(0, wallMs - decodeMs - setupMs - inferMs);
|
|
8120
|
-
return {
|
|
8121
|
-
...result,
|
|
8122
|
-
debug: {
|
|
8123
|
-
...result.debug,
|
|
8124
|
-
decodeMs: Math.round(decodeMs * 100) / 100,
|
|
8125
|
-
setupMs: Math.round(setupMs * 100) / 100,
|
|
8126
|
-
wallMs: Math.round(wallMs * 100) / 100,
|
|
8127
|
-
overheadMs: Math.round(overheadMs * 100) / 100
|
|
8128
|
-
}
|
|
8129
|
-
};
|
|
8130
|
-
} finally {
|
|
8131
|
-
if (differs && this.overrideCache) {
|
|
8132
|
-
this.overrideCache.executor = this.executor;
|
|
8133
|
-
this.overrideCache.lastUsedMs = Date.now();
|
|
8134
|
-
this.scheduleOverrideCacheEviction();
|
|
8784
|
+
throw new Error(`${verdict.reason}: ${verdict.detail}`);
|
|
8135
8785
|
}
|
|
8136
|
-
if (differs) {
|
|
8137
|
-
this.currentEngine = priorEngine;
|
|
8138
|
-
this.engineFactory = priorFactory;
|
|
8139
|
-
this.executor = priorExecutor;
|
|
8140
|
-
}
|
|
8141
|
-
}
|
|
8142
|
-
}
|
|
8143
|
-
/**
|
|
8144
|
-
* Apply a benchmark engine override (mirroring the body of `runPipeline`
|
|
8145
|
-
* lines 863-953). Swaps `currentEngine` / `engineFactory` / `executor`
|
|
8146
|
-
* to the override's warm factory and returns a restore function the
|
|
8147
|
-
* caller MUST invoke in `finally`. When `override` is undefined or
|
|
8148
|
-
* matches the current engine, this is a no-op and the restore function
|
|
8149
|
-
* does nothing — same shape so callers don't need a conditional path.
|
|
8150
|
-
*
|
|
8151
|
-
* Used by both `runPipeline` (single-frame benchmark) and
|
|
8152
|
-
* `runPipelineBatch` (batched fast path). Without this, the batch
|
|
8153
|
-
* path silently ignored `input.engine` and ran the user's override
|
|
8154
|
-
* against the addon's saved engine — making the override matrix in
|
|
8155
|
-
* the bench UI a no-op for batchSize > 1.
|
|
8156
|
-
*/
|
|
8157
|
-
async applyEngineOverride(override) {
|
|
8158
|
-
const priorEngine = this.currentEngine;
|
|
8159
|
-
const priorFactory = this.engineFactory;
|
|
8160
|
-
const priorExecutor = this.executor;
|
|
8161
|
-
if (!override) return () => {};
|
|
8162
|
-
if (!backendPossibleOnPlatform(override.backend)) {
|
|
8163
|
-
this.log.warn("Engine override impossible on this platform — running on own engine", { meta: {
|
|
8164
|
-
override: `${override.runtime}/${override.backend}/${override.device ?? "default"}`,
|
|
8165
|
-
platform: process.platform,
|
|
8166
|
-
arch: process.arch,
|
|
8167
|
-
engine: `${priorEngine.runtime}/${priorEngine.backend}`
|
|
8168
|
-
} });
|
|
8169
|
-
return () => {};
|
|
8170
8786
|
}
|
|
8171
|
-
const
|
|
8172
|
-
|
|
8173
|
-
|
|
8174
|
-
|
|
8175
|
-
|
|
8176
|
-
|
|
8177
|
-
|
|
8178
|
-
|
|
8179
|
-
|
|
8180
|
-
|
|
8181
|
-
this.log.debug("Benchmark: reusing warm override factory", { meta: { engine: `${runEngine.runtime}/${runEngine.backend}/${runEngine.device ?? "default"}` } });
|
|
8182
|
-
this.currentEngine = runEngine;
|
|
8183
|
-
this.engineFactory = cached.factory;
|
|
8184
|
-
this.executor = cached.executor;
|
|
8185
|
-
} else {
|
|
8186
|
-
if (this.overrideCache) {
|
|
8187
|
-
try {
|
|
8188
|
-
await this.overrideCache.factory.dispose();
|
|
8189
|
-
} catch {}
|
|
8190
|
-
this.overrideCache = null;
|
|
8787
|
+
const appliesCameraGates = isRuntime || input.replay === true;
|
|
8788
|
+
await this.ensureEngineFactory();
|
|
8789
|
+
const runEngine = this.currentEngine;
|
|
8790
|
+
const deviceLabel = runEngine.device ?? "default";
|
|
8791
|
+
emit(`Engine: ${runEngine.runtime}/${runEngine.backend}/${deviceLabel}`, {
|
|
8792
|
+
step: "engine",
|
|
8793
|
+
engine: {
|
|
8794
|
+
runtime: runEngine.runtime,
|
|
8795
|
+
backend: runEngine.backend,
|
|
8796
|
+
device: deviceLabel
|
|
8191
8797
|
}
|
|
8192
|
-
|
|
8193
|
-
|
|
8194
|
-
|
|
8798
|
+
});
|
|
8799
|
+
const runtimeStr = `${runEngine.runtime}+${runEngine.backend}`;
|
|
8800
|
+
const executor = this.executor ?? new PipelineExecutor({
|
|
8801
|
+
engineRuntime: runtimeStr,
|
|
8802
|
+
logger: this.log
|
|
8803
|
+
});
|
|
8804
|
+
this.executor = executor;
|
|
8805
|
+
const dispatchEngine = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey) : runEngine;
|
|
8806
|
+
const dispatchFactory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8807
|
+
const needed = enabledSteps.filter((s) => dispatchFactory.needsPoolUpdate(s));
|
|
8808
|
+
if (needed.length > 0) {
|
|
8809
|
+
this.log.info("Benchmark: models to load", { meta: {
|
|
8810
|
+
count: needed.length,
|
|
8811
|
+
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8195
8812
|
} });
|
|
8196
|
-
|
|
8197
|
-
|
|
8198
|
-
|
|
8199
|
-
|
|
8200
|
-
logger: this.log.child("engine-override"),
|
|
8201
|
-
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8202
|
-
provisioning: this.executorOptions.provisioning,
|
|
8203
|
-
resolveCustomModel: this.customModelResolver
|
|
8813
|
+
for (const s of needed) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8814
|
+
step: s.addonId,
|
|
8815
|
+
addonId: s.addonId,
|
|
8816
|
+
modelId: s.modelId
|
|
8204
8817
|
});
|
|
8205
|
-
|
|
8206
|
-
|
|
8207
|
-
|
|
8208
|
-
|
|
8209
|
-
|
|
8210
|
-
|
|
8211
|
-
|
|
8818
|
+
await this.ensureModelsForSteps(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
|
|
8819
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
8820
|
+
deviceKey: deviceKeyOf(dispatchEngine)
|
|
8821
|
+
});
|
|
8822
|
+
emit(`All models loaded`);
|
|
8823
|
+
}
|
|
8824
|
+
emit("Running inference...");
|
|
8825
|
+
const tree = buildExecutableTree(benchmarkSteps, (stepId) => dispatchFactory.getEngine(stepId), this.customModelResolver);
|
|
8826
|
+
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8827
|
+
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8828
|
+
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
8829
|
+
const effectiveTree = Object.keys(deviceOverrides).length > 0 ? applyDeviceOverridesToTree(tree, "object-detection", deviceOverrides) : tree;
|
|
8830
|
+
const nativeCropProvider = this.buildNativeCropProviderFromRef(input.nativeCropRef) ?? this.buildNativeFaceCropProvider(input.frameHandle);
|
|
8831
|
+
let cropZoneBbox;
|
|
8832
|
+
if (appliesCameraGates && effectiveDeviceId > 0 && effectiveTree.roots.some((r) => r.definition.extractMode === "crop-zone")) {
|
|
8833
|
+
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8834
|
+
const proxy = this.deviceProxies.get(effectiveDeviceId);
|
|
8835
|
+
if (proxy) cropZoneBbox = resolvePackageCropBbox(proxy.state.zones.value?.zones ?? [], proxy.state.zoneRules.value?.package ?? [], imageWidth, imageHeight) ?? void 0;
|
|
8836
|
+
}
|
|
8837
|
+
let frameViewResolver;
|
|
8838
|
+
let rootInputViewProvider;
|
|
8839
|
+
if (runtimeFrameRef) {
|
|
8840
|
+
frameViewResolver = createRootFrameViewResolver(require_default_detection_model.localFrameRegistry, runtimeFrameRef);
|
|
8841
|
+
rootInputViewProvider = async (root, crop) => {
|
|
8842
|
+
const view = await frameViewResolver(root, crop ? {
|
|
8843
|
+
left: crop[0],
|
|
8844
|
+
top: crop[1],
|
|
8845
|
+
width: crop[2] - crop[0],
|
|
8846
|
+
height: crop[3] - crop[1]
|
|
8847
|
+
} : void 0);
|
|
8848
|
+
const data = Buffer.from(view.data.buffer, view.data.byteOffset, view.data.byteLength);
|
|
8849
|
+
return {
|
|
8850
|
+
input: {
|
|
8851
|
+
kind: "jpeg",
|
|
8852
|
+
data
|
|
8853
|
+
},
|
|
8854
|
+
jpegProvider: async () => data,
|
|
8855
|
+
width: view.width,
|
|
8856
|
+
height: view.height,
|
|
8857
|
+
geometry: view.geometry
|
|
8858
|
+
};
|
|
8212
8859
|
};
|
|
8213
|
-
|
|
8214
|
-
|
|
8215
|
-
|
|
8216
|
-
|
|
8217
|
-
|
|
8218
|
-
this.
|
|
8219
|
-
|
|
8220
|
-
|
|
8221
|
-
|
|
8222
|
-
|
|
8223
|
-
|
|
8860
|
+
}
|
|
8861
|
+
let executionSucceeded = false;
|
|
8862
|
+
let execution;
|
|
8863
|
+
try {
|
|
8864
|
+
execution = await executor.run(effectiveTree, rootInput, jpegProvider, imageWidth, imageHeight, effectiveDeviceId, {
|
|
8865
|
+
traceVerbosity: isRuntime ? this.eventBus ? "summary" : "off" : "full",
|
|
8866
|
+
plane: input.plane,
|
|
8867
|
+
occasion: input.replay === true ? "replay" : "live"
|
|
8868
|
+
}, nativeCropProvider, cropZoneBbox, rootInputViewProvider);
|
|
8869
|
+
executionSucceeded = true;
|
|
8870
|
+
} finally {
|
|
8871
|
+
frameViewResolver?.release(executionSucceeded ? "success" : "error");
|
|
8872
|
+
}
|
|
8873
|
+
const { result, trace } = execution;
|
|
8874
|
+
if (isRuntime) {
|
|
8875
|
+
if (trace && this.eventBus) this.eventBus.emit(require_dist.createEvent(require_dist.EventCategory.PipelineTrace, {
|
|
8876
|
+
type: "device",
|
|
8877
|
+
id: trace.deviceId,
|
|
8878
|
+
nodeId: "hub"
|
|
8879
|
+
}, trace));
|
|
8880
|
+
if (effectiveDeviceId > 0) {
|
|
8881
|
+
await this.ensureDeviceProxy(effectiveDeviceId);
|
|
8882
|
+
return stampRoot(this.gateDetectionsByZoneRules(effectiveDeviceId, result));
|
|
8224
8883
|
}
|
|
8225
|
-
|
|
8226
|
-
this.engineFactory = newFactory;
|
|
8227
|
-
this.executor = null;
|
|
8884
|
+
return stampRoot(result);
|
|
8228
8885
|
}
|
|
8229
|
-
|
|
8230
|
-
|
|
8231
|
-
|
|
8232
|
-
|
|
8233
|
-
|
|
8886
|
+
for (const t of result.debug?.stepTimings ?? []) emit(`${t.source}${t.modelId ? ` (${t.modelId})` : ""} → ${t.ms}ms`, {
|
|
8887
|
+
step: t.source,
|
|
8888
|
+
addonId: t.source,
|
|
8889
|
+
modelId: t.modelId ?? void 0,
|
|
8890
|
+
ms: t.ms
|
|
8891
|
+
});
|
|
8892
|
+
emit(`Done — ${result.detections.length} detection(s) in ${result.debug?.totalInferenceMs ?? 0}ms`, { ms: result.debug?.totalInferenceMs });
|
|
8893
|
+
const wallMs = performance.now() - wallT0;
|
|
8894
|
+
const inferMs = result.debug?.totalInferenceMs ?? 0;
|
|
8895
|
+
const overheadMs = Math.max(0, wallMs - decodeMs - setupMs - inferMs);
|
|
8896
|
+
const gated = input.replay === true && effectiveDeviceId > 0 ? await this.ensureDeviceProxy(effectiveDeviceId).then(() => this.gateDetectionsByZoneRules(effectiveDeviceId, result)) : result;
|
|
8897
|
+
return {
|
|
8898
|
+
...stampRoot(gated),
|
|
8899
|
+
debug: {
|
|
8900
|
+
...gated.debug,
|
|
8901
|
+
decodeMs: Math.round(decodeMs * 100) / 100,
|
|
8902
|
+
setupMs: Math.round(setupMs * 100) / 100,
|
|
8903
|
+
wallMs: Math.round(wallMs * 100) / 100,
|
|
8904
|
+
overheadMs: Math.round(overheadMs * 100) / 100
|
|
8234
8905
|
}
|
|
8235
|
-
this.currentEngine = priorEngine;
|
|
8236
|
-
this.engineFactory = priorFactory;
|
|
8237
|
-
this.executor = priorExecutor;
|
|
8238
8906
|
};
|
|
8239
8907
|
}
|
|
8240
8908
|
/**
|
|
8241
|
-
* Schedule eviction of the override cache after `OVERRIDE_CACHE_TTL_MS`
|
|
8242
|
-
* of idleness. Rescheduling resets the timer so an active benchmark
|
|
8243
|
-
* keeps its warm factory alive until iterations stop arriving.
|
|
8244
|
-
*/
|
|
8245
|
-
scheduleOverrideCacheEviction() {
|
|
8246
|
-
if (this.overrideCacheTimer) clearTimeout(this.overrideCacheTimer);
|
|
8247
|
-
this.overrideCacheTimer = setTimeout(() => {
|
|
8248
|
-
this.evictOverrideCache("idle TTL");
|
|
8249
|
-
}, DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS);
|
|
8250
|
-
}
|
|
8251
|
-
async evictOverrideCache(reason) {
|
|
8252
|
-
if (!this.overrideCache) return;
|
|
8253
|
-
const entry = this.overrideCache;
|
|
8254
|
-
this.overrideCache = null;
|
|
8255
|
-
if (this.overrideCacheTimer) {
|
|
8256
|
-
clearTimeout(this.overrideCacheTimer);
|
|
8257
|
-
this.overrideCacheTimer = null;
|
|
8258
|
-
}
|
|
8259
|
-
try {
|
|
8260
|
-
await entry.factory.dispose();
|
|
8261
|
-
this.log.info("Override cache evicted", { meta: {
|
|
8262
|
-
reason,
|
|
8263
|
-
engine: `${entry.engine.runtime}/${entry.engine.backend}/${entry.engine.device ?? "default"}`,
|
|
8264
|
-
ageMs: Date.now() - entry.lastUsedMs
|
|
8265
|
-
} });
|
|
8266
|
-
} catch (err) {
|
|
8267
|
-
this.log.warn("Override cache dispose failed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
8268
|
-
}
|
|
8269
|
-
}
|
|
8270
|
-
/**
|
|
8271
8909
|
* Convert PipelineStepInput[] → PipelineDefaultStep[] by enriching from
|
|
8272
8910
|
* StepDefinitions. This is the LIVE per-camera dispatch path — it runs
|
|
8273
8911
|
* once per decoded frame (up to detectionFps × N cameras/sec) — so it
|
|
@@ -8283,13 +8921,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8283
8921
|
* `resolveStepModels` (see its doc comment); a second writer here would
|
|
8284
8922
|
* corrupt it.
|
|
8285
8923
|
*
|
|
8286
|
-
* `format` defaults to `this.currentEngine.format` (the
|
|
8287
|
-
*
|
|
8288
|
-
* per-run engine override (`input.engine`) MUST pass the override's format
|
|
8289
|
-
* explicitly — `this.currentEngine` only gets swapped to the override
|
|
8290
|
-
* LATER in `runPipeline`/`runPipelineBatch`, so relying on the default here
|
|
8291
|
-
* would resolve models against the node's persisted format instead of the
|
|
8292
|
-
* format actually being benchmarked.
|
|
8924
|
+
* `format` defaults to `this.currentEngine.format` (the node-default
|
|
8925
|
+
* pool); a `deviceKey` dispatch passes that device's format explicitly.
|
|
8293
8926
|
*/
|
|
8294
8927
|
inputStepsToPipelineSteps(steps, format = this.currentEngine.format, engine = {
|
|
8295
8928
|
backend: this.currentEngine.backend,
|
|
@@ -8371,8 +9004,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8371
9004
|
* devices — so the whole-tree gate stays.
|
|
8372
9005
|
*
|
|
8373
9006
|
* Fallback is strictly NODE-LOCAL and ordered: the node's own selected
|
|
8374
|
-
* engine
|
|
8375
|
-
* fallback — frames never migrate to another node from here (that is the
|
|
9007
|
+
* engine is the one and only fallback — frames never migrate to another node from here (that is the
|
|
8376
9008
|
* orchestrator's tier). The fallback is logged ONCE per distinct
|
|
8377
9009
|
* (deviceKey, steps, format) signature, with `tags: { deviceId }` when the
|
|
8378
9010
|
* dispatch is camera-scoped.
|
|
@@ -8383,12 +9015,9 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8383
9015
|
* configuration must stay loud, not vanish into a silent fallback chain.
|
|
8384
9016
|
*/
|
|
8385
9017
|
resolveStepsForDispatch(args) {
|
|
8386
|
-
const { deviceKey,
|
|
8387
|
-
const nodeFormat =
|
|
8388
|
-
const nodeEngine =
|
|
8389
|
-
backend: engineOverride.backend,
|
|
8390
|
-
device: engineOverride.device ?? null
|
|
8391
|
-
} : {
|
|
9018
|
+
const { deviceKey, deviceId } = args;
|
|
9019
|
+
const nodeFormat = this.currentEngine.format;
|
|
9020
|
+
const nodeEngine = {
|
|
8392
9021
|
backend: this.currentEngine.backend,
|
|
8393
9022
|
device: this.currentEngine.device ?? null
|
|
8394
9023
|
};
|
|
@@ -8486,53 +9115,193 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8486
9115
|
throw new Error(message);
|
|
8487
9116
|
}
|
|
8488
9117
|
/**
|
|
8489
|
-
*
|
|
8490
|
-
*
|
|
8491
|
-
*
|
|
8492
|
-
*
|
|
8493
|
-
*
|
|
8494
|
-
*
|
|
8495
|
-
*
|
|
8496
|
-
*/
|
|
8497
|
-
modelLoadInFlight =
|
|
9118
|
+
* In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
|
|
9119
|
+
* needing the same models on the same pool share ONE load and its outcome;
|
|
9120
|
+
* a dispatch needing a different set is never handed another set's failure
|
|
9121
|
+
* — it queues behind the pool's current load ({@link modelLoadTail}) and
|
|
9122
|
+
* then decides for itself. Keyed by pool, not node-wide: one GPU compile
|
|
9123
|
+
* that never returned held EVERY load on the node — the NPU's and the CPU's
|
|
9124
|
+
* too — behind it.
|
|
9125
|
+
*/
|
|
9126
|
+
modelLoadInFlight = /* @__PURE__ */ new Map();
|
|
9127
|
+
/** The last load issued on each pool — loads into one pool run one at a time. */
|
|
9128
|
+
modelLoadTail = /* @__PURE__ */ new Map();
|
|
9129
|
+
/** Stands in for "the node-default pool" before it exists. */
|
|
9130
|
+
nullPoolKey = {};
|
|
9131
|
+
/** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
|
|
9132
|
+
loadGovernor = new ModelLoadGovernor();
|
|
8498
9133
|
/** Ensure all models needed by steps are downloaded and loaded in the engine pool.
|
|
8499
9134
|
* `factory`/`format` default to the node's engine; a per-device dispatch passes
|
|
8500
|
-
* that device's factory + format (Phase 2 multi-device).
|
|
8501
|
-
|
|
8502
|
-
|
|
8503
|
-
const
|
|
9135
|
+
* that device's factory + format (Phase 2 multi-device). `context` names the
|
|
9136
|
+
* camera and the accelerator on the line a rejected load writes. */
|
|
9137
|
+
async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
|
|
9138
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9139
|
+
if (needed.length === 0) return;
|
|
9140
|
+
const pool = factory ?? this.nullPoolKey;
|
|
9141
|
+
const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
|
|
9142
|
+
const models = needed.map(stepModelKey);
|
|
9143
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9144
|
+
if (refusal !== null) {
|
|
9145
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
|
|
9146
|
+
throw refusal;
|
|
9147
|
+
}
|
|
9148
|
+
const setKey = [...models].sort().join(",");
|
|
9149
|
+
let perPool = this.modelLoadInFlight.get(pool);
|
|
9150
|
+
if (perPool === void 0) {
|
|
9151
|
+
perPool = /* @__PURE__ */ new Map();
|
|
9152
|
+
this.modelLoadInFlight.set(pool, perPool);
|
|
9153
|
+
}
|
|
9154
|
+
const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
|
|
9155
|
+
try {
|
|
9156
|
+
await flight.work;
|
|
9157
|
+
} catch (err) {
|
|
9158
|
+
if (err instanceof ModelLoadRefusedError) {
|
|
9159
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
|
|
9160
|
+
throw err;
|
|
9161
|
+
}
|
|
9162
|
+
const failed = err instanceof StepVariantLoadError ? `${err.stepId}/${err.modelId}` : models[0] ?? "";
|
|
9163
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, require_dist.errMsg(err));
|
|
9164
|
+
throw err;
|
|
9165
|
+
}
|
|
9166
|
+
}
|
|
9167
|
+
/**
|
|
9168
|
+
* Why `models` may not be loaded on `pool` right now — a back-off after a
|
|
9169
|
+
* failure, or a refusal by name after repeated compile timeouts — or `null`.
|
|
9170
|
+
*/
|
|
9171
|
+
loadRefusal(pool, deviceKey, models) {
|
|
9172
|
+
const verdict = this.loadGovernor.check(pool, deviceKey, models);
|
|
9173
|
+
if (verdict.kind === "go") return null;
|
|
9174
|
+
return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
|
|
9175
|
+
}
|
|
9176
|
+
/**
|
|
9177
|
+
* A compile that outlived its soft bound came back on `factory`'s pool
|
|
9178
|
+
* (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
|
|
9179
|
+
* again, so that model's back-off is lifted at once rather than waited out;
|
|
9180
|
+
* no other model's is. A FAILURE is one more failure of that model, and its
|
|
9181
|
+
* back-off grows.
|
|
9182
|
+
*/
|
|
9183
|
+
noteLateCompile(factory, deviceKey, event) {
|
|
9184
|
+
const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
|
|
9185
|
+
if (event.ok) {
|
|
9186
|
+
this.log.info("model loadable again after late compile", { meta: {
|
|
9187
|
+
deviceKey,
|
|
9188
|
+
model: event.model,
|
|
9189
|
+
backoffLifted: models
|
|
9190
|
+
} });
|
|
9191
|
+
return;
|
|
9192
|
+
}
|
|
9193
|
+
this.log.warn("late compile failed — the model stays backed off", { meta: {
|
|
9194
|
+
deviceKey,
|
|
9195
|
+
model: event.model,
|
|
9196
|
+
backoffExtended: models,
|
|
9197
|
+
error: event.error ?? null
|
|
9198
|
+
} });
|
|
9199
|
+
}
|
|
9200
|
+
/** The enabled steps the factory's pool does not reflect yet. */
|
|
9201
|
+
stepsNeedingLoad(steps, factory) {
|
|
8504
9202
|
const needed = [];
|
|
8505
|
-
for (const step of
|
|
9203
|
+
for (const step of flattenSteps(steps)) {
|
|
8506
9204
|
if (!step.enabled) continue;
|
|
8507
9205
|
if (factory !== null && !factory.needsPoolUpdate(step)) continue;
|
|
8508
9206
|
needed.push(step);
|
|
8509
9207
|
}
|
|
8510
|
-
|
|
8511
|
-
|
|
8512
|
-
|
|
8513
|
-
|
|
8514
|
-
|
|
8515
|
-
|
|
8516
|
-
|
|
8517
|
-
|
|
8518
|
-
|
|
8519
|
-
|
|
8520
|
-
|
|
8521
|
-
|
|
8522
|
-
|
|
8523
|
-
|
|
9208
|
+
return needed;
|
|
9209
|
+
}
|
|
9210
|
+
/**
|
|
9211
|
+
* Issue ONE load for a model set on a pool, queued behind the pool's
|
|
9212
|
+
* previous load, and record its outcome with the governor. The set is
|
|
9213
|
+
* re-derived once the queue reaches it: the load ahead may have brought
|
|
9214
|
+
* some of it in.
|
|
9215
|
+
*/
|
|
9216
|
+
issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
|
|
9217
|
+
const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
|
|
9218
|
+
const flight = {
|
|
9219
|
+
work: Promise.resolve(),
|
|
9220
|
+
generation: null
|
|
9221
|
+
};
|
|
9222
|
+
flight.work = (async () => {
|
|
9223
|
+
await previous;
|
|
9224
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9225
|
+
if (needed.length === 0) return;
|
|
9226
|
+
const models = needed.map(stepModelKey);
|
|
9227
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9228
|
+
if (refusal !== null) throw refusal;
|
|
9229
|
+
try {
|
|
9230
|
+
await this.downloadNeededModels(needed, format);
|
|
9231
|
+
this.log.info("Loading additional models for benchmark", { meta: {
|
|
9232
|
+
count: needed.length,
|
|
9233
|
+
models,
|
|
9234
|
+
deviceKey
|
|
9235
|
+
} });
|
|
9236
|
+
await factory.loadAdditional(needed);
|
|
9237
|
+
} catch (err) {
|
|
9238
|
+
flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
|
|
9239
|
+
throw err;
|
|
8524
9240
|
}
|
|
8525
|
-
this.
|
|
8526
|
-
count: needed.length,
|
|
8527
|
-
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8528
|
-
} });
|
|
8529
|
-
await factory.loadAdditional(needed);
|
|
9241
|
+
this.loadGovernor.recordSuccess(pool, deviceKey, models);
|
|
8530
9242
|
})();
|
|
8531
|
-
|
|
8532
|
-
|
|
8533
|
-
|
|
8534
|
-
|
|
8535
|
-
if (
|
|
9243
|
+
perPool.set(setKey, flight);
|
|
9244
|
+
const tail = flight.work.catch(() => void 0);
|
|
9245
|
+
this.modelLoadTail.set(pool, tail);
|
|
9246
|
+
tail.then(() => {
|
|
9247
|
+
if (perPool.get(setKey) === flight) perPool.delete(setKey);
|
|
9248
|
+
if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
|
|
9249
|
+
});
|
|
9250
|
+
return flight;
|
|
9251
|
+
}
|
|
9252
|
+
/** Charge a failed load to the model that failed; say so once if it is now refused. */
|
|
9253
|
+
recordModelLoadFailure(pool, deviceKey, models, err) {
|
|
9254
|
+
const failedModels = err instanceof StepVariantLoadError ? [`${err.stepId}/${err.modelId}`] : models;
|
|
9255
|
+
const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
|
|
9256
|
+
const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
|
|
9257
|
+
let generation = 0;
|
|
9258
|
+
for (const model of failedModels) {
|
|
9259
|
+
const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, require_dist.errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
|
|
9260
|
+
generation = outcome.generation;
|
|
9261
|
+
if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
|
|
9262
|
+
deviceKey,
|
|
9263
|
+
model,
|
|
9264
|
+
compileTimeouts: outcome.compileTimeouts,
|
|
9265
|
+
error: require_dist.errMsg(err),
|
|
9266
|
+
device: "keeps serving every other model",
|
|
9267
|
+
rearm: "pipelineExecutor.rearmInferenceDevice"
|
|
9268
|
+
} });
|
|
9269
|
+
}
|
|
9270
|
+
return generation;
|
|
9271
|
+
}
|
|
9272
|
+
/**
|
|
9273
|
+
* One WARN per camera per state change of the model that cost it its frame.
|
|
9274
|
+
* The ERROR for the failed load itself is written once, where the load
|
|
9275
|
+
* failed (`Step variant load failed`); this line answers "which cameras
|
|
9276
|
+
* did it cost?", tagged so it can be counted per camera.
|
|
9277
|
+
*/
|
|
9278
|
+
reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
|
|
9279
|
+
if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
|
|
9280
|
+
this.log.warn("model load for dispatch failed", {
|
|
9281
|
+
...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
|
|
9282
|
+
meta: {
|
|
9283
|
+
deviceKey,
|
|
9284
|
+
models,
|
|
9285
|
+
model,
|
|
9286
|
+
state,
|
|
9287
|
+
error
|
|
9288
|
+
}
|
|
9289
|
+
});
|
|
9290
|
+
}
|
|
9291
|
+
/** Download any model of `needed` missing in `format` (the device's). */
|
|
9292
|
+
async downloadNeededModels(needed, format) {
|
|
9293
|
+
for (const step of needed) {
|
|
9294
|
+
const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
|
|
9295
|
+
if (modelEntry && !(0, _camstack_system_addon_utils.isModelDownloaded)(this.modelsDir, modelEntry, format)) {
|
|
9296
|
+
if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
|
|
9297
|
+
if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
|
|
9298
|
+
this.log.info("Downloading model for step", { meta: {
|
|
9299
|
+
modelId: step.modelId,
|
|
9300
|
+
format,
|
|
9301
|
+
step: step.addonId
|
|
9302
|
+
} });
|
|
9303
|
+
await this.downloadWithRetry(modelEntry, format, 3);
|
|
9304
|
+
}
|
|
8536
9305
|
}
|
|
8537
9306
|
}
|
|
8538
9307
|
/** Download a model with retry + exponential backoff */
|
|
@@ -8633,12 +9402,11 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8633
9402
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
8634
9403
|
steps: input.steps,
|
|
8635
9404
|
deviceKey: input.deviceKey,
|
|
8636
|
-
engineOverride: input.engine,
|
|
8637
9405
|
deviceId: input.deviceId,
|
|
8638
9406
|
plane: void 0
|
|
8639
9407
|
});
|
|
8640
9408
|
const dispatchDeviceKey = dispatchResolution.deviceKey;
|
|
8641
|
-
const resolveFormat = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format :
|
|
9409
|
+
const resolveFormat = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey).format : this.currentEngine.format;
|
|
8642
9410
|
const benchmarkSteps = dispatchResolution.steps;
|
|
8643
9411
|
const enabledSteps = flattenEnabledVideoSteps(benchmarkSteps);
|
|
8644
9412
|
if (enabledSteps.length === 0) throw new Error("runPipelineBatch: no enabled steps");
|
|
@@ -8659,53 +9427,48 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8659
9427
|
const uniformDims = input.frames.every((f) => f.width === firstFrame.width && f.height === firstFrame.height);
|
|
8660
9428
|
const singleRoot = enabledSteps.length === 1;
|
|
8661
9429
|
await this.ensureEngineFactory();
|
|
8662
|
-
const
|
|
8663
|
-
|
|
8664
|
-
|
|
8665
|
-
|
|
8666
|
-
|
|
8667
|
-
|
|
8668
|
-
|
|
8669
|
-
|
|
8670
|
-
|
|
8671
|
-
|
|
8672
|
-
|
|
8673
|
-
|
|
8674
|
-
|
|
8675
|
-
|
|
8676
|
-
|
|
8677
|
-
|
|
8678
|
-
|
|
8679
|
-
|
|
8680
|
-
|
|
8681
|
-
|
|
8682
|
-
|
|
8683
|
-
|
|
8684
|
-
|
|
8685
|
-
|
|
8686
|
-
|
|
8687
|
-
|
|
8688
|
-
|
|
8689
|
-
|
|
8690
|
-
|
|
8691
|
-
|
|
8692
|
-
|
|
8693
|
-
|
|
8694
|
-
|
|
8695
|
-
|
|
8696
|
-
|
|
8697
|
-
|
|
8698
|
-
|
|
8699
|
-
|
|
8700
|
-
|
|
8701
|
-
|
|
8702
|
-
|
|
8703
|
-
|
|
8704
|
-
const perFrameMs = totalMs / rawResults.length;
|
|
8705
|
-
return { results: rawResults.map((raw, i) => assembleBatchedFrameResult(raw, input.frames[i], rootStep.addonId, input.deviceId ?? 0, perFrameMs, perFrameMs)) };
|
|
8706
|
-
} finally {
|
|
8707
|
-
restoreEngine();
|
|
8708
|
-
}
|
|
9430
|
+
const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
9431
|
+
if (!factory) throw new Error("runPipelineBatch: factory not initialised");
|
|
9432
|
+
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
|
|
9433
|
+
const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
|
|
9434
|
+
this.log.info("runPipelineBatch path decision", { meta: {
|
|
9435
|
+
phase: "batch",
|
|
9436
|
+
canFastPath,
|
|
9437
|
+
singleRoot,
|
|
9438
|
+
allRaw,
|
|
9439
|
+
uniformDims,
|
|
9440
|
+
supportsBatch: factory.supportsBatch()
|
|
9441
|
+
} });
|
|
9442
|
+
if (!canFastPath) return { results: await Promise.all(input.frames.map((frame) => this.runPipeline({
|
|
9443
|
+
steps: input.steps,
|
|
9444
|
+
frame,
|
|
9445
|
+
deviceId: input.deviceId,
|
|
9446
|
+
sessionId: input.sessionId,
|
|
9447
|
+
deviceKey: dispatchDeviceKey
|
|
9448
|
+
}))) };
|
|
9449
|
+
const items = input.frames.map((frame) => ({
|
|
9450
|
+
raw: Buffer.from(frame.data),
|
|
9451
|
+
width: frame.width,
|
|
9452
|
+
height: frame.height,
|
|
9453
|
+
format: frame.format
|
|
9454
|
+
}));
|
|
9455
|
+
this.log.info("runPipelineBatch fast path dispatch", { meta: {
|
|
9456
|
+
phase: "batch",
|
|
9457
|
+
stepId: rootStep.addonId,
|
|
9458
|
+
modelId: rootStep.modelId,
|
|
9459
|
+
items: items.length,
|
|
9460
|
+
totalRawBytes: items.reduce((s, it) => s + it.raw.length, 0)
|
|
9461
|
+
} });
|
|
9462
|
+
const start = performance.now();
|
|
9463
|
+
const rawResults = await factory.batchInferRaw(rootStep.addonId, items, rootStep.modelId, input.frameId ?? 0);
|
|
9464
|
+
const totalMs = performance.now() - start;
|
|
9465
|
+
this.log.info("runPipelineBatch fast path returned", { meta: {
|
|
9466
|
+
phase: "batch",
|
|
9467
|
+
resultCount: rawResults.length,
|
|
9468
|
+
ms: Math.round(totalMs)
|
|
9469
|
+
} });
|
|
9470
|
+
const perFrameMs = totalMs / rawResults.length;
|
|
9471
|
+
return { results: rawResults.map((raw, i) => assembleBatchedFrameResult(raw, input.frames[i], rootStep.addonId, input.deviceId ?? 0, perFrameMs, perFrameMs)) };
|
|
8709
9472
|
}
|
|
8710
9473
|
async listReferenceImages() {
|
|
8711
9474
|
const dir = this.resolveRefImagesDir();
|
|
@@ -8810,7 +9573,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8810
9573
|
this.poolMemoryGuard = null;
|
|
8811
9574
|
this.inferenceTimeoutGuard?.stop();
|
|
8812
9575
|
this.inferenceTimeoutGuard = null;
|
|
8813
|
-
await this.evictOverrideCache("shutdown");
|
|
8814
9576
|
if (this.engineFactory) {
|
|
8815
9577
|
await this.engineFactory.dispose();
|
|
8816
9578
|
this.engineFactory = null;
|
|
@@ -8892,13 +9654,22 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8892
9654
|
* own runtime's row, so this is not "the node's tuning" — the per-pool truth
|
|
8893
9655
|
* is on each pool's `Python inference pool` log line and in `listLoadedEngines`.
|
|
8894
9656
|
*/
|
|
8895
|
-
|
|
8896
|
-
|
|
9657
|
+
/**
|
|
9658
|
+
* One pool's provisioning — the device pool named by `deviceKey`, else the
|
|
9659
|
+
* node default. Resolved with the SAME override `resolveDeviceFactory` and
|
|
9660
|
+
* `ensureEngineFactory` build the pool with, so the answer describes the
|
|
9661
|
+
* pool that ran the work (D646).
|
|
9662
|
+
*/
|
|
9663
|
+
async getEffectiveTuning(input = {}) {
|
|
9664
|
+
const p = resolvePoolProvisioning(input.deviceKey !== void 0 ? resolveDeviceEngine(input.deviceKey) : this.currentEngine, this.executorOptions.provisioning);
|
|
8897
9665
|
return {
|
|
9666
|
+
deviceKey: input.deviceKey ?? null,
|
|
9667
|
+
runtime: p.runtime,
|
|
8898
9668
|
batchMode: p.batchMode,
|
|
8899
9669
|
windowMs: p.windowMs,
|
|
8900
9670
|
maxBatchSize: p.maxBatchSize,
|
|
8901
|
-
concurrency: p.concurrency
|
|
9671
|
+
concurrency: p.concurrency,
|
|
9672
|
+
numWorkers: p.numWorkers
|
|
8902
9673
|
};
|
|
8903
9674
|
}
|
|
8904
9675
|
/**
|
|
@@ -8913,7 +9684,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8913
9684
|
if (this.engineFactory.isReady()) return;
|
|
8914
9685
|
const dead = this.engineFactory;
|
|
8915
9686
|
this.engineFactory = null;
|
|
8916
|
-
this.
|
|
9687
|
+
this.noteFactoryDeath(defaultDeviceKey, dead);
|
|
8917
9688
|
await dead.dispose().catch(() => void 0);
|
|
8918
9689
|
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8919
9690
|
}
|
|
@@ -8931,7 +9702,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8931
9702
|
logger: this.log.child("engine"),
|
|
8932
9703
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8933
9704
|
provisioning: this.executorOptions.provisioning,
|
|
8934
|
-
resolveCustomModel: this.customModelResolver
|
|
9705
|
+
resolveCustomModel: this.customModelResolver,
|
|
9706
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
|
|
8935
9707
|
});
|
|
8936
9708
|
try {
|
|
8937
9709
|
await factory.initialize([]);
|
|
@@ -9051,15 +9823,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9051
9823
|
* dispose — which rejects whatever was still in flight on it with a reason
|
|
9052
9824
|
* rather than letting those requests sit until their own deadlines.
|
|
9053
9825
|
*/
|
|
9054
|
-
async condemnDeviceFactory(deviceKey, factory
|
|
9826
|
+
async condemnDeviceFactory(deviceKey, factory) {
|
|
9055
9827
|
if (this.factoriesByDevice.get(deviceKey) === factory) {
|
|
9056
9828
|
this.factoriesByDevice.delete(deviceKey);
|
|
9057
9829
|
this.deviceReaper.cancel(deviceKey);
|
|
9058
|
-
this.
|
|
9830
|
+
this.noteFactoryDeath(deviceKey, factory);
|
|
9059
9831
|
}
|
|
9060
9832
|
await factory.dispose().catch(() => void 0);
|
|
9061
9833
|
}
|
|
9062
9834
|
/**
|
|
9835
|
+
* Charge one dead pool to its device — unless it died of a compile that was
|
|
9836
|
+
* still running at its hard bound (D653, fix round 1). That death is not a
|
|
9837
|
+
* crash of the DEVICE: the model that hung was already counted when its
|
|
9838
|
+
* load answered `compile-timeout`, and the same model timing out again on
|
|
9839
|
+
* the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
|
|
9840
|
+
* the respawns. A crash stays a crash.
|
|
9841
|
+
*/
|
|
9842
|
+
noteFactoryDeath(deviceKey, factory) {
|
|
9843
|
+
const cause = factory.getDeathCause();
|
|
9844
|
+
if (cause !== null && !cause.chargesDeviceBudget) {
|
|
9845
|
+
this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
|
|
9846
|
+
deviceKey,
|
|
9847
|
+
reason: cause.reason,
|
|
9848
|
+
cause: cause.message
|
|
9849
|
+
} });
|
|
9850
|
+
return;
|
|
9851
|
+
}
|
|
9852
|
+
this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
|
|
9853
|
+
}
|
|
9854
|
+
/**
|
|
9063
9855
|
* Every inference device this node currently refuses, and why.
|
|
9064
9856
|
*
|
|
9065
9857
|
* The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
|
|
@@ -9094,7 +9886,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9094
9886
|
state: "backoff",
|
|
9095
9887
|
since: Date.now(),
|
|
9096
9888
|
deaths: 0,
|
|
9097
|
-
lastError:
|
|
9889
|
+
lastError: deathReasonOf(factory)
|
|
9098
9890
|
});
|
|
9099
9891
|
}
|
|
9100
9892
|
return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
|
|
@@ -9111,10 +9903,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9111
9903
|
* no-op, reported as such.
|
|
9112
9904
|
*/
|
|
9113
9905
|
async rearmInferenceDevice(input) {
|
|
9114
|
-
const
|
|
9906
|
+
const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
|
|
9907
|
+
const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
|
|
9908
|
+
const rearmed = rearmedDevice || rearmedModels > 0;
|
|
9115
9909
|
this.log.info("inference device re-armed by operator", { meta: {
|
|
9116
9910
|
deviceKey: input.deviceKey,
|
|
9117
|
-
rearmed
|
|
9911
|
+
rearmed,
|
|
9912
|
+
rearmedModels
|
|
9118
9913
|
} });
|
|
9119
9914
|
return { rearmed };
|
|
9120
9915
|
}
|
|
@@ -9131,7 +9926,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9131
9926
|
this.deviceReaper.touch(deviceKey);
|
|
9132
9927
|
return existing;
|
|
9133
9928
|
}
|
|
9134
|
-
await this.condemnDeviceFactory(deviceKey, existing
|
|
9929
|
+
await this.condemnDeviceFactory(deviceKey, existing);
|
|
9135
9930
|
this.refuseIfDeviceUnusable(deviceKey);
|
|
9136
9931
|
}
|
|
9137
9932
|
const inflight = this.deviceFactoryInflight.get(deviceKey);
|
|
@@ -9144,7 +9939,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9144
9939
|
logger: this.log.child(`engine:${deviceKey}`),
|
|
9145
9940
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
9146
9941
|
provisioning: this.executorOptions.provisioning,
|
|
9147
|
-
resolveCustomModel: this.customModelResolver
|
|
9942
|
+
resolveCustomModel: this.customModelResolver,
|
|
9943
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
|
|
9148
9944
|
});
|
|
9149
9945
|
try {
|
|
9150
9946
|
await factory.initialize([]);
|
|
@@ -9286,9 +10082,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9286
10082
|
}
|
|
9287
10083
|
async listLoadedEngines() {
|
|
9288
10084
|
const out = [];
|
|
9289
|
-
|
|
9290
|
-
const runtimeFactoryIsOverride = this.engineFactory !== null && this.engineFactory === overrideFactory;
|
|
9291
|
-
if (this.engineFactory && !runtimeFactoryIsOverride) {
|
|
10085
|
+
if (this.engineFactory) {
|
|
9292
10086
|
const eng = this.currentEngine;
|
|
9293
10087
|
const engineKey = `${eng.runtime}/${eng.backend}/${eng.device ?? "default"}`;
|
|
9294
10088
|
const loaded = this.engineFactory.listLoaded();
|
|
@@ -9303,22 +10097,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9303
10097
|
idleTtlMs: null
|
|
9304
10098
|
});
|
|
9305
10099
|
}
|
|
9306
|
-
if (this.overrideCache) {
|
|
9307
|
-
const eng = this.overrideCache.engine;
|
|
9308
|
-
const engineKey = `${eng.runtime}/${eng.backend}/${eng.device ?? "default"} (warm)`;
|
|
9309
|
-
const loaded = this.overrideCache.factory.listLoaded();
|
|
9310
|
-
const idleMs = Math.max(0, Date.now() - this.overrideCache.lastUsedMs);
|
|
9311
|
-
if (idleMs <= DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS) out.push({
|
|
9312
|
-
engineKey,
|
|
9313
|
-
engine: eng,
|
|
9314
|
-
modelsLoaded: loaded.map((l) => `${l.stepId}/${l.modelId}`),
|
|
9315
|
-
inUseByCameras: [],
|
|
9316
|
-
kind: "warm-override",
|
|
9317
|
-
poolPid: this.overrideCache.factory.getPoolPid(),
|
|
9318
|
-
idleMs,
|
|
9319
|
-
idleTtlMs: DetectionPipelineProvider.OVERRIDE_CACHE_TTL_MS
|
|
9320
|
-
});
|
|
9321
|
-
}
|
|
9322
10100
|
for (const [deviceKey, factory] of this.factoriesByDevice) {
|
|
9323
10101
|
const eng = resolveDeviceEngine(deviceKey);
|
|
9324
10102
|
const loaded = factory.listLoaded();
|
|
@@ -9347,14 +10125,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
9347
10125
|
return { success: true };
|
|
9348
10126
|
}
|
|
9349
10127
|
async killEngine(input) {
|
|
9350
|
-
if (this.overrideCache && enginesEqual(this.overrideCache.engine, input.engine)) {
|
|
9351
|
-
await this.evictOverrideCache("killEngine");
|
|
9352
|
-
this.log.info("Override engine killed", { meta: {
|
|
9353
|
-
...input.engine,
|
|
9354
|
-
force: input.force ?? false
|
|
9355
|
-
} });
|
|
9356
|
-
return { success: true };
|
|
9357
|
-
}
|
|
9358
10128
|
if (!this.engineFactory) return {
|
|
9359
10129
|
success: false,
|
|
9360
10130
|
reason: "not loaded"
|
|
@@ -9872,7 +10642,7 @@ var DetectionPipelineAddon = class extends require_dist.BaseAddon {
|
|
|
9872
10642
|
/**
|
|
9873
10643
|
* Embedded Python path resolved once at boot via
|
|
9874
10644
|
* `ctx.deps.ensurePython()`. Empty string means the download failed
|
|
9875
|
-
*
|
|
10645
|
+
* — the provider's
|
|
9876
10646
|
* `ensureBackendDeps` and EngineFactory's `initPythonPool` raise a
|
|
9877
10647
|
* clear error in that case.
|
|
9878
10648
|
*/
|