@camstack/addon-pipeline 1.2.295 → 1.2.297
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +8 -0
- package/dist/audio-analyzer/index.js +2 -2
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/{default-detection-model-wNEISVA9.js → default-detection-model-B2n_SFXt.js} +442 -50
- package/dist/{default-detection-model-CAUBVbgK.mjs → default-detection-model-D6DypCkI.mjs} +413 -51
- package/dist/detection-pipeline/index.js +1516 -269
- package/dist/detection-pipeline/index.mjs +1515 -268
- package/dist/{dist-CuxSNLKW.mjs → dist-CxHIpsQs.mjs} +834 -398
- package/dist/{dist-BsVcf5wO.js → dist-DUqr1zyq.js} +845 -409
- package/dist/motion-wasm/index.js +1 -1
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/{node-UFk6I2f6.js → node-Dx6DLQ1j.js} +1 -1
- package/dist/{node-Cqb1QiXk.mjs → node-DyeWu78a.mjs} +1 -1
- package/dist/pipeline-runner/index.js +906 -310
- package/dist/pipeline-runner/index.mjs +906 -310
- package/dist/{process-memory-D0Nvs9rr.mjs → process-memory-BVR4592X.mjs} +1 -1
- package/dist/{process-memory-CJV29sPf.js → process-memory-CMJ3NY-s.js} +1 -1
- package/dist/recorder/index.js +4 -6
- package/dist/recorder/index.mjs +4 -6
- package/dist/{segment-demux-js-DGwmf5hu.js → segment-demux-js-CmGIF_uK.js} +1 -1
- package/dist/{segment-demux-js-DIDWw1iE.mjs → segment-demux-js-Dzga-XgE.mjs} +1 -1
- package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
- package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
- package/dist/stream-broker/_stub.js +2 -2
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-B7nBFqva.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-P71ze8cu.mjs} +2 -2
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CZkpFU-J.mjs +26 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-Dti8Ex88.mjs +26 -0
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-BKlb3qac.mjs → hostInit-CRlStbz4.mjs} +2 -2
- package/dist/stream-broker/index.js +4 -4
- package/dist/stream-broker/index.mjs +4 -4
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
- package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
- package/package.json +1 -1
- package/python/inference_pool.py +422 -64
- package/python/test_inference_pool_compile_off_loop.py +414 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CCIyBvRa.mjs +0 -26
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DeD_UFdb.mjs +0 -26
|
@@ -3,11 +3,11 @@ Object.defineProperties(exports, {
|
|
|
3
3
|
[Symbol.toStringTag]: { value: "Module" }
|
|
4
4
|
});
|
|
5
5
|
const require_chunk = require("../chunk-emK7D4bc.js");
|
|
6
|
-
const require_dist = require("../dist-
|
|
7
|
-
const require_node = require("../node-
|
|
8
|
-
const require_default_detection_model = require("../default-detection-model-
|
|
6
|
+
const require_dist = require("../dist-DUqr1zyq.js");
|
|
7
|
+
const require_node = require("../node-Dx6DLQ1j.js");
|
|
8
|
+
const require_default_detection_model = require("../default-detection-model-B2n_SFXt.js");
|
|
9
9
|
const require_lazy_sharp = require("../lazy-sharp-BuwKlQLR.js");
|
|
10
|
-
const require_process_memory = require("../process-memory-
|
|
10
|
+
const require_process_memory = require("../process-memory-CMJ3NY-s.js");
|
|
11
11
|
let node_os = require("node:os");
|
|
12
12
|
node_os = require_chunk.__toESM(node_os);
|
|
13
13
|
let node_fs = require("node:fs");
|
|
@@ -367,12 +367,14 @@ function normalizeEngineNodeId(rawNodeId) {
|
|
|
367
367
|
* builders eventually reaches ten of them, and the cameras attached by the
|
|
368
368
|
* forgotten one quietly write a foreign feature space into the shared index.
|
|
369
369
|
*
|
|
370
|
-
* ## Why the failure mode keeps the last value
|
|
370
|
+
* ## Why the failure mode keeps the last value — and refuses before the first
|
|
371
371
|
*
|
|
372
372
|
* A failed read must never revert to the registry default MID-PASS: that would
|
|
373
373
|
* silently change what every remaining vector of that dispatch means, which is
|
|
374
374
|
* the exact class of harm the scope exists to prevent. So a failure keeps the
|
|
375
|
-
* last known row and says so,
|
|
375
|
+
* last known row and says so, once per change. Before the FIRST answer there is
|
|
376
|
+
* no row to keep: the step is refused by name (D658) — a default run then made
|
|
377
|
+
* the default model the pool's active variant for the process lifetime.
|
|
376
378
|
*/
|
|
377
379
|
/**
|
|
378
380
|
* The addon whose GLOBAL settings own the cluster row. Same owner as the crop
|
|
@@ -381,35 +383,58 @@ function normalizeEngineNodeId(rawNodeId) {
|
|
|
381
383
|
* enforces it live — neither should own a value the other depends on).
|
|
382
384
|
*/
|
|
383
385
|
var CLUSTER_MODEL_OWNER_ADDON_ID = "pipeline-orchestrator";
|
|
384
|
-
/**
|
|
386
|
+
/**
|
|
387
|
+
* TTL-cached read of the cluster model row.
|
|
388
|
+
*
|
|
389
|
+
* A step's model is KNOWN only once the owner has answered for it (`stored` or
|
|
390
|
+
* `absent`). Until then it is UNKNOWN, and a cluster-scoped step is refused by
|
|
391
|
+
* name rather than run with the registry default (D49, D658): a default run
|
|
392
|
+
* during the owner's boot window made that model the pool's active variant for
|
|
393
|
+
* the whole process, and the face default was ArcFace until D651. Once known,
|
|
394
|
+
* an unanswered read keeps the LAST KNOWN model — never reverts to a default.
|
|
395
|
+
*/
|
|
385
396
|
var ClusterModelSource = class {
|
|
386
|
-
models =
|
|
397
|
+
models = {};
|
|
398
|
+
unknown = new Map(require_dist.CLUSTER_MODEL_SCOPED_STEPS.map((s) => [s.stepId, "not-yet-asked"]));
|
|
387
399
|
settings = require_dist.DEFAULT_CLUSTER_STEP_SETTINGS;
|
|
388
400
|
lastReadAtMs = Number.NEGATIVE_INFINITY;
|
|
389
401
|
inFlight = null;
|
|
390
402
|
logger;
|
|
391
403
|
now;
|
|
392
404
|
ttlMs;
|
|
405
|
+
unknownRetryMs;
|
|
406
|
+
/** The last "not known" state logged, so it is said once per change. */
|
|
407
|
+
lastUnknownLogKey = "";
|
|
393
408
|
constructor(options) {
|
|
394
409
|
this.logger = options.logger;
|
|
395
410
|
this.now = options.now ?? (() => Date.now());
|
|
396
411
|
this.ttlMs = options.ttlMs ?? 6e4;
|
|
412
|
+
this.unknownRetryMs = options.unknownRetryMs ?? 5e3;
|
|
397
413
|
}
|
|
398
|
-
/**
|
|
414
|
+
/**
|
|
415
|
+
* The KNOWN models, `stepId → modelId`. A step absent here is unknown (see
|
|
416
|
+
* {@link unknownSteps}) and must be refused, never defaulted.
|
|
417
|
+
*/
|
|
399
418
|
current() {
|
|
400
419
|
return this.models;
|
|
401
420
|
}
|
|
421
|
+
/** Cluster-scoped steps whose model the owner has not answered yet, with why. */
|
|
422
|
+
unknownSteps() {
|
|
423
|
+
return this.unknown;
|
|
424
|
+
}
|
|
402
425
|
/** Cluster-scoped step knobs in force right now (admission floors, …). */
|
|
403
426
|
currentSettings() {
|
|
404
427
|
return this.settings;
|
|
405
428
|
}
|
|
406
429
|
/**
|
|
407
|
-
* Re-read if the cached row has expired
|
|
408
|
-
* in-flight read; a failure keeps the LAST
|
|
430
|
+
* Re-read if the cached row has expired (sooner while a step is unknown).
|
|
431
|
+
* Concurrent callers share one in-flight read; a failure keeps the LAST
|
|
432
|
+
* KNOWN row.
|
|
409
433
|
*/
|
|
410
434
|
async refresh(api) {
|
|
411
435
|
if (api === void 0) return this.models;
|
|
412
|
-
|
|
436
|
+
const ttl = this.unknown.size > 0 ? Math.min(this.ttlMs, this.unknownRetryMs) : this.ttlMs;
|
|
437
|
+
if (this.now() - this.lastReadAtMs < ttl) return this.models;
|
|
413
438
|
this.inFlight ??= this.read(api).finally(() => {
|
|
414
439
|
this.inFlight = null;
|
|
415
440
|
});
|
|
@@ -417,38 +442,186 @@ var ClusterModelSource = class {
|
|
|
417
442
|
return this.models;
|
|
418
443
|
}
|
|
419
444
|
async read(api) {
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
to: next[stepId]
|
|
436
|
-
})) } });
|
|
437
|
-
this.models = next;
|
|
438
|
-
}
|
|
439
|
-
if (settingsChanged) {
|
|
440
|
-
this.logger.info("cluster step settings changed", { meta: {
|
|
441
|
-
from: { ...this.settings },
|
|
442
|
-
to: { ...nextSettings }
|
|
443
|
-
} });
|
|
444
|
-
this.settings = nextSettings;
|
|
445
|
+
this.lastReadAtMs = this.now();
|
|
446
|
+
await Promise.all([this.readModels(api), this.readSettings(api)]);
|
|
447
|
+
}
|
|
448
|
+
async readModels(api) {
|
|
449
|
+
const answers = await Promise.all(require_dist.CLUSTER_MODEL_SCOPED_STEPS.map(async (step) => ({
|
|
450
|
+
stepId: step.stepId,
|
|
451
|
+
answer: await askOwner(api, step.stepId)
|
|
452
|
+
})));
|
|
453
|
+
const nextModels = { ...this.models };
|
|
454
|
+
const nextUnknown = /* @__PURE__ */ new Map();
|
|
455
|
+
const unanswered = [];
|
|
456
|
+
for (const { stepId, answer } of answers) {
|
|
457
|
+
if (answer.kind === "model") {
|
|
458
|
+
nextModels[stepId] = answer.modelId;
|
|
459
|
+
continue;
|
|
445
460
|
}
|
|
461
|
+
const keeping = this.models[stepId] ?? null;
|
|
462
|
+
unanswered.push({
|
|
463
|
+
stepId,
|
|
464
|
+
reason: answer.reason,
|
|
465
|
+
keeping
|
|
466
|
+
});
|
|
467
|
+
if (keeping === null) nextUnknown.set(stepId, answer.reason);
|
|
468
|
+
}
|
|
469
|
+
const changed = Object.keys(nextModels).filter((stepId) => this.models[stepId] !== void 0 && nextModels[stepId] !== this.models[stepId]);
|
|
470
|
+
if (changed.length > 0) this.logger.info("cluster step model changed — vectors already stored by the previous model need a re-embed pass", { meta: { changed: changed.map((stepId) => ({
|
|
471
|
+
stepId,
|
|
472
|
+
from: this.models[stepId],
|
|
473
|
+
to: nextModels[stepId]
|
|
474
|
+
})) } });
|
|
475
|
+
const learned = Object.keys(nextModels).filter((stepId) => this.models[stepId] === void 0);
|
|
476
|
+
if (learned.length > 0) this.logger.info("cluster step model known — cluster-scoped steps run it", { meta: { known: Object.fromEntries(learned.map((id) => [id, nextModels[id]])) } });
|
|
477
|
+
this.models = nextModels;
|
|
478
|
+
this.unknown = nextUnknown;
|
|
479
|
+
this.reportUnanswered(unanswered);
|
|
480
|
+
}
|
|
481
|
+
/** Once per change of the unanswered set — never once per retry. */
|
|
482
|
+
reportUnanswered(unanswered) {
|
|
483
|
+
const key = JSON.stringify(unanswered);
|
|
484
|
+
if (key === this.lastUnknownLogKey) return;
|
|
485
|
+
this.lastUnknownLogKey = key;
|
|
486
|
+
if (unanswered.length === 0) return;
|
|
487
|
+
this.logger.warn("cluster step model not answered by its owner — a step never known is REFUSED, a known one keeps its last model", { meta: {
|
|
488
|
+
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
489
|
+
unanswered: unanswered.map((u) => ({
|
|
490
|
+
stepId: u.stepId,
|
|
491
|
+
reason: u.reason,
|
|
492
|
+
...u.keeping !== null ? { keeping: u.keeping } : { refused: true }
|
|
493
|
+
}))
|
|
494
|
+
} });
|
|
495
|
+
}
|
|
496
|
+
async readSettings(api) {
|
|
497
|
+
let view;
|
|
498
|
+
try {
|
|
499
|
+
view = await api.addonSettings.getGlobalSettings.query({ addonId: CLUSTER_MODEL_OWNER_ADDON_ID });
|
|
446
500
|
} catch (err) {
|
|
447
|
-
this.logger.warn("cluster step
|
|
501
|
+
this.logger.warn("cluster step settings read failed — keeping the last known knobs", { meta: {
|
|
448
502
|
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
449
|
-
inUse: { ...this.models },
|
|
450
503
|
error: err instanceof Error ? err.message : String(err)
|
|
451
504
|
} });
|
|
505
|
+
return;
|
|
506
|
+
}
|
|
507
|
+
if (view === null) return;
|
|
508
|
+
const nextSettings = require_dist.pickClusterStepSettings(view);
|
|
509
|
+
if (Object.keys(nextSettings).some((stepId) => {
|
|
510
|
+
const from = this.settings[stepId] ?? {};
|
|
511
|
+
const to = nextSettings[stepId] ?? {};
|
|
512
|
+
return [...new Set([...Object.keys(from), ...Object.keys(to)])].some((key) => from[key] !== to[key]);
|
|
513
|
+
})) {
|
|
514
|
+
this.logger.info("cluster step settings changed", { meta: {
|
|
515
|
+
from: { ...this.settings },
|
|
516
|
+
to: { ...nextSettings }
|
|
517
|
+
} });
|
|
518
|
+
this.settings = nextSettings;
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
};
|
|
522
|
+
/** One step's model from the owner's three-way answer; a throw is no answer. */
|
|
523
|
+
async function askOwner(api, stepId) {
|
|
524
|
+
try {
|
|
525
|
+
const choice = await api.pipelineOrchestrator.getClusterModelChoice.query({ stepId });
|
|
526
|
+
switch (choice.kind) {
|
|
527
|
+
case "stored": return {
|
|
528
|
+
kind: "model",
|
|
529
|
+
modelId: choice.modelId
|
|
530
|
+
};
|
|
531
|
+
case "absent": return {
|
|
532
|
+
kind: "model",
|
|
533
|
+
modelId: choice.defaultModelId
|
|
534
|
+
};
|
|
535
|
+
case "unreadable": return {
|
|
536
|
+
kind: "none",
|
|
537
|
+
reason: choice.reason
|
|
538
|
+
};
|
|
539
|
+
}
|
|
540
|
+
} catch (err) {
|
|
541
|
+
return {
|
|
542
|
+
kind: "none",
|
|
543
|
+
reason: `read-failed: ${err instanceof Error ? err.message : String(err)}`
|
|
544
|
+
};
|
|
545
|
+
}
|
|
546
|
+
}
|
|
547
|
+
//#endregion
|
|
548
|
+
//#region src/detection-pipeline/cluster-model-switch-log.ts
|
|
549
|
+
function* clusterScopedSteps(steps) {
|
|
550
|
+
for (const step of steps) {
|
|
551
|
+
if (step.definition.modelScope === "cluster") yield step;
|
|
552
|
+
yield* clusterScopedSteps(step.children);
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
var ClusterModelSwitchLog = class {
|
|
556
|
+
logger;
|
|
557
|
+
/** `deviceId|stepId` → the model that step last ran for that camera. */
|
|
558
|
+
lastRun = /* @__PURE__ */ new Map();
|
|
559
|
+
constructor(logger) {
|
|
560
|
+
this.logger = logger;
|
|
561
|
+
}
|
|
562
|
+
/** Record what `tree` runs for `deviceId`; say so when a cluster step changed model. */
|
|
563
|
+
note(deviceId, tree) {
|
|
564
|
+
if (deviceId <= 0) return;
|
|
565
|
+
for (const step of clusterScopedSteps(tree.roots)) {
|
|
566
|
+
const key = `${String(deviceId)}|${step.stepId}`;
|
|
567
|
+
const previous = this.lastRun.get(key);
|
|
568
|
+
if (previous === step.modelId) continue;
|
|
569
|
+
this.lastRun.set(key, step.modelId);
|
|
570
|
+
if (previous === void 0) continue;
|
|
571
|
+
this.logger.info("cluster-scoped step re-provisioned for this camera — now runs the new model", {
|
|
572
|
+
tags: { deviceId },
|
|
573
|
+
meta: {
|
|
574
|
+
stepId: step.stepId,
|
|
575
|
+
from: previous,
|
|
576
|
+
to: step.modelId
|
|
577
|
+
}
|
|
578
|
+
});
|
|
579
|
+
}
|
|
580
|
+
}
|
|
581
|
+
};
|
|
582
|
+
function* enabledStepIds(steps) {
|
|
583
|
+
for (const step of steps) {
|
|
584
|
+
if (!step.enabled) continue;
|
|
585
|
+
yield step.addonId;
|
|
586
|
+
if (step.children) yield* enabledStepIds(step.children);
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
/** The enabled steps of `steps` (any depth) whose cluster model is unknown. */
|
|
590
|
+
function unknownStepsNamed(steps, unknown) {
|
|
591
|
+
if (unknown.size === 0) return [];
|
|
592
|
+
return [...enabledStepIds(steps)].filter((id) => unknown.has(id));
|
|
593
|
+
}
|
|
594
|
+
/**
|
|
595
|
+
* One line per (camera, step) while a cluster-scoped step is REFUSED because
|
|
596
|
+
* the owner has not answered which model the cluster runs (D658). Said once per
|
|
597
|
+
* episode: when every step becomes known the memory is cleared, so the next
|
|
598
|
+
* episode is said again.
|
|
599
|
+
*/
|
|
600
|
+
var ClusterUnknownRefusalLog = class {
|
|
601
|
+
logger;
|
|
602
|
+
said = /* @__PURE__ */ new Set();
|
|
603
|
+
constructor(logger) {
|
|
604
|
+
this.logger = logger;
|
|
605
|
+
}
|
|
606
|
+
note(deviceId, steps, unknown) {
|
|
607
|
+
if (unknown.size === 0) {
|
|
608
|
+
if (this.said.size > 0) this.said = /* @__PURE__ */ new Set();
|
|
609
|
+
return;
|
|
610
|
+
}
|
|
611
|
+
if (deviceId <= 0) return;
|
|
612
|
+
for (const stepId of enabledStepIds(steps)) {
|
|
613
|
+
const reason = unknown.get(stepId);
|
|
614
|
+
if (reason === void 0) continue;
|
|
615
|
+
const key = `${String(deviceId)}|${stepId}`;
|
|
616
|
+
if (this.said.has(key)) continue;
|
|
617
|
+
this.said = new Set(this.said).add(key);
|
|
618
|
+
this.logger.warn("cluster-scoped step refused for this camera — the owner has not answered which model the cluster runs", {
|
|
619
|
+
tags: { deviceId },
|
|
620
|
+
meta: {
|
|
621
|
+
stepId,
|
|
622
|
+
reason
|
|
623
|
+
}
|
|
624
|
+
});
|
|
452
625
|
}
|
|
453
626
|
}
|
|
454
627
|
};
|
|
@@ -569,6 +742,98 @@ var DeviceOverrideMirror = class {
|
|
|
569
742
|
}
|
|
570
743
|
};
|
|
571
744
|
//#endregion
|
|
745
|
+
//#region src/detection-pipeline/engine/pool-worker-health.ts
|
|
746
|
+
/**
|
|
747
|
+
* The one table of load bounds (D653, fix round 1).
|
|
748
|
+
*
|
|
749
|
+
* The soft bound decides how soon the operator HEARS about a slow compile; the
|
|
750
|
+
* hard bound decides when a compile is given up for hung. A single 120 s bound
|
|
751
|
+
* that also killed the compile was wrong twice: a slow but finite compile was
|
|
752
|
+
* killed before it could write its cache, so every respawn faced the same cold
|
|
753
|
+
* compile and three of them walked the device to `failed`; and on CoreML the
|
|
754
|
+
* bound was shorter than the cross-process compile-cache lock wait alone.
|
|
755
|
+
*/
|
|
756
|
+
var POOL_MODEL_LOAD_BOUNDS = {
|
|
757
|
+
openvino: {
|
|
758
|
+
softMs: 12e4,
|
|
759
|
+
hardMs: 6e5,
|
|
760
|
+
why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
|
|
761
|
+
gilMayBeHeldDuringCompile: false,
|
|
762
|
+
gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
|
|
763
|
+
},
|
|
764
|
+
coreml: {
|
|
765
|
+
softMs: 3e5,
|
|
766
|
+
hardMs: 9e5,
|
|
767
|
+
why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
|
|
768
|
+
gilMayBeHeldDuringCompile: true,
|
|
769
|
+
gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
|
|
770
|
+
},
|
|
771
|
+
onnxruntime: {
|
|
772
|
+
softMs: 12e4,
|
|
773
|
+
hardMs: 6e5,
|
|
774
|
+
why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
|
|
775
|
+
gilMayBeHeldDuringCompile: false,
|
|
776
|
+
gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
|
|
777
|
+
},
|
|
778
|
+
edgetpu: {
|
|
779
|
+
softMs: 12e4,
|
|
780
|
+
hardMs: 6e5,
|
|
781
|
+
why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
|
|
782
|
+
gilMayBeHeldDuringCompile: false,
|
|
783
|
+
gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
|
|
784
|
+
}
|
|
785
|
+
};
|
|
786
|
+
/**
|
|
787
|
+
* The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
|
|
788
|
+
* so the worker's NAMED answer always lands first.
|
|
789
|
+
*/
|
|
790
|
+
var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
|
|
791
|
+
function chargesDeviceBudget(reason) {
|
|
792
|
+
return reason !== "compile-hung";
|
|
793
|
+
}
|
|
794
|
+
/**
|
|
795
|
+
* Tracks live inference outcomes on one worker and says when its executor has
|
|
796
|
+
* produced nothing for too long. Consulted on demand — no timer.
|
|
797
|
+
*
|
|
798
|
+
* A SHED (`dropped`) does not count as a result: the Python loop answers sheds
|
|
799
|
+
* itself, so a worker whose executor threads are hung keeps shedding happily.
|
|
800
|
+
*/
|
|
801
|
+
var InferStallDetector = class {
|
|
802
|
+
unanswered = 0;
|
|
803
|
+
lastResultAt;
|
|
804
|
+
/**
|
|
805
|
+
* When the current streak of unanswered requests began — the first one after
|
|
806
|
+
* a result. IDLE time is not silence: quiet cameras send nothing, and a
|
|
807
|
+
* worker that was asked nothing has failed nothing. Measuring from the last
|
|
808
|
+
* result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
|
|
809
|
+
* cameras) poison a healthy worker and charge the device.
|
|
810
|
+
*/
|
|
811
|
+
streakStartedAt = null;
|
|
812
|
+
constructor(now) {
|
|
813
|
+
this.lastResultAt = now;
|
|
814
|
+
}
|
|
815
|
+
/** A reply that carried a real result (or a real error) from the executor. */
|
|
816
|
+
noteResult(now) {
|
|
817
|
+
this.unanswered = 0;
|
|
818
|
+
this.lastResultAt = now;
|
|
819
|
+
this.streakStartedAt = null;
|
|
820
|
+
}
|
|
821
|
+
/**
|
|
822
|
+
* A live request that ended without a result (deadline expiry or a shed).
|
|
823
|
+
* Returns the stall when the worker has crossed both thresholds.
|
|
824
|
+
*/
|
|
825
|
+
noteUnanswered(now) {
|
|
826
|
+
this.unanswered += 1;
|
|
827
|
+
if (this.streakStartedAt === null) this.streakStartedAt = now;
|
|
828
|
+
const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
|
|
829
|
+
if (this.unanswered >= 20 && silentMs >= 3e4) return {
|
|
830
|
+
unanswered: this.unanswered,
|
|
831
|
+
silentMs
|
|
832
|
+
};
|
|
833
|
+
return null;
|
|
834
|
+
}
|
|
835
|
+
};
|
|
836
|
+
//#endregion
|
|
572
837
|
//#region src/detection-pipeline/engine/shared-inference-pool.ts
|
|
573
838
|
/**
|
|
574
839
|
* SharedInferencePool — TypeScript wrapper for inference_pool.py.
|
|
@@ -610,6 +875,27 @@ var RAW_FMT_CODE = {
|
|
|
610
875
|
gray: 2
|
|
611
876
|
};
|
|
612
877
|
/**
|
|
878
|
+
* A load the worker answered with a machine-readable failure class. The
|
|
879
|
+
* provider tells a `compile-timeout` (counted against THAT MODEL on the
|
|
880
|
+
* device) from an ordinary failure by `reason`, never by parsing text.
|
|
881
|
+
*/
|
|
882
|
+
var PoolModelLoadError = class extends Error {
|
|
883
|
+
reason;
|
|
884
|
+
/** The model's file stem, as the worker named it. */
|
|
885
|
+
model;
|
|
886
|
+
constructor(message, reason, model) {
|
|
887
|
+
super(message);
|
|
888
|
+
this.name = "PoolModelLoadError";
|
|
889
|
+
this.reason = reason;
|
|
890
|
+
this.model = model;
|
|
891
|
+
}
|
|
892
|
+
};
|
|
893
|
+
/**
|
|
894
|
+
* Request id the worker uses for UNSOLICITED events — a reply to no request
|
|
895
|
+
* (`compile-finished-late`). Never allocated to a request.
|
|
896
|
+
*/
|
|
897
|
+
var WORKER_EVENT_REQ_ID = 4294967295;
|
|
898
|
+
/**
|
|
613
899
|
* Per-inference-request reply timeout (ms). Turns a wedged request (worker
|
|
614
900
|
* alive but no reply) into a rejection so every caller settles — runtime frame
|
|
615
901
|
* dispatch drops the frame; the benchmark full-tree run rejects the wedged
|
|
@@ -618,6 +904,22 @@ var RAW_FMT_CODE = {
|
|
|
618
904
|
* large frame) never trips; override via env for constrained hardware.
|
|
619
905
|
*/
|
|
620
906
|
var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
|
|
907
|
+
/** Commands that compile a model — the only ones that get the load deadline. */
|
|
908
|
+
var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
|
|
909
|
+
/**
|
|
910
|
+
* Deadline of the liveness probe (`mem_stats`) sent after a command times out.
|
|
911
|
+
* The worker answers `mem_stats` on its event loop, never behind a compile
|
|
912
|
+
* (D653), so a healthy worker replies in milliseconds even mid-compile; ten
|
|
913
|
+
* seconds only has to outlast a GC pause or a burst of inference replies.
|
|
914
|
+
*/
|
|
915
|
+
var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
|
|
916
|
+
/**
|
|
917
|
+
* How long a poisoned worker may keep its in-flight LIVE requests before it is
|
|
918
|
+
* killed anyway. A poisoned worker takes no new work, and every live request
|
|
919
|
+
* carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
|
|
920
|
+
* by then; the second of slack keeps the backstop from racing the last reply.
|
|
921
|
+
*/
|
|
922
|
+
var POOL_POISON_DRAIN_SLACK_MS = 1e3;
|
|
621
923
|
/**
|
|
622
924
|
* Max inference requests outstanding to ONE worker before new ones are SHED.
|
|
623
925
|
*
|
|
@@ -805,6 +1107,27 @@ var PoolWorker = class {
|
|
|
805
1107
|
wireVersion = 1;
|
|
806
1108
|
nextRequestId = 1;
|
|
807
1109
|
ready = false;
|
|
1110
|
+
/** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
|
|
1111
|
+
* a poisoned worker is recycled, not revived. */
|
|
1112
|
+
poisonVerdict = null;
|
|
1113
|
+
/** The kill of a poisoned worker has been started. */
|
|
1114
|
+
recycling = false;
|
|
1115
|
+
/** Kills a poisoned worker whose drain did not finish in time. */
|
|
1116
|
+
recycleBackstop = null;
|
|
1117
|
+
/** A liveness probe is in flight — one at a time, never a probe storm. */
|
|
1118
|
+
probeInFlight = false;
|
|
1119
|
+
/** The child has exited (any cause). */
|
|
1120
|
+
exited = false;
|
|
1121
|
+
/** A compile past its soft bound, still running in the worker (D653). */
|
|
1122
|
+
abandonedCompile = null;
|
|
1123
|
+
/** Says when live inference has produced nothing for too long (D653). */
|
|
1124
|
+
inferStall = new InferStallDetector(Date.now());
|
|
1125
|
+
/**
|
|
1126
|
+
* A load's HOST deadline expired with no reply since the last reply of any
|
|
1127
|
+
* kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
|
|
1128
|
+
* is over for this worker until something — anything — comes back.
|
|
1129
|
+
*/
|
|
1130
|
+
loadDeadlineMissed = false;
|
|
808
1131
|
log;
|
|
809
1132
|
opts;
|
|
810
1133
|
constructor(opts) {
|
|
@@ -829,6 +1152,24 @@ var PoolWorker = class {
|
|
|
829
1152
|
isReady() {
|
|
830
1153
|
return this.ready;
|
|
831
1154
|
}
|
|
1155
|
+
/** Why this worker was recycled, typed for the provider's budget, or `null`. */
|
|
1156
|
+
getDeathCause() {
|
|
1157
|
+
const v = this.poisonVerdict;
|
|
1158
|
+
const message = this.getPoisonDescription();
|
|
1159
|
+
if (v === null || message === null) return null;
|
|
1160
|
+
return {
|
|
1161
|
+
reason: v.reason,
|
|
1162
|
+
chargesDeviceBudget: chargesDeviceBudget(v.reason),
|
|
1163
|
+
message
|
|
1164
|
+
};
|
|
1165
|
+
}
|
|
1166
|
+
/** Why this live worker was declared unusable, or `null` (D653). */
|
|
1167
|
+
getPoisonDescription() {
|
|
1168
|
+
const v = this.poisonVerdict;
|
|
1169
|
+
if (v === null) return null;
|
|
1170
|
+
const target = v.command.model !== null ? ` of ${v.command.model}` : "";
|
|
1171
|
+
return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
|
|
1172
|
+
}
|
|
832
1173
|
async initialize(initialModels) {
|
|
833
1174
|
this.process = (0, node_child_process.spawn)(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
|
|
834
1175
|
"pipe",
|
|
@@ -852,7 +1193,11 @@ var PoolWorker = class {
|
|
|
852
1193
|
const spawnedProcess = this.process;
|
|
853
1194
|
spawnedProcess.on("exit", (code, signal) => {
|
|
854
1195
|
this.ready = false;
|
|
1196
|
+
this.exited = true;
|
|
1197
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1198
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
855
1199
|
if (this.process !== spawnedProcess) return;
|
|
1200
|
+
const poisonedBy = this.getPoisonDescription();
|
|
856
1201
|
this.log.error("Worker process exited", { meta: {
|
|
857
1202
|
worker: this.opts.workerLabel,
|
|
858
1203
|
pid: spawnedProcess.pid ?? null,
|
|
@@ -860,9 +1205,10 @@ var PoolWorker = class {
|
|
|
860
1205
|
device: this.opts.device ?? "default",
|
|
861
1206
|
code,
|
|
862
1207
|
signal,
|
|
863
|
-
inFlight: this.pending.size
|
|
1208
|
+
inFlight: this.pending.size,
|
|
1209
|
+
...poisonedBy !== null ? { poisonedBy } : {}
|
|
864
1210
|
} });
|
|
865
|
-
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
|
|
1211
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
|
|
866
1212
|
});
|
|
867
1213
|
this.process.stdout.on("data", (chunk) => {
|
|
868
1214
|
const t0 = Date.now();
|
|
@@ -876,6 +1222,7 @@ var PoolWorker = class {
|
|
|
876
1222
|
runtime: this.opts.poolRuntime,
|
|
877
1223
|
concurrency: this.opts.concurrency,
|
|
878
1224
|
protocolVersion: 2,
|
|
1225
|
+
modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
|
|
879
1226
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
880
1227
|
};
|
|
881
1228
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -898,6 +1245,7 @@ var PoolWorker = class {
|
|
|
898
1245
|
clearTimeout(timeout);
|
|
899
1246
|
if (result["status"] === "ready") {
|
|
900
1247
|
this.ready = true;
|
|
1248
|
+
this.inferStall.noteResult(Date.now());
|
|
901
1249
|
const loadedCount = result["models"];
|
|
902
1250
|
const startupMs = result["startupMs"];
|
|
903
1251
|
const workers = result["workers"] ?? 1;
|
|
@@ -920,7 +1268,7 @@ var PoolWorker = class {
|
|
|
920
1268
|
async infer(modelByte, jpeg, deviceId) {
|
|
921
1269
|
this.ensureReady();
|
|
922
1270
|
const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
|
|
923
|
-
return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
|
|
1271
|
+
return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
|
|
924
1272
|
}
|
|
925
1273
|
async inferRaw(modelByte, raw, width, height, format, deviceId) {
|
|
926
1274
|
this.ensureReady();
|
|
@@ -973,12 +1321,18 @@ var PoolWorker = class {
|
|
|
973
1321
|
const payload = Buffer.allocUnsafe(5);
|
|
974
1322
|
payload[0] = modelByte;
|
|
975
1323
|
payload.writeUInt32LE(frameId, 1);
|
|
976
|
-
return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
|
|
1324
|
+
return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
|
|
977
1325
|
}
|
|
978
1326
|
async sendCommand(cmd) {
|
|
979
1327
|
this.ensureReady();
|
|
1328
|
+
const command = describeCommand(cmd);
|
|
980
1329
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
981
|
-
|
|
1330
|
+
const sentAt = Date.now();
|
|
1331
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
|
|
1332
|
+
const raw = await this.dispatch(MSG_COMMAND, payload, { command });
|
|
1333
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
|
|
1334
|
+
this.inspectCommandReply(raw, command, sentAt);
|
|
1335
|
+
return raw;
|
|
982
1336
|
}
|
|
983
1337
|
/** Command whose reply shape is NOT the load/unload/replace/status envelope
|
|
984
1338
|
* (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
|
|
@@ -986,13 +1340,15 @@ var PoolWorker = class {
|
|
|
986
1340
|
async sendRawCommand(cmd) {
|
|
987
1341
|
this.ensureReady();
|
|
988
1342
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
989
|
-
return this.dispatch(MSG_COMMAND, payload);
|
|
1343
|
+
return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
|
|
990
1344
|
}
|
|
991
1345
|
async dispose() {
|
|
992
1346
|
const proc = this.process;
|
|
993
1347
|
if (!proc) return;
|
|
994
1348
|
this.process = null;
|
|
995
1349
|
this.ready = false;
|
|
1350
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1351
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
996
1352
|
await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
|
|
997
1353
|
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
|
|
998
1354
|
}
|
|
@@ -1035,8 +1391,11 @@ var PoolWorker = class {
|
|
|
1035
1391
|
}
|
|
1036
1392
|
/** Live inference gets seconds; commands and model loads keep the long
|
|
1037
1393
|
* timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
|
|
1038
|
-
deadlineFor(msgType) {
|
|
1039
|
-
|
|
1394
|
+
deadlineFor(msgType, opts = {}) {
|
|
1395
|
+
if (opts.timeoutMs !== void 0) return opts.timeoutMs;
|
|
1396
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
|
|
1397
|
+
if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
|
|
1398
|
+
return POOL_INFER_TIMEOUT_MS;
|
|
1040
1399
|
}
|
|
1041
1400
|
/**
|
|
1042
1401
|
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
@@ -1053,16 +1412,18 @@ var PoolWorker = class {
|
|
|
1053
1412
|
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1054
1413
|
return stamp;
|
|
1055
1414
|
}
|
|
1056
|
-
dispatch(msgType, payload,
|
|
1057
|
-
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1415
|
+
dispatch(msgType, payload, opts = {}) {
|
|
1416
|
+
const shed = this.shedIfSaturated(msgType, opts.deviceId);
|
|
1058
1417
|
if (shed) return Promise.resolve(shed);
|
|
1059
1418
|
const reqId = this.allocRequestId();
|
|
1060
1419
|
return new Promise((resolve, reject) => {
|
|
1061
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1420
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
|
|
1062
1421
|
this.pending.set(reqId, {
|
|
1063
1422
|
resolve,
|
|
1064
1423
|
reject,
|
|
1065
|
-
timer
|
|
1424
|
+
timer,
|
|
1425
|
+
...opts.command !== void 0 ? { command: opts.command } : {},
|
|
1426
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1066
1427
|
});
|
|
1067
1428
|
try {
|
|
1068
1429
|
const stamp = this.deadlineStamp(msgType);
|
|
@@ -1093,8 +1454,9 @@ var PoolWorker = class {
|
|
|
1093
1454
|
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1094
1455
|
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1095
1456
|
*/
|
|
1096
|
-
armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1097
|
-
const
|
|
1457
|
+
armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
|
|
1458
|
+
const { deviceId, command } = opts;
|
|
1459
|
+
const timeoutMs = this.deadlineFor(msgType, opts);
|
|
1098
1460
|
const timer = setTimeout(() => {
|
|
1099
1461
|
if (this.pending.delete(reqId)) {
|
|
1100
1462
|
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
@@ -1118,6 +1480,7 @@ var PoolWorker = class {
|
|
|
1118
1480
|
dropped: true,
|
|
1119
1481
|
shedReason: "deadline-expired"
|
|
1120
1482
|
});
|
|
1483
|
+
this.noteInferUnanswered();
|
|
1121
1484
|
return;
|
|
1122
1485
|
}
|
|
1123
1486
|
this.timedOutCount++;
|
|
@@ -1130,10 +1493,20 @@ var PoolWorker = class {
|
|
|
1130
1493
|
device: this.opts.device ?? "default",
|
|
1131
1494
|
inFlight: this.pending.size,
|
|
1132
1495
|
reqId,
|
|
1133
|
-
timeoutMs
|
|
1496
|
+
timeoutMs,
|
|
1497
|
+
...command !== void 0 ? {
|
|
1498
|
+
command: command.cmd,
|
|
1499
|
+
modelIndex: command.modelIndex,
|
|
1500
|
+
model: command.model
|
|
1501
|
+
} : {}
|
|
1134
1502
|
}
|
|
1135
1503
|
});
|
|
1136
|
-
|
|
1504
|
+
const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
|
|
1505
|
+
if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
|
|
1506
|
+
this.loadDeadlineMissed = true;
|
|
1507
|
+
reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
|
|
1508
|
+
} else reject(new Error(message));
|
|
1509
|
+
if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
|
|
1137
1510
|
}
|
|
1138
1511
|
}, timeoutMs);
|
|
1139
1512
|
timer.unref?.();
|
|
@@ -1144,11 +1517,12 @@ var PoolWorker = class {
|
|
|
1144
1517
|
if (shed) return Promise.resolve(shed);
|
|
1145
1518
|
const reqId = this.allocRequestId();
|
|
1146
1519
|
return new Promise((resolve, reject) => {
|
|
1147
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1520
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
|
|
1148
1521
|
this.pending.set(reqId, {
|
|
1149
1522
|
resolve,
|
|
1150
1523
|
reject,
|
|
1151
|
-
timer
|
|
1524
|
+
timer,
|
|
1525
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1152
1526
|
});
|
|
1153
1527
|
try {
|
|
1154
1528
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
@@ -1170,10 +1544,10 @@ var PoolWorker = class {
|
|
|
1170
1544
|
}
|
|
1171
1545
|
allocRequestId() {
|
|
1172
1546
|
let id = this.nextRequestId;
|
|
1173
|
-
this.nextRequestId = id >=
|
|
1547
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1174
1548
|
while (this.pending.has(id)) {
|
|
1175
1549
|
id = this.nextRequestId;
|
|
1176
|
-
this.nextRequestId = id >=
|
|
1550
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1177
1551
|
}
|
|
1178
1552
|
return id;
|
|
1179
1553
|
}
|
|
@@ -1188,6 +1562,8 @@ var PoolWorker = class {
|
|
|
1188
1562
|
this.process.stdin.write(payload);
|
|
1189
1563
|
}
|
|
1190
1564
|
ensureReady() {
|
|
1565
|
+
const poisonedBy = this.getPoisonDescription();
|
|
1566
|
+
if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
|
|
1191
1567
|
if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
|
|
1192
1568
|
}
|
|
1193
1569
|
/** Time spent in `Buffer.concat` since the last report. */
|
|
@@ -1227,23 +1603,273 @@ var PoolWorker = class {
|
|
|
1227
1603
|
const reqId = this.receiveBuffer.readUInt32LE(4);
|
|
1228
1604
|
const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
|
|
1229
1605
|
this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
|
|
1606
|
+
if (reqId === WORKER_EVENT_REQ_ID) {
|
|
1607
|
+
this.handleWorkerEvent(jsonBytes);
|
|
1608
|
+
continue;
|
|
1609
|
+
}
|
|
1230
1610
|
const entry = this.pending.get(reqId);
|
|
1231
1611
|
if (!entry) {
|
|
1232
1612
|
this.log.warn("Response for unknown request id", { meta: {
|
|
1233
1613
|
worker: this.opts.workerLabel,
|
|
1234
1614
|
reqId
|
|
1235
1615
|
} });
|
|
1616
|
+
this.noteLateReply(jsonBytes);
|
|
1236
1617
|
continue;
|
|
1237
1618
|
}
|
|
1238
1619
|
this.pending.delete(reqId);
|
|
1239
1620
|
if (entry.timer) clearTimeout(entry.timer);
|
|
1240
1621
|
try {
|
|
1241
1622
|
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1623
|
+
if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
|
|
1624
|
+
else this.inferStall.noteResult(Date.now());
|
|
1242
1625
|
entry.resolve(parsed);
|
|
1243
1626
|
} catch (err) {
|
|
1244
1627
|
entry.reject(err instanceof Error ? err : new Error(String(err)));
|
|
1245
1628
|
}
|
|
1246
1629
|
}
|
|
1630
|
+
this.noteAnyReply();
|
|
1631
|
+
if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
|
|
1632
|
+
}
|
|
1633
|
+
/**
|
|
1634
|
+
* Something came back from the worker: its loop was alive just now, so the
|
|
1635
|
+
* missed-deadline verdict is lifted.
|
|
1636
|
+
*/
|
|
1637
|
+
noteAnyReply() {
|
|
1638
|
+
this.loadDeadlineMissed = false;
|
|
1639
|
+
}
|
|
1640
|
+
/**
|
|
1641
|
+
* The invariant the load deadline depends on (D653 § 2): loads into one
|
|
1642
|
+
* pool are serialised, so at most ONE load is outstanding per worker. The
|
|
1643
|
+
* worker runs model commands in order and starts a load's soft bound only
|
|
1644
|
+
* when it begins, while the host's deadline for it runs from the SEND. A
|
|
1645
|
+
* load queued behind another could therefore see its host deadline fire
|
|
1646
|
+
* before the worker's named answer. Nothing in the provider issues that; if
|
|
1647
|
+
* anything ever does, this says so.
|
|
1648
|
+
*/
|
|
1649
|
+
warnIfLoadQueued(next) {
|
|
1650
|
+
const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
|
|
1651
|
+
if (ahead.length === 0) return;
|
|
1652
|
+
this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
|
|
1653
|
+
worker: this.opts.workerLabel,
|
|
1654
|
+
pid: this.getPid(),
|
|
1655
|
+
runtime: this.opts.poolRuntime,
|
|
1656
|
+
device: this.opts.device ?? "default",
|
|
1657
|
+
model: next.model,
|
|
1658
|
+
queuedBehind: ahead.map((c) => c.model)
|
|
1659
|
+
} });
|
|
1660
|
+
}
|
|
1661
|
+
hasPendingLoad() {
|
|
1662
|
+
for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
|
|
1663
|
+
return false;
|
|
1664
|
+
}
|
|
1665
|
+
/**
|
|
1666
|
+
* A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
|
|
1667
|
+
* or that an earlier one still has (`worker-poisoned`). The worker is not
|
|
1668
|
+
* recycled for it (fix round 1): the compile keeps running so a slow but
|
|
1669
|
+
* finite one writes its cache, the loaded models keep serving, and the worker
|
|
1670
|
+
* starts no new load. Only a compile still running at the HARD bound costs
|
|
1671
|
+
* the process.
|
|
1672
|
+
*/
|
|
1673
|
+
inspectCommandReply(raw, command, sentAt) {
|
|
1674
|
+
const reason = raw["reason"];
|
|
1675
|
+
if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
|
|
1676
|
+
if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
|
|
1677
|
+
const model = raw["modelId"];
|
|
1678
|
+
const abandoned = {
|
|
1679
|
+
...command,
|
|
1680
|
+
model: typeof model === "string" ? model : command.model
|
|
1681
|
+
};
|
|
1682
|
+
const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
|
|
1683
|
+
const reported = raw["compileElapsedMs"];
|
|
1684
|
+
const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
|
|
1685
|
+
const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
|
|
1686
|
+
hardTimer.unref?.();
|
|
1687
|
+
this.abandonedCompile = {
|
|
1688
|
+
command: abandoned,
|
|
1689
|
+
since: Date.now(),
|
|
1690
|
+
hardTimer
|
|
1691
|
+
};
|
|
1692
|
+
this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
|
|
1693
|
+
worker: this.opts.workerLabel,
|
|
1694
|
+
pid: this.getPid(),
|
|
1695
|
+
runtime: this.opts.poolRuntime,
|
|
1696
|
+
device: this.opts.device ?? "default",
|
|
1697
|
+
model: abandoned.model,
|
|
1698
|
+
modelIndex: abandoned.modelIndex,
|
|
1699
|
+
softMs: bounds.softMs,
|
|
1700
|
+
hardMs: bounds.hardMs,
|
|
1701
|
+
loads: "refused until it returns"
|
|
1702
|
+
} });
|
|
1703
|
+
}
|
|
1704
|
+
/** An unsolicited worker event (`compile-finished-late`). */
|
|
1705
|
+
handleWorkerEvent(jsonBytes) {
|
|
1706
|
+
let event;
|
|
1707
|
+
try {
|
|
1708
|
+
event = JSON.parse(jsonBytes.toString("utf8"));
|
|
1709
|
+
} catch {
|
|
1710
|
+
return;
|
|
1711
|
+
}
|
|
1712
|
+
if (event["event"] !== "compile-finished-late") return;
|
|
1713
|
+
const abandoned = this.abandonedCompile;
|
|
1714
|
+
if (abandoned !== null) clearTimeout(abandoned.hardTimer);
|
|
1715
|
+
this.abandonedCompile = null;
|
|
1716
|
+
this.inferStall.noteResult(Date.now());
|
|
1717
|
+
const ok = event["ok"] === true;
|
|
1718
|
+
const meta = {
|
|
1719
|
+
worker: this.opts.workerLabel,
|
|
1720
|
+
pid: this.getPid(),
|
|
1721
|
+
runtime: this.opts.poolRuntime,
|
|
1722
|
+
device: this.opts.device ?? "default",
|
|
1723
|
+
model: typeof event["modelId"] === "string" ? event["modelId"] : null,
|
|
1724
|
+
elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
|
|
1725
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1726
|
+
};
|
|
1727
|
+
if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
|
|
1728
|
+
else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
|
|
1729
|
+
this.opts.onCompileFinishedLate?.({
|
|
1730
|
+
model: meta.model,
|
|
1731
|
+
ok,
|
|
1732
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1733
|
+
});
|
|
1734
|
+
}
|
|
1735
|
+
/**
|
|
1736
|
+
* The GIL grace (D653 round 2): may a silent loop be a compile holding the
|
|
1737
|
+
* GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
|
|
1738
|
+
* on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
|
|
1739
|
+
*
|
|
1740
|
+
* NOT while a compile runs past its soft bound: the worker REPLIED
|
|
1741
|
+
* `compile-timeout`, so its loop is proven alive, and an inference hang
|
|
1742
|
+
* behind that compile — the 2026-09-26 incident exactly — must be seen on
|
|
1743
|
+
* the normal rule, not ~600-900 s later at the hard bound.
|
|
1744
|
+
*/
|
|
1745
|
+
gilGraceActive() {
|
|
1746
|
+
if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
|
|
1747
|
+
if (this.loadDeadlineMissed) return false;
|
|
1748
|
+
return this.hasPendingLoad();
|
|
1749
|
+
}
|
|
1750
|
+
/**
|
|
1751
|
+
* The reason a silent worker is recycled with: a freeze that began with a
|
|
1752
|
+
* load missing its deadline is that compile's doing (`compile-hung`, charged
|
|
1753
|
+
* to the model, not the device); anything else is the device's.
|
|
1754
|
+
*/
|
|
1755
|
+
silenceReason(fallback) {
|
|
1756
|
+
return this.loadDeadlineMissed ? "compile-hung" : fallback;
|
|
1757
|
+
}
|
|
1758
|
+
/** A live request ended with no result: judge the executor (D653, item 4). */
|
|
1759
|
+
noteInferUnanswered() {
|
|
1760
|
+
const stall = this.inferStall.noteUnanswered(Date.now());
|
|
1761
|
+
if (stall === null || this.gilGraceActive()) return;
|
|
1762
|
+
this.poison(this.silenceReason("infer-unresponsive"), {
|
|
1763
|
+
cmd: "infer",
|
|
1764
|
+
modelIndex: null,
|
|
1765
|
+
model: null
|
|
1766
|
+
}, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
|
|
1767
|
+
}
|
|
1768
|
+
/** A reply whose request was already abandoned: a real result still proves the executor runs. */
|
|
1769
|
+
noteLateReply(jsonBytes) {
|
|
1770
|
+
try {
|
|
1771
|
+
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1772
|
+
if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
|
|
1773
|
+
} catch {}
|
|
1774
|
+
}
|
|
1775
|
+
/**
|
|
1776
|
+
* Ask a worker whose command just timed out whether its loop still answers.
|
|
1777
|
+
* `mem_stats` is served on the loop, never behind a compile, so a worker
|
|
1778
|
+
* that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
|
|
1779
|
+
* slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
|
|
1780
|
+
* `mem_stats` lost while the process looked alive.
|
|
1781
|
+
*/
|
|
1782
|
+
probeLiveness(timedOut) {
|
|
1783
|
+
if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
|
|
1784
|
+
this.probeInFlight = true;
|
|
1785
|
+
this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
|
|
1786
|
+
command: {
|
|
1787
|
+
cmd: "mem_stats",
|
|
1788
|
+
modelIndex: null,
|
|
1789
|
+
model: null
|
|
1790
|
+
},
|
|
1791
|
+
timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
|
|
1792
|
+
probe: true
|
|
1793
|
+
}).then(() => {
|
|
1794
|
+
this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
|
|
1795
|
+
worker: this.opts.workerLabel,
|
|
1796
|
+
pid: this.getPid(),
|
|
1797
|
+
runtime: this.opts.poolRuntime,
|
|
1798
|
+
device: this.opts.device ?? "default",
|
|
1799
|
+
command: timedOut.cmd,
|
|
1800
|
+
modelIndex: timedOut.modelIndex,
|
|
1801
|
+
model: timedOut.model
|
|
1802
|
+
} });
|
|
1803
|
+
}, (err) => {
|
|
1804
|
+
if (this.gilGraceActive()) {
|
|
1805
|
+
this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
|
|
1806
|
+
worker: this.opts.workerLabel,
|
|
1807
|
+
pid: this.getPid(),
|
|
1808
|
+
runtime: this.opts.poolRuntime,
|
|
1809
|
+
device: this.opts.device ?? "default",
|
|
1810
|
+
command: timedOut.cmd,
|
|
1811
|
+
model: timedOut.model
|
|
1812
|
+
} });
|
|
1813
|
+
return;
|
|
1814
|
+
}
|
|
1815
|
+
this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1816
|
+
}).finally(() => {
|
|
1817
|
+
this.probeInFlight = false;
|
|
1818
|
+
});
|
|
1819
|
+
}
|
|
1820
|
+
/**
|
|
1821
|
+
* Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
|
|
1822
|
+
* and recycle it once it has drained.
|
|
1823
|
+
*
|
|
1824
|
+
* `ready = false` is what hands the pool back to the provider: its next
|
|
1825
|
+
* dispatch finds the factory not ready and condemns it under the per-device
|
|
1826
|
+
* restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
|
|
1827
|
+
* excludes) — the same path a crashed worker takes, except that a
|
|
1828
|
+
* `compile-hung` death is not charged to the device (see PoolPoisonReason).
|
|
1829
|
+
* An unresponsive worker is killed at once (it will answer nothing it
|
|
1830
|
+
* holds); a compile-hung one keeps serving its loaded models until its
|
|
1831
|
+
* in-flight requests are answered, bounded by the live deadline.
|
|
1832
|
+
*/
|
|
1833
|
+
poison(reason, command, detail) {
|
|
1834
|
+
if (this.poisonVerdict !== null || this.exited || this.process === null) return;
|
|
1835
|
+
this.poisonVerdict = {
|
|
1836
|
+
reason,
|
|
1837
|
+
command,
|
|
1838
|
+
detail,
|
|
1839
|
+
pid: this.getPid()
|
|
1840
|
+
};
|
|
1841
|
+
this.ready = false;
|
|
1842
|
+
const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
|
|
1843
|
+
this.log.error("pool worker POISONED — recycling it", { meta: {
|
|
1844
|
+
worker: this.opts.workerLabel,
|
|
1845
|
+
pid: this.getPid(),
|
|
1846
|
+
runtime: this.opts.poolRuntime,
|
|
1847
|
+
device: this.opts.device ?? "default",
|
|
1848
|
+
reason,
|
|
1849
|
+
command: command.cmd,
|
|
1850
|
+
modelIndex: command.modelIndex,
|
|
1851
|
+
model: command.model,
|
|
1852
|
+
detail,
|
|
1853
|
+
inFlight: this.pending.size,
|
|
1854
|
+
inFlightCommands
|
|
1855
|
+
} });
|
|
1856
|
+
if (reason !== "compile-hung" || this.pending.size === 0) {
|
|
1857
|
+
this.recycle();
|
|
1858
|
+
return;
|
|
1859
|
+
}
|
|
1860
|
+
this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
|
|
1861
|
+
this.recycleBackstop.unref?.();
|
|
1862
|
+
}
|
|
1863
|
+
/**
|
|
1864
|
+
* Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
|
|
1865
|
+
* reported and rejects whatever it still held — a deliberate dispose would
|
|
1866
|
+
* silence both, and this is not one.
|
|
1867
|
+
*/
|
|
1868
|
+
recycle() {
|
|
1869
|
+
if (this.recycling || this.exited || this.process === null) return;
|
|
1870
|
+
this.recycling = true;
|
|
1871
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1872
|
+
terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
|
|
1247
1873
|
}
|
|
1248
1874
|
rejectAll(err) {
|
|
1249
1875
|
const entries = [...this.pending.values()];
|
|
@@ -1290,8 +1916,10 @@ var SharedInferencePool = class {
|
|
|
1290
1916
|
this.tuning = options.tuning ?? null;
|
|
1291
1917
|
this.numWorkers = Math.max(1, options.numWorkers ?? 1);
|
|
1292
1918
|
this.device = options.device;
|
|
1919
|
+
this.onCompileFinishedLate = options.onCompileFinishedLate;
|
|
1293
1920
|
}
|
|
1294
1921
|
device;
|
|
1922
|
+
onCompileFinishedLate;
|
|
1295
1923
|
/** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
|
|
1296
1924
|
/** Summed backlog and shed count across the pool's workers. A rising
|
|
1297
1925
|
* `inFlight` with a rising `shed` is a worker falling behind; a rising
|
|
@@ -1330,7 +1958,8 @@ var SharedInferencePool = class {
|
|
|
1330
1958
|
tuning: this.tuning,
|
|
1331
1959
|
logger: this.log,
|
|
1332
1960
|
workerLabel: `w${i}`,
|
|
1333
|
-
...this.device ? { device: this.device } : {}
|
|
1961
|
+
...this.device ? { device: this.device } : {},
|
|
1962
|
+
...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
|
|
1334
1963
|
}));
|
|
1335
1964
|
const t0 = performance.now();
|
|
1336
1965
|
const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
|
|
@@ -1425,7 +2054,7 @@ var SharedInferencePool = class {
|
|
|
1425
2054
|
index,
|
|
1426
2055
|
config: serializeModelConfig(config)
|
|
1427
2056
|
})));
|
|
1428
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
2057
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1429
2058
|
if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
|
|
1430
2059
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1431
2060
|
}
|
|
@@ -1461,7 +2090,7 @@ var SharedInferencePool = class {
|
|
|
1461
2090
|
index,
|
|
1462
2091
|
config: serializeModelConfig(config)
|
|
1463
2092
|
})));
|
|
1464
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
2093
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1465
2094
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1466
2095
|
}
|
|
1467
2096
|
/**
|
|
@@ -1511,9 +2140,35 @@ var SharedInferencePool = class {
|
|
|
1511
2140
|
allocateIndex() {
|
|
1512
2141
|
return this.nextFreeIndex++;
|
|
1513
2142
|
}
|
|
2143
|
+
/**
|
|
2144
|
+
* Give back an index whose load FAILED, so the retry reuses it. Only the most
|
|
2145
|
+
* recent allocation can be returned. Loads into one pool are serialised by
|
|
2146
|
+
* the provider, so a failed load is normally the latest one; anything else
|
|
2147
|
+
* is left allocated rather than risk handing out a live slot twice.
|
|
2148
|
+
*/
|
|
2149
|
+
releaseIndex(index) {
|
|
2150
|
+
if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
|
|
2151
|
+
}
|
|
1514
2152
|
isReady() {
|
|
1515
2153
|
return this.workers.length > 0 && this.workers.every((w) => w.isReady());
|
|
1516
2154
|
}
|
|
2155
|
+
/**
|
|
2156
|
+
* Why a worker of this pool was declared unusable while alive (D653), or
|
|
2157
|
+
* `null`. The provider charges the restart budget with THIS — or, for a
|
|
2158
|
+
* `compile-hung` death, does not charge the device at all — so the
|
|
2159
|
+
* `inference device FAILED` line names the cause instead of "pool worker is
|
|
2160
|
+
* not ready".
|
|
2161
|
+
*/
|
|
2162
|
+
getDeathCause() {
|
|
2163
|
+
for (const w of this.workers) {
|
|
2164
|
+
const cause = w.getDeathCause();
|
|
2165
|
+
if (cause !== null) return {
|
|
2166
|
+
...cause,
|
|
2167
|
+
message: `worker ${cause.message}`
|
|
2168
|
+
};
|
|
2169
|
+
}
|
|
2170
|
+
return null;
|
|
2171
|
+
}
|
|
1517
2172
|
async dispose() {
|
|
1518
2173
|
await Promise.all(this.workers.map((w) => w.dispose()));
|
|
1519
2174
|
this.workers.length = 0;
|
|
@@ -1578,6 +2233,23 @@ var SharedInferencePool = class {
|
|
|
1578
2233
|
return found;
|
|
1579
2234
|
}
|
|
1580
2235
|
};
|
|
2236
|
+
/** A failed command's reply as one message: its reason class first, when it has one. */
|
|
2237
|
+
function describeCommandFailure(resp) {
|
|
2238
|
+
const error = resp.error ?? "unknown";
|
|
2239
|
+
return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
|
|
2240
|
+
}
|
|
2241
|
+
/** Name a command for the lines that must say which one hung (D653). */
|
|
2242
|
+
function describeCommand(cmd) {
|
|
2243
|
+
const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
|
|
2244
|
+
const index = cmd["index"];
|
|
2245
|
+
const config = cmd["config"];
|
|
2246
|
+
const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
|
|
2247
|
+
return {
|
|
2248
|
+
cmd: name,
|
|
2249
|
+
modelIndex: typeof index === "number" ? index : null,
|
|
2250
|
+
model: typeof modelPath === "string" && modelPath.length > 0 ? node_path.basename(modelPath, node_path.extname(modelPath)) : null
|
|
2251
|
+
};
|
|
2252
|
+
}
|
|
1581
2253
|
function serializeModelConfig(config) {
|
|
1582
2254
|
const result = {
|
|
1583
2255
|
path: config.path,
|
|
@@ -1649,7 +2321,8 @@ function collectFramePlaneSteps(steps, isDetailStep) {
|
|
|
1649
2321
|
}
|
|
1650
2322
|
/**
|
|
1651
2323
|
* Immutably remove from a frame-plane dispatch tree every enabled DETAIL
|
|
1652
|
-
* subtree (at any depth below the top level)
|
|
2324
|
+
* subtree (at any depth below the top level) that `isUnrunnable` — judged
|
|
2325
|
+
* WHOLE at its top, because its children ride its call (D657).
|
|
1653
2326
|
* Those steps can never load into the dispatch device's pool — leaving them
|
|
1654
2327
|
* in the tree makes `ensureModelsForSteps` fail the whole call on a format
|
|
1655
2328
|
* build that does not exist. They are NOT dropped work: each runs later as
|
|
@@ -1661,6 +2334,11 @@ function collectFramePlaneSteps(steps, isDetailStep) {
|
|
|
1661
2334
|
*/
|
|
1662
2335
|
function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
1663
2336
|
const prunedAddonIds = [];
|
|
2337
|
+
const removedAddonIds = [];
|
|
2338
|
+
const collect = (step) => {
|
|
2339
|
+
removedAddonIds.push(step.addonId);
|
|
2340
|
+
for (const child of step.children ?? []) collect(child);
|
|
2341
|
+
};
|
|
1664
2342
|
const walk = (nodes, topLevel) => {
|
|
1665
2343
|
const kept = [];
|
|
1666
2344
|
for (const step of nodes) {
|
|
@@ -1668,8 +2346,9 @@ function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
|
1668
2346
|
kept.push(step);
|
|
1669
2347
|
continue;
|
|
1670
2348
|
}
|
|
1671
|
-
if (!topLevel && isDetailStep(step.addonId) && isUnrunnable(step
|
|
2349
|
+
if (!topLevel && isDetailStep(step.addonId) && isUnrunnable(step)) {
|
|
1672
2350
|
prunedAddonIds.push(step.addonId);
|
|
2351
|
+
collect(step);
|
|
1673
2352
|
continue;
|
|
1674
2353
|
}
|
|
1675
2354
|
kept.push(step.children?.length ? {
|
|
@@ -1681,7 +2360,8 @@ function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
|
1681
2360
|
};
|
|
1682
2361
|
return {
|
|
1683
2362
|
steps: walk(steps, true),
|
|
1684
|
-
prunedAddonIds
|
|
2363
|
+
prunedAddonIds,
|
|
2364
|
+
removedAddonIds
|
|
1685
2365
|
};
|
|
1686
2366
|
}
|
|
1687
2367
|
//#endregion
|
|
@@ -1865,6 +2545,28 @@ function poolDecodeSettingsEqual(a, b) {
|
|
|
1865
2545
|
}
|
|
1866
2546
|
//#endregion
|
|
1867
2547
|
//#region src/detection-pipeline/engine/pipeline-model-manager.ts
|
|
2548
|
+
/**
|
|
2549
|
+
* A (step, model) load the pool rejected — carrying the pool's failure class
|
|
2550
|
+
* (`compile-timeout`, …) so the provider can count a timeout against THAT
|
|
2551
|
+
* model without parsing a message (D653).
|
|
2552
|
+
*/
|
|
2553
|
+
var StepVariantLoadError = class extends Error {
|
|
2554
|
+
stepId;
|
|
2555
|
+
modelId;
|
|
2556
|
+
poolIndex;
|
|
2557
|
+
reason;
|
|
2558
|
+
/** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
|
|
2559
|
+
poolModel;
|
|
2560
|
+
constructor(stepId, modelId, poolIndex, cause) {
|
|
2561
|
+
super(cause instanceof Error ? cause.message : String(cause));
|
|
2562
|
+
this.name = "StepVariantLoadError";
|
|
2563
|
+
this.stepId = stepId;
|
|
2564
|
+
this.modelId = modelId;
|
|
2565
|
+
this.poolIndex = poolIndex;
|
|
2566
|
+
this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
|
|
2567
|
+
this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
|
|
2568
|
+
}
|
|
2569
|
+
};
|
|
1868
2570
|
var PipelineModelManager = class {
|
|
1869
2571
|
pool;
|
|
1870
2572
|
source;
|
|
@@ -1876,11 +2578,20 @@ var PipelineModelManager = class {
|
|
|
1876
2578
|
lruClock = 0;
|
|
1877
2579
|
log;
|
|
1878
2580
|
maxModelsPerStep;
|
|
2581
|
+
deviceKey;
|
|
2582
|
+
/**
|
|
2583
|
+
* Loads issued and not yet answered, keyed `stepId::modelId`. A second
|
|
2584
|
+
* caller for the same pair JOINS the pending load instead of issuing its own
|
|
2585
|
+
* (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
|
|
2586
|
+
* compile, each allocating a pool index of its own.
|
|
2587
|
+
*/
|
|
2588
|
+
pendingLoads = /* @__PURE__ */ new Map();
|
|
1879
2589
|
constructor(pool, source, logger, options) {
|
|
1880
2590
|
this.pool = pool;
|
|
1881
2591
|
this.source = source;
|
|
1882
2592
|
this.log = logger;
|
|
1883
2593
|
this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
|
|
2594
|
+
this.deviceKey = options?.deviceKey ?? null;
|
|
1884
2595
|
}
|
|
1885
2596
|
/**
|
|
1886
2597
|
* Apply a new pipeline configuration — driven by the runtime config
|
|
@@ -1923,16 +2634,26 @@ var PipelineModelManager = class {
|
|
|
1923
2634
|
for (const { entry, step } of diff.unchanged) await this.reconcileDecode(entry, step.settings);
|
|
1924
2635
|
}
|
|
1925
2636
|
/**
|
|
1926
|
-
*
|
|
1927
|
-
*
|
|
1928
|
-
*
|
|
1929
|
-
*
|
|
1930
|
-
*
|
|
1931
|
-
|
|
1932
|
-
|
|
1933
|
-
|
|
2637
|
+
* The warm variant that runs `(stepId, modelId)` — its handle AND the model
|
|
2638
|
+
* loaded at that pool index, which is the only honest answer to "which model
|
|
2639
|
+
* produced this output" (D658).
|
|
2640
|
+
*
|
|
2641
|
+
* No active-variant fallback, deliberately. The step-only lookup this
|
|
2642
|
+
* replaces (`getHandle(stepId)`, deleted) answered the step's ACTIVE
|
|
2643
|
+
* variant — the first one loaded after boot — and a tree that paired that
|
|
2644
|
+
* handle with the step's REQUESTED id stamped SigLIP2 on MobileCLIP vectors
|
|
2645
|
+
* for twenty minutes on 2026-09-27. A pair that is not resident throws,
|
|
2646
|
+
* naming both.
|
|
2647
|
+
*/
|
|
2648
|
+
getVariant(stepId, modelId) {
|
|
2649
|
+
const entry = this.loaded.get(stepId)?.get(modelId);
|
|
2650
|
+
if (entry === void 0) throw new Error(`Step "${stepId}" has no warm variant for model "${modelId}" in the inference pool — refusing rather than running another variant`);
|
|
1934
2651
|
this.touch(entry);
|
|
1935
|
-
return
|
|
2652
|
+
return {
|
|
2653
|
+
engine: this.pool.getHandle(entry.poolIndex),
|
|
2654
|
+
modelId: entry.modelId,
|
|
2655
|
+
poolIndex: entry.poolIndex
|
|
2656
|
+
};
|
|
1936
2657
|
}
|
|
1937
2658
|
/** True iff the step has any model loaded. */
|
|
1938
2659
|
isLoaded(stepId) {
|
|
@@ -1963,13 +2684,14 @@ var PipelineModelManager = class {
|
|
|
1963
2684
|
return this.activeByStep.get(stepId);
|
|
1964
2685
|
}
|
|
1965
2686
|
/**
|
|
1966
|
-
* Pool index for
|
|
1967
|
-
*
|
|
1968
|
-
*
|
|
2687
|
+
* Pool index for exactly `(stepId, modelId)`, or `null` when that pair is not
|
|
2688
|
+
* warm. Used by the inference fast paths that bypass `getVariant` and call
|
|
2689
|
+
* `pool.inferBatch` directly. The model is REQUIRED: the active-variant
|
|
2690
|
+
* fallback this had timed a benchmark on whichever variant was active (D658).
|
|
1969
2691
|
*/
|
|
1970
2692
|
getPoolIndex(stepId, modelId) {
|
|
1971
|
-
const entry = this.
|
|
1972
|
-
if (
|
|
2693
|
+
const entry = this.loaded.get(stepId)?.get(modelId);
|
|
2694
|
+
if (entry === void 0) return null;
|
|
1973
2695
|
this.touch(entry);
|
|
1974
2696
|
return entry.poolIndex;
|
|
1975
2697
|
}
|
|
@@ -2030,6 +2752,23 @@ var PipelineModelManager = class {
|
|
|
2030
2752
|
await this.reconcileDecode(existing, settings);
|
|
2031
2753
|
return existing;
|
|
2032
2754
|
}
|
|
2755
|
+
const key = `${stepId}::${modelId}`;
|
|
2756
|
+
const pending = this.pendingLoads.get(key);
|
|
2757
|
+
if (pending) {
|
|
2758
|
+
const joined = await pending;
|
|
2759
|
+
await this.reconcileDecode(joined, settings);
|
|
2760
|
+
return joined;
|
|
2761
|
+
}
|
|
2762
|
+
const load = this.issueLoad(perStep, stepId, modelId, settings);
|
|
2763
|
+
this.pendingLoads.set(key, load);
|
|
2764
|
+
try {
|
|
2765
|
+
return await load;
|
|
2766
|
+
} finally {
|
|
2767
|
+
this.pendingLoads.delete(key);
|
|
2768
|
+
}
|
|
2769
|
+
}
|
|
2770
|
+
/** Evict if at capacity, then send ONE load command and record the result. */
|
|
2771
|
+
async issueLoad(perStep, stepId, modelId, settings) {
|
|
2033
2772
|
while (perStep.size >= this.maxModelsPerStep) {
|
|
2034
2773
|
const evicted = this.pickEvictionTarget(stepId);
|
|
2035
2774
|
if (!evicted) break;
|
|
@@ -2041,16 +2780,31 @@ var PipelineModelManager = class {
|
|
|
2041
2780
|
cap: this.maxModelsPerStep
|
|
2042
2781
|
} });
|
|
2043
2782
|
}
|
|
2044
|
-
const index = this.pool.allocateIndex();
|
|
2045
2783
|
const config = this.source.buildConfig(stepId, modelId, settings);
|
|
2046
2784
|
const decode = poolDecodeSettingsOf(config);
|
|
2785
|
+
const index = this.pool.allocateIndex();
|
|
2047
2786
|
this.log.info("Loading step variant", { meta: {
|
|
2048
2787
|
step: stepId,
|
|
2049
2788
|
modelId,
|
|
2050
2789
|
poolIndex: index,
|
|
2790
|
+
deviceKey: this.deviceKey,
|
|
2051
2791
|
...decode
|
|
2052
2792
|
} });
|
|
2053
|
-
|
|
2793
|
+
let loadMs;
|
|
2794
|
+
try {
|
|
2795
|
+
({loadMs} = await this.pool.loadModel(index, config));
|
|
2796
|
+
} catch (err) {
|
|
2797
|
+
this.pool.releaseIndex(index);
|
|
2798
|
+
this.log.error("Step variant load failed", { meta: {
|
|
2799
|
+
step: stepId,
|
|
2800
|
+
modelId,
|
|
2801
|
+
poolIndex: index,
|
|
2802
|
+
deviceKey: this.deviceKey,
|
|
2803
|
+
reason: err instanceof PoolModelLoadError ? err.reason : null,
|
|
2804
|
+
error: err instanceof Error ? err.message : String(err)
|
|
2805
|
+
} });
|
|
2806
|
+
throw new StepVariantLoadError(stepId, modelId, index, err);
|
|
2807
|
+
}
|
|
2054
2808
|
this.log.info("Step variant loaded", { meta: {
|
|
2055
2809
|
step: stepId,
|
|
2056
2810
|
modelId,
|
|
@@ -2157,18 +2911,6 @@ var PipelineModelManager = class {
|
|
|
2157
2911
|
await this.unloadEntry(evicted);
|
|
2158
2912
|
}
|
|
2159
2913
|
}
|
|
2160
|
-
resolve(stepId, modelId) {
|
|
2161
|
-
const perStep = this.loaded.get(stepId);
|
|
2162
|
-
if (!perStep) return null;
|
|
2163
|
-
const targetModelId = modelId ?? this.activeByStep.get(stepId);
|
|
2164
|
-
if (!targetModelId) return null;
|
|
2165
|
-
return perStep.get(targetModelId) ?? null;
|
|
2166
|
-
}
|
|
2167
|
-
resolveOrThrow(stepId, modelId) {
|
|
2168
|
-
const entry = this.resolve(stepId, modelId);
|
|
2169
|
-
if (!entry) throw new Error(`Step "${stepId}"${modelId ? ` (model "${modelId}")` : ""} is not loaded in the inference pool`);
|
|
2170
|
-
return entry;
|
|
2171
|
-
}
|
|
2172
2914
|
touch(entry) {
|
|
2173
2915
|
entry.lruTick = ++this.lruClock;
|
|
2174
2916
|
}
|
|
@@ -2364,16 +3106,15 @@ var EngineFactory = class {
|
|
|
2364
3106
|
await this.poolManager.applyConfig(steps);
|
|
2365
3107
|
}
|
|
2366
3108
|
/**
|
|
2367
|
-
*
|
|
2368
|
-
*
|
|
2369
|
-
*
|
|
2370
|
-
* paths picking a non-runtime model).
|
|
3109
|
+
* The resident variant for exactly `(stepId, modelId)`, with the model it
|
|
3110
|
+
* runs — never the step's active variant in its place (D658). The executable
|
|
3111
|
+
* tree is built from this, and a result's model stamp comes from it.
|
|
2371
3112
|
*
|
|
2372
|
-
* @throws if
|
|
3113
|
+
* @throws if that exact pair is not warm.
|
|
2373
3114
|
*/
|
|
2374
|
-
|
|
3115
|
+
getStepVariant(stepId, modelId) {
|
|
2375
3116
|
if (!this.poolManager) throw new Error("EngineFactory not initialized");
|
|
2376
|
-
return this.poolManager.
|
|
3117
|
+
return this.poolManager.getVariant(stepId, modelId);
|
|
2377
3118
|
}
|
|
2378
3119
|
/** Check if a step is loaded (any variant). */
|
|
2379
3120
|
isLoaded(stepId) {
|
|
@@ -2420,6 +3161,15 @@ var EngineFactory = class {
|
|
|
2420
3161
|
isReady() {
|
|
2421
3162
|
return this.pool?.isReady() ?? false;
|
|
2422
3163
|
}
|
|
3164
|
+
/**
|
|
3165
|
+
* Why this factory's pool stopped being usable while its process was alive —
|
|
3166
|
+
* a compile that never returned, or a worker that answered nothing (D653) —
|
|
3167
|
+
* or `null`. A pool that merely crashed says nothing here; its `Worker
|
|
3168
|
+
* process exited` line already names the signal.
|
|
3169
|
+
*/
|
|
3170
|
+
getDeathCause() {
|
|
3171
|
+
return this.pool?.getDeathCause() ?? null;
|
|
3172
|
+
}
|
|
2423
3173
|
/** Native pid of the underlying Python pool, if any. */
|
|
2424
3174
|
getPoolPid() {
|
|
2425
3175
|
return this.pool?.getPid() ?? null;
|
|
@@ -2468,7 +3218,7 @@ var EngineFactory = class {
|
|
|
2468
3218
|
async batchInferRaw(stepId, items, modelId, frameId = 0) {
|
|
2469
3219
|
if (!this.poolManager) throw new Error("EngineFactory.batchInferRaw: pool not initialized");
|
|
2470
3220
|
const idx = this.poolManager.getPoolIndex(stepId, modelId);
|
|
2471
|
-
if (idx === null) throw new Error(`EngineFactory.batchInferRaw: step "${stepId}"
|
|
3221
|
+
if (idx === null) throw new Error(`EngineFactory.batchInferRaw: step "${stepId}" (model "${modelId}") is not loaded`);
|
|
2472
3222
|
return this.poolManager.getPool().inferBatch(idx, items, frameId);
|
|
2473
3223
|
}
|
|
2474
3224
|
async cacheFrameInPool(raw, width, height, format) {
|
|
@@ -2478,7 +3228,7 @@ var EngineFactory = class {
|
|
|
2478
3228
|
async inferCached(stepId, frameId, modelId) {
|
|
2479
3229
|
if (!this.poolManager) throw new Error("inferCached: pool not available");
|
|
2480
3230
|
const idx = this.poolManager.getPoolIndex(stepId, modelId);
|
|
2481
|
-
if (idx === null) throw new Error(`inferCached: step "${stepId}"
|
|
3231
|
+
if (idx === null) throw new Error(`inferCached: step "${stepId}" (model "${modelId}") not loaded`);
|
|
2482
3232
|
return this.poolManager.getPool().inferCached(idx, frameId);
|
|
2483
3233
|
}
|
|
2484
3234
|
async uncacheFrame(frameId) {
|
|
@@ -2546,12 +3296,13 @@ var EngineFactory = class {
|
|
|
2546
3296
|
concurrency,
|
|
2547
3297
|
tuning: resolvedTuning,
|
|
2548
3298
|
numWorkers,
|
|
2549
|
-
...this.opts.engine.device ? { device: this.opts.engine.device } : {}
|
|
3299
|
+
...this.opts.engine.device ? { device: this.opts.engine.device } : {},
|
|
3300
|
+
...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
|
|
2550
3301
|
});
|
|
2551
3302
|
this.poolManager = new PipelineModelManager(this.pool, {
|
|
2552
3303
|
buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
|
|
2553
3304
|
resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
|
|
2554
|
-
}, this.log.child("model-mgr"));
|
|
3305
|
+
}, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
|
|
2555
3306
|
await this.pool.initialize([]);
|
|
2556
3307
|
await this.poolManager.applyConfig(steps);
|
|
2557
3308
|
}
|
|
@@ -2667,6 +3418,175 @@ function buildPoolModelConfigForStep(inputs) {
|
|
|
2667
3418
|
};
|
|
2668
3419
|
}
|
|
2669
3420
|
//#endregion
|
|
3421
|
+
//#region src/detection-pipeline/engine/model-load-governor.ts
|
|
3422
|
+
/**
|
|
3423
|
+
* What the dispatch path may ask of a pool's model loads, and what it must say
|
|
3424
|
+
* when it may not (D653, fix round 1).
|
|
3425
|
+
*
|
|
3426
|
+
* Three pieces of state, consulted ON DEMAND by the dispatch that needs a
|
|
3427
|
+
* model — there is no timer anywhere in here:
|
|
3428
|
+
*
|
|
3429
|
+
* 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
|
|
3430
|
+
* failed load, so without it every frame of every camera re-issued the
|
|
3431
|
+
* failing load and wrote two ERRORs. After a failure the model is not
|
|
3432
|
+
* re-issued on that pool for an interval that doubles per consecutive
|
|
3433
|
+
* failure up to a cap; a success clears it.
|
|
3434
|
+
* 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
|
|
3435
|
+
* bound is not a crash of the device. The SAME model timing out
|
|
3436
|
+
* {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
|
|
3437
|
+
* since a slow compile now runs on and writes its cache, only a compile
|
|
3438
|
+
* that never finishes can do — refuses THAT MODEL there, by name. The
|
|
3439
|
+
* device keeps serving every other model.
|
|
3440
|
+
* 3. Which camera has already been TOLD about the current state of a model,
|
|
3441
|
+
* so each camera gets one line per state change, never one per frame.
|
|
3442
|
+
*/
|
|
3443
|
+
/** The first back-off after a failed load. */
|
|
3444
|
+
var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
|
|
3445
|
+
/** The back-off doubles per consecutive failure up to this. */
|
|
3446
|
+
var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
|
|
3447
|
+
/** Timeouts older than this no longer count toward a refusal. */
|
|
3448
|
+
var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
|
|
3449
|
+
var ModelLoadGovernor = class {
|
|
3450
|
+
/** Keyed by the pool OBJECT, so a respawned pool starts clean. */
|
|
3451
|
+
backoff = /* @__PURE__ */ new WeakMap();
|
|
3452
|
+
timeouts = /* @__PURE__ */ new Map();
|
|
3453
|
+
reported = /* @__PURE__ */ new Map();
|
|
3454
|
+
generation = 0;
|
|
3455
|
+
/** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
|
|
3456
|
+
check(pool, deviceKey, models, now = Date.now()) {
|
|
3457
|
+
for (const model of models) {
|
|
3458
|
+
const t = this.timeouts.get(timeoutKey(deviceKey, model));
|
|
3459
|
+
if (t?.refused === true) return {
|
|
3460
|
+
kind: "refused",
|
|
3461
|
+
model,
|
|
3462
|
+
timeouts: t.at.length,
|
|
3463
|
+
error: t.lastError
|
|
3464
|
+
};
|
|
3465
|
+
}
|
|
3466
|
+
const perPool = this.backoff.get(pool);
|
|
3467
|
+
if (perPool === void 0) return { kind: "go" };
|
|
3468
|
+
for (const model of models) {
|
|
3469
|
+
const b = perPool.get(model);
|
|
3470
|
+
if (b !== void 0 && now < b.retryAtMs) return {
|
|
3471
|
+
kind: "backoff",
|
|
3472
|
+
model,
|
|
3473
|
+
retryInMs: b.retryAtMs - now,
|
|
3474
|
+
failures: b.failures,
|
|
3475
|
+
error: b.error,
|
|
3476
|
+
generation: b.generation
|
|
3477
|
+
};
|
|
3478
|
+
}
|
|
3479
|
+
return { kind: "go" };
|
|
3480
|
+
}
|
|
3481
|
+
/** One load of `model` on `pool` failed. */
|
|
3482
|
+
recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
|
|
3483
|
+
let perPool = this.backoff.get(pool);
|
|
3484
|
+
if (perPool === void 0) {
|
|
3485
|
+
perPool = /* @__PURE__ */ new Map();
|
|
3486
|
+
this.backoff.set(pool, perPool);
|
|
3487
|
+
}
|
|
3488
|
+
const previousEntry = perPool.get(model);
|
|
3489
|
+
const failures = (previousEntry?.failures ?? 0) + 1;
|
|
3490
|
+
const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
|
|
3491
|
+
this.generation += 1;
|
|
3492
|
+
const generation = this.generation;
|
|
3493
|
+
perPool.set(model, {
|
|
3494
|
+
failures,
|
|
3495
|
+
retryAtMs: now + retryInMs,
|
|
3496
|
+
error,
|
|
3497
|
+
generation,
|
|
3498
|
+
poolModel: poolModel ?? previousEntry?.poolModel ?? null,
|
|
3499
|
+
reason
|
|
3500
|
+
});
|
|
3501
|
+
if (!compileTimeout) return {
|
|
3502
|
+
generation,
|
|
3503
|
+
retryInMs,
|
|
3504
|
+
refusedNow: false,
|
|
3505
|
+
compileTimeouts: 0
|
|
3506
|
+
};
|
|
3507
|
+
const key = timeoutKey(deviceKey, model);
|
|
3508
|
+
const previous = this.timeouts.get(key);
|
|
3509
|
+
const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
|
|
3510
|
+
const refused = at.length >= 2;
|
|
3511
|
+
this.timeouts.set(key, {
|
|
3512
|
+
at,
|
|
3513
|
+
refused,
|
|
3514
|
+
lastError: error
|
|
3515
|
+
});
|
|
3516
|
+
return {
|
|
3517
|
+
generation,
|
|
3518
|
+
retryInMs,
|
|
3519
|
+
refusedNow: refused && previous?.refused !== true,
|
|
3520
|
+
compileTimeouts: at.length
|
|
3521
|
+
};
|
|
3522
|
+
}
|
|
3523
|
+
/**
|
|
3524
|
+
* A compile that outlived its bound came back on `pool` (D653 round 3).
|
|
3525
|
+
*
|
|
3526
|
+
* `ok`: THAT model's cache is written and the worker loads again, so its
|
|
3527
|
+
* back-off is lifted — and only its: another model's failure on the same
|
|
3528
|
+
* pool is not the compile that just finished. Not ok: it is one more
|
|
3529
|
+
* failure of that model, and its back-off grows. Refusals by name are never
|
|
3530
|
+
* lifted here; a model refused for timing out twice stays refused until the
|
|
3531
|
+
* operator re-arms. Returns the governor keys it touched.
|
|
3532
|
+
*/
|
|
3533
|
+
settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
|
|
3534
|
+
const perPool = this.backoff.get(pool);
|
|
3535
|
+
if (perPool === void 0) return [];
|
|
3536
|
+
const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
|
|
3537
|
+
for (const key of refusedMeanwhile) perPool.delete(key);
|
|
3538
|
+
const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
|
|
3539
|
+
for (const key of own) if (ok) perPool.delete(key);
|
|
3540
|
+
else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
|
|
3541
|
+
return [...refusedMeanwhile, ...own];
|
|
3542
|
+
}
|
|
3543
|
+
/** `models` loaded on `pool`: forget their failures there. */
|
|
3544
|
+
recordSuccess(pool, deviceKey, models) {
|
|
3545
|
+
const perPool = this.backoff.get(pool);
|
|
3546
|
+
for (const model of models) {
|
|
3547
|
+
perPool?.delete(model);
|
|
3548
|
+
this.timeouts.delete(timeoutKey(deviceKey, model));
|
|
3549
|
+
}
|
|
3550
|
+
}
|
|
3551
|
+
/**
|
|
3552
|
+
* Has `camera` been told about this `state` of `model` on `deviceKey` yet?
|
|
3553
|
+
* Returns `true` exactly once per state change.
|
|
3554
|
+
*/
|
|
3555
|
+
shouldReport(camera, deviceKey, model, state) {
|
|
3556
|
+
const key = `${camera ?? "-"}|${deviceKey}|${model}`;
|
|
3557
|
+
if (this.reported.get(key) === state) return false;
|
|
3558
|
+
this.reported.set(key, state);
|
|
3559
|
+
return true;
|
|
3560
|
+
}
|
|
3561
|
+
/** Every model refused on some device. */
|
|
3562
|
+
refused() {
|
|
3563
|
+
const out = [];
|
|
3564
|
+
for (const [key, t] of this.timeouts) {
|
|
3565
|
+
if (!t.refused) continue;
|
|
3566
|
+
const [deviceKey = "", model = ""] = key.split("\0");
|
|
3567
|
+
out.push({
|
|
3568
|
+
deviceKey,
|
|
3569
|
+
model,
|
|
3570
|
+
timeouts: t.at.length,
|
|
3571
|
+
lastError: t.lastError
|
|
3572
|
+
});
|
|
3573
|
+
}
|
|
3574
|
+
return out;
|
|
3575
|
+
}
|
|
3576
|
+
/** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
|
|
3577
|
+
rearm(deviceKey) {
|
|
3578
|
+
let cleared = 0;
|
|
3579
|
+
for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
|
|
3580
|
+
this.timeouts.delete(key);
|
|
3581
|
+
cleared += 1;
|
|
3582
|
+
}
|
|
3583
|
+
return cleared;
|
|
3584
|
+
}
|
|
3585
|
+
};
|
|
3586
|
+
function timeoutKey(deviceKey, model) {
|
|
3587
|
+
return `${deviceKey}\u0000${model}`;
|
|
3588
|
+
}
|
|
3589
|
+
//#endregion
|
|
2670
3590
|
//#region src/detection-pipeline/engine/idle-pool-reaper.ts
|
|
2671
3591
|
var IdlePoolReaper = class {
|
|
2672
3592
|
lastUsed = /* @__PURE__ */ new Map();
|
|
@@ -3203,13 +4123,16 @@ function checkReplayPin(input) {
|
|
|
3203
4123
|
}
|
|
3204
4124
|
var PRIMARY_ROOT_STEP = "object-detection";
|
|
3205
4125
|
/**
|
|
3206
|
-
* The
|
|
3207
|
-
*
|
|
3208
|
-
*
|
|
4126
|
+
* The same rule over an EXECUTABLE tree's roots (D658): their model ids are the
|
|
4127
|
+
* pool variants that will run — after resolution, per-step loads and drops —
|
|
4128
|
+
* so this is what a live `runPipeline` stamps. The tree holds only enabled
|
|
4129
|
+
* video roots already.
|
|
3209
4130
|
*/
|
|
3210
|
-
function
|
|
3211
|
-
|
|
3212
|
-
|
|
4131
|
+
function resolveTreeRootModelId(roots) {
|
|
4132
|
+
return primaryRootModelId(roots);
|
|
4133
|
+
}
|
|
4134
|
+
function primaryRootModelId(roots) {
|
|
4135
|
+
return (roots.find((s) => s.stepId === PRIMARY_ROOT_STEP) ?? roots[0])?.modelId;
|
|
3213
4136
|
}
|
|
3214
4137
|
//#endregion
|
|
3215
4138
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
@@ -6002,26 +6925,47 @@ function resolveEffectiveClassMap(definition, modelId, resolveModel) {
|
|
|
6002
6925
|
* Build an executable tree from user config.
|
|
6003
6926
|
*
|
|
6004
6927
|
* @param steps - User-configured pipeline steps (from PipelineDefaultStep[])
|
|
6005
|
-
* @param getEngine -
|
|
6006
|
-
* Throws if
|
|
6928
|
+
* @param getEngine - Returns the engine that runs exactly `(stepId, modelId)`
|
|
6929
|
+
* and the model it runs. Throws if that pair is not loaded.
|
|
6007
6930
|
*/
|
|
6008
|
-
function buildExecutableTree(steps, getEngine, resolveModel) {
|
|
6009
|
-
return { roots: steps.filter((s) => s.enabled).filter((s) => s.slot !== "audio-classifier").map((s) => buildNode(s, getEngine, resolveModel)) };
|
|
6931
|
+
function buildExecutableTree(steps, getEngine, resolveModel, onMissingVariant) {
|
|
6932
|
+
return { roots: steps.filter((s) => s.enabled).filter((s) => s.slot !== "audio-classifier").map((s) => buildNode(s, getEngine, resolveModel, onMissingVariant)) };
|
|
6933
|
+
}
|
|
6934
|
+
function subtreeIds(step) {
|
|
6935
|
+
return [step.addonId, ...(step.children ?? []).filter((c) => c.enabled).flatMap((c) => subtreeIds(c))];
|
|
6010
6936
|
}
|
|
6011
|
-
function
|
|
6937
|
+
function buildChild(child, parentStepId, getEngine, resolveModel, onMissingVariant) {
|
|
6938
|
+
if (onMissingVariant === void 0) return buildNode(child, getEngine, resolveModel, onMissingVariant);
|
|
6939
|
+
let variant;
|
|
6940
|
+
try {
|
|
6941
|
+
variant = getEngine(child.addonId, child.modelId);
|
|
6942
|
+
} catch (err) {
|
|
6943
|
+
onMissingVariant({
|
|
6944
|
+
step: child,
|
|
6945
|
+
stepId: child.addonId,
|
|
6946
|
+
modelId: child.modelId,
|
|
6947
|
+
parentStepId,
|
|
6948
|
+
droppedSubtree: subtreeIds(child),
|
|
6949
|
+
error: err instanceof Error ? err.message : String(err)
|
|
6950
|
+
});
|
|
6951
|
+
return null;
|
|
6952
|
+
}
|
|
6953
|
+
return buildNode(child, getEngine, resolveModel, onMissingVariant, variant);
|
|
6954
|
+
}
|
|
6955
|
+
function buildNode(step, getEngine, resolveModel, onMissingVariant, resolved) {
|
|
6012
6956
|
const definition = require_default_detection_model.getStepDefinition(step.addonId);
|
|
6013
|
-
const
|
|
6014
|
-
const children = (step.children ?? []).filter((c) => c.enabled).map((c) =>
|
|
6957
|
+
const variant = resolved ?? getEngine(step.addonId, step.modelId);
|
|
6958
|
+
const children = (step.children ?? []).filter((c) => c.enabled).map((c) => buildChild(c, step.addonId, getEngine, resolveModel, onMissingVariant)).filter((c) => c !== null);
|
|
6015
6959
|
const mergedSettings = {
|
|
6016
6960
|
...collectSchemaDefaults(step.addonId),
|
|
6017
6961
|
...step.settings
|
|
6018
6962
|
};
|
|
6019
|
-
const effectiveClassMap = resolveEffectiveClassMap(definition,
|
|
6963
|
+
const effectiveClassMap = resolveEffectiveClassMap(definition, variant.modelId, resolveModel);
|
|
6020
6964
|
return {
|
|
6021
6965
|
stepId: step.addonId,
|
|
6022
6966
|
definition,
|
|
6023
|
-
engine,
|
|
6024
|
-
modelId:
|
|
6967
|
+
engine: variant.engine,
|
|
6968
|
+
modelId: variant.modelId,
|
|
6025
6969
|
inputClasses: definition.inputClasses ?? [],
|
|
6026
6970
|
enabled: step.enabled,
|
|
6027
6971
|
children,
|
|
@@ -6051,43 +6995,6 @@ function walkFieldsForDefaults(fields, out) {
|
|
|
6051
6995
|
}
|
|
6052
6996
|
}
|
|
6053
6997
|
//#endregion
|
|
6054
|
-
//#region src/detection-pipeline/registry/custom-models.ts
|
|
6055
|
-
/**
|
|
6056
|
-
* Group a flat list of custom-model descriptors (as returned by the
|
|
6057
|
-
* `custom-model-registry` collection cap) into a `stepId → entries` map for
|
|
6058
|
-
* the picker / resolution union. Pure; order within a step preserved.
|
|
6059
|
-
*/
|
|
6060
|
-
function groupCustomModelsByStep(descriptors) {
|
|
6061
|
-
const byStep = /* @__PURE__ */ new Map();
|
|
6062
|
-
for (const d of descriptors) {
|
|
6063
|
-
const arr = byStep.get(d.stepId) ?? [];
|
|
6064
|
-
arr.push(d.entry);
|
|
6065
|
-
byStep.set(d.stepId, arr);
|
|
6066
|
-
}
|
|
6067
|
-
return byStep;
|
|
6068
|
-
}
|
|
6069
|
-
/**
|
|
6070
|
-
* Union a step's static catalog models with operator-registered custom
|
|
6071
|
-
* models. On an `id` collision the static catalog entry wins — a custom
|
|
6072
|
-
* model can never shadow a built-in one.
|
|
6073
|
-
*
|
|
6074
|
-
* Pure + side-effect-free so it can be unit-tested in isolation and called
|
|
6075
|
-
* from the (free) `buildSchemaSlots` builder without any addon context.
|
|
6076
|
-
*/
|
|
6077
|
-
function mergeCustomModels(staticModels, customModels) {
|
|
6078
|
-
const seen = new Set(staticModels.map((m) => m.id));
|
|
6079
|
-
const merged = [...staticModels];
|
|
6080
|
-
for (const m of customModels) {
|
|
6081
|
-
if (seen.has(m.id)) continue;
|
|
6082
|
-
seen.add(m.id);
|
|
6083
|
-
merged.push({
|
|
6084
|
-
...m,
|
|
6085
|
-
provider: require_dist.inferModelProvider(m)
|
|
6086
|
-
});
|
|
6087
|
-
}
|
|
6088
|
-
return merged;
|
|
6089
|
-
}
|
|
6090
|
-
//#endregion
|
|
6091
6998
|
//#region src/detection-pipeline/cluster-model-resolution.ts
|
|
6092
6999
|
/**
|
|
6093
7000
|
* The cluster model rule, enforced on the node that has to obey it.
|
|
@@ -6192,8 +7099,10 @@ function collectSubstitutions(steps, format) {
|
|
|
6192
7099
|
return result;
|
|
6193
7100
|
}
|
|
6194
7101
|
/**
|
|
6195
|
-
* Return one {@link ZeroBuildIssue} per step in `steps`
|
|
6196
|
-
*
|
|
7102
|
+
* Return one {@link ZeroBuildIssue} per step in `steps` that a pool of `format`
|
|
7103
|
+
* cannot run: neither the model it names nor any NON-legacy model of its
|
|
7104
|
+
* catalog (via `getStepDef`) ships a build for `format`
|
|
7105
|
+
* (`stepModelRunsOnFormat`, D657). `steps` is
|
|
6197
7106
|
* expected to already be flattened + filtered to the set the caller cares
|
|
6198
7107
|
* about (e.g. enabled video steps) — this does not recurse into children or
|
|
6199
7108
|
* filter by `enabled`, unlike {@link collectSubstitutions}.
|
|
@@ -6214,7 +7123,7 @@ function collectZeroBuildIssues(steps, format, getStepDef = require_default_dete
|
|
|
6214
7123
|
} catch {
|
|
6215
7124
|
continue;
|
|
6216
7125
|
}
|
|
6217
|
-
if (!def.models.
|
|
7126
|
+
if (!require_dist.stepModelRunsOnFormat(def.models, step.modelId, format)) result.push({
|
|
6218
7127
|
addonId: step.addonId,
|
|
6219
7128
|
format
|
|
6220
7129
|
});
|
|
@@ -6264,11 +7173,20 @@ function collectUnknownAddonIssues(steps, getStepDef = require_default_detection
|
|
|
6264
7173
|
* `diagnostics.substitutions` instead of happening silently. The caller
|
|
6265
7174
|
* (the provider) owns logging + dedup for both.
|
|
6266
7175
|
*/
|
|
6267
|
-
function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings) {
|
|
7176
|
+
function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings, unknownClusterSteps) {
|
|
6268
7177
|
const resolvedSteps = [];
|
|
6269
7178
|
const unknownAddonIds = [];
|
|
6270
7179
|
const substitutions = [];
|
|
6271
7180
|
const clusterRefusals = [];
|
|
7181
|
+
const clusterUnknown = [];
|
|
7182
|
+
const recurse = (children) => {
|
|
7183
|
+
const r = resolveInputSteps(children, format, engine, clusterModels, clusterSettings, unknownClusterSteps);
|
|
7184
|
+
unknownAddonIds.push(...r.diagnostics.unknownAddonIds);
|
|
7185
|
+
substitutions.push(...r.diagnostics.substitutions);
|
|
7186
|
+
clusterRefusals.push(...r.diagnostics.clusterRefusals);
|
|
7187
|
+
clusterUnknown.push(...r.diagnostics.clusterUnknown);
|
|
7188
|
+
return r;
|
|
7189
|
+
};
|
|
6272
7190
|
for (const s of steps) {
|
|
6273
7191
|
let def;
|
|
6274
7192
|
try {
|
|
@@ -6277,6 +7195,10 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6277
7195
|
unknownAddonIds.push(s.addonId);
|
|
6278
7196
|
continue;
|
|
6279
7197
|
}
|
|
7198
|
+
if (def.modelScope === "cluster" && unknownClusterSteps?.has(s.addonId) === true) {
|
|
7199
|
+
clusterUnknown.push(s.addonId);
|
|
7200
|
+
continue;
|
|
7201
|
+
}
|
|
6280
7202
|
if (clusterModels !== void 0) {
|
|
6281
7203
|
const verdict = resolveClusterStepModel(s.addonId, clusterModels, format, () => def);
|
|
6282
7204
|
if (verdict.kind !== "node-scoped") {
|
|
@@ -6286,12 +7208,7 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6286
7208
|
format: verdict.format,
|
|
6287
7209
|
formatsShipped: verdict.formatsShipped
|
|
6288
7210
|
});
|
|
6289
|
-
const childResult = s.children ?
|
|
6290
|
-
if (childResult) {
|
|
6291
|
-
unknownAddonIds.push(...childResult.diagnostics.unknownAddonIds);
|
|
6292
|
-
substitutions.push(...childResult.diagnostics.substitutions);
|
|
6293
|
-
clusterRefusals.push(...childResult.diagnostics.clusterRefusals);
|
|
6294
|
-
}
|
|
7211
|
+
const childResult = s.children ? recurse(s.children) : null;
|
|
6295
7212
|
const settings = require_dist.overlayClusterStepSettings(s.addonId, s.settings, clusterSettings ?? {});
|
|
6296
7213
|
resolvedSteps.push({
|
|
6297
7214
|
addonId: s.addonId,
|
|
@@ -6315,12 +7232,7 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6315
7232
|
running: runningModelId,
|
|
6316
7233
|
format
|
|
6317
7234
|
});
|
|
6318
|
-
const childResult = s.children ?
|
|
6319
|
-
if (childResult) {
|
|
6320
|
-
unknownAddonIds.push(...childResult.diagnostics.unknownAddonIds);
|
|
6321
|
-
substitutions.push(...childResult.diagnostics.substitutions);
|
|
6322
|
-
clusterRefusals.push(...childResult.diagnostics.clusterRefusals);
|
|
6323
|
-
}
|
|
7235
|
+
const childResult = s.children ? recurse(s.children) : null;
|
|
6324
7236
|
const settings = require_dist.overlayClusterStepSettings(s.addonId, s.settings, clusterSettings ?? {});
|
|
6325
7237
|
resolvedSteps.push({
|
|
6326
7238
|
addonId: s.addonId,
|
|
@@ -6339,7 +7251,8 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6339
7251
|
diagnostics: {
|
|
6340
7252
|
unknownAddonIds,
|
|
6341
7253
|
substitutions,
|
|
6342
|
-
clusterRefusals
|
|
7254
|
+
clusterRefusals,
|
|
7255
|
+
clusterUnknown
|
|
6343
7256
|
}
|
|
6344
7257
|
};
|
|
6345
7258
|
}
|
|
@@ -6563,6 +7476,90 @@ function parseWavToAudioChunk(filePath) {
|
|
|
6563
7476
|
function enginesEqual(a, b) {
|
|
6564
7477
|
return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
|
|
6565
7478
|
}
|
|
7479
|
+
/** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
|
|
7480
|
+
var ModelLoadRefusedError = class extends Error {
|
|
7481
|
+
model;
|
|
7482
|
+
state;
|
|
7483
|
+
constructor(message, model, state) {
|
|
7484
|
+
super(message);
|
|
7485
|
+
this.name = "ModelLoadRefusedError";
|
|
7486
|
+
this.model = model;
|
|
7487
|
+
this.state = state;
|
|
7488
|
+
}
|
|
7489
|
+
};
|
|
7490
|
+
/** `step/model` — how the governor and the log lines name a model. */
|
|
7491
|
+
function stepModelKey(step) {
|
|
7492
|
+
return `${step.addonId}/${step.modelId}`;
|
|
7493
|
+
}
|
|
7494
|
+
/** The reporting state a refusal verdict stands for. */
|
|
7495
|
+
function stateOf(verdict) {
|
|
7496
|
+
return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
|
|
7497
|
+
}
|
|
7498
|
+
/** The load policy of a `runPipeline` call — see {@link DispatchLoadPolicy}. */
|
|
7499
|
+
function dispatchLoadPolicyOf(input) {
|
|
7500
|
+
return input.deviceId !== void 0 && input.replay !== true ? "partial" : "all-or-nothing";
|
|
7501
|
+
}
|
|
7502
|
+
/**
|
|
7503
|
+
* An all-or-nothing call lost steps to failed loads (D657). Names every one,
|
|
7504
|
+
* so a benchmark never reports timings for a tree it did not run.
|
|
7505
|
+
*/
|
|
7506
|
+
var DispatchStepsNotLoadedError = class extends Error {
|
|
7507
|
+
failedSteps;
|
|
7508
|
+
constructor(failures) {
|
|
7509
|
+
const named = failures.map((f) => `${f.step.addonId}/${f.step.modelId} (${f.error instanceof Error ? f.error.message : String(f.error)})`);
|
|
7510
|
+
super(`pipeline steps failed to load — the call runs all of them or none: ${named.join("; ")}`, { cause: failures[0]?.error });
|
|
7511
|
+
this.name = "DispatchStepsNotLoadedError";
|
|
7512
|
+
this.failedSteps = failures.map((f) => `${f.step.addonId}/${f.step.modelId}`);
|
|
7513
|
+
}
|
|
7514
|
+
};
|
|
7515
|
+
/**
|
|
7516
|
+
* A step's model artifact cannot be made present for the pool's format — no
|
|
7517
|
+
* build for that format, a custom model missing from this node, a download
|
|
7518
|
+
* that failed. Named by step and model so the failure is charged to THAT
|
|
7519
|
+
* model only (D657): a plain `Error` was charged to every model of the set,
|
|
7520
|
+
* and one face model with no tflite build backed the Coral's root detector
|
|
7521
|
+
* off for minutes.
|
|
7522
|
+
*/
|
|
7523
|
+
var StepModelArtifactError = class extends Error {
|
|
7524
|
+
stepId;
|
|
7525
|
+
modelId;
|
|
7526
|
+
constructor(stepId, modelId, cause) {
|
|
7527
|
+
super(cause instanceof Error ? cause.message : String(cause), { cause });
|
|
7528
|
+
this.name = "StepModelArtifactError";
|
|
7529
|
+
this.stepId = stepId;
|
|
7530
|
+
this.modelId = modelId;
|
|
7531
|
+
}
|
|
7532
|
+
};
|
|
7533
|
+
/** The one model a failed load is charged to, when the error names it. */
|
|
7534
|
+
function failedModelOf(err) {
|
|
7535
|
+
if (err instanceof StepVariantLoadError || err instanceof StepModelArtifactError) return `${err.stepId}/${err.modelId}`;
|
|
7536
|
+
return null;
|
|
7537
|
+
}
|
|
7538
|
+
/**
|
|
7539
|
+
* `steps` without the subtrees of the steps whose model did not load (D657).
|
|
7540
|
+
* A step's children consume its output, so they leave with it. Matched by
|
|
7541
|
+
* INSTANCE: two steps of one addon in a tree are two steps.
|
|
7542
|
+
*/
|
|
7543
|
+
function withoutFailedSteps(steps, failed) {
|
|
7544
|
+
const kept = [];
|
|
7545
|
+
for (const step of steps) {
|
|
7546
|
+
if (failed.has(step)) continue;
|
|
7547
|
+
kept.push(step.children && step.children.length > 0 ? {
|
|
7548
|
+
...step,
|
|
7549
|
+
children: withoutFailedSteps(step.children, failed)
|
|
7550
|
+
} : step);
|
|
7551
|
+
}
|
|
7552
|
+
return kept;
|
|
7553
|
+
}
|
|
7554
|
+
/**
|
|
7555
|
+
* The reason a condemned pool is charged against its device's restart budget.
|
|
7556
|
+
* A pool recycled because it HUNG (D653) names the hang — `inference device
|
|
7557
|
+
* FAILED … reason: pool worker is not ready` could not tell a GPU compile that
|
|
7558
|
+
* never returned from a crash.
|
|
7559
|
+
*/
|
|
7560
|
+
function deathReasonOf(factory) {
|
|
7561
|
+
return factory.getDeathCause()?.message ?? "pool worker is not ready";
|
|
7562
|
+
}
|
|
6566
7563
|
/** Build a `RuntimeEnv` from the running process + probed hardware. */
|
|
6567
7564
|
function runtimeEnvFromProcess(hardware) {
|
|
6568
7565
|
return {
|
|
@@ -6661,18 +7658,18 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6661
7658
|
* paths don't issue a cap round-trip on every call. Empty map = no provider
|
|
6662
7659
|
* (or a query failure) → behaviour identical to the static-catalog-only path.
|
|
6663
7660
|
*/
|
|
6664
|
-
|
|
7661
|
+
customModels = new require_default_detection_model.CustomModelCatalog({ warn: (message, extras) => this.log.warn(message, extras) });
|
|
6665
7662
|
/**
|
|
6666
7663
|
* SYNC custom-model lookup handed to every live {@link EngineFactory} so
|
|
6667
7664
|
* `buildPoolModelConfigForStep` can resolve an operator-registered model at
|
|
6668
|
-
* pool-load time. Reads the {@link
|
|
7665
|
+
* pool-load time. Reads the {@link customModels} snapshot — every load
|
|
6669
7666
|
* path (`ensureModelsForSteps`, `ensureModelsForCurrentSteps`) refreshes the
|
|
6670
7667
|
* cache via `resolveModelEntry` BEFORE `loadAdditional`/`applyConfig`, so
|
|
6671
7668
|
* the snapshot is fresh when the pool asks. Without this seam a selected
|
|
6672
7669
|
* custom model failed the WHOLE pool applyConfig with
|
|
6673
7670
|
* `Model "X" not found in step "Y" catalog`.
|
|
6674
7671
|
*/
|
|
6675
|
-
customModelResolver = (stepId, modelId) => this.
|
|
7672
|
+
customModelResolver = (stepId, modelId) => this.customModels.current().get(stepId)?.find((m) => m.id === modelId);
|
|
6676
7673
|
/**
|
|
6677
7674
|
* Per-device {@link DeviceProxy} cache used for zone gating at the
|
|
6678
7675
|
* runtime path. Reads `state.zones.value` + `state.zoneRules.value`
|
|
@@ -6695,6 +7692,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6695
7692
|
info: (message, extras) => this.log.info(message, extras),
|
|
6696
7693
|
warn: (message, extras) => this.log.warn(message, extras)
|
|
6697
7694
|
} });
|
|
7695
|
+
/** One line per camera when a cluster-scoped step starts running another model (D658). */
|
|
7696
|
+
clusterModelSwitchLog = new ClusterModelSwitchLog({ info: (message, extras) => this.log.info(message, extras) });
|
|
7697
|
+
/** One line per (camera, step) while a cluster step is refused as unknown (D658). */
|
|
7698
|
+
clusterUnknownLog = new ClusterUnknownRefusalLog({ warn: (message, extras) => this.log.warn(message, extras) });
|
|
6698
7699
|
/**
|
|
6699
7700
|
* Last logged "step configured" signature — used to dedupe the
|
|
6700
7701
|
* per-step debug trail so it only fires on actual config CHANGE
|
|
@@ -7161,7 +8162,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7161
8162
|
if (!steps) return;
|
|
7162
8163
|
const format = this.currentEngine.format;
|
|
7163
8164
|
const substitutionIssues = this.getActiveModelSubstitutions().map((s) => `${s.addonId}: chose ${s.chosen}, running ${s.running} (${s.format})`);
|
|
7164
|
-
const zeroBuildIssues = collectZeroBuildIssues(flattenSteps(steps), format).map((i) => `${i.addonId}: no model has a ${i.format} build`);
|
|
8165
|
+
const zeroBuildIssues = collectZeroBuildIssues(flattenSteps(steps), format, this.mergedStepDefOrThrow).map((i) => `${i.addonId}: no model has a ${i.format} build`);
|
|
7165
8166
|
this.configIssues = [...substitutionIssues, ...zeroBuildIssues];
|
|
7166
8167
|
} catch (err) {
|
|
7167
8168
|
this.configIssues = [];
|
|
@@ -7295,24 +8296,21 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7295
8296
|
* a warning — callers then behave exactly like the static-catalog path.
|
|
7296
8297
|
*/
|
|
7297
8298
|
async getCustomModels() {
|
|
7298
|
-
|
|
7299
|
-
if (this.customModelsCache && now - this.customModelsCache.at < 5e3) return this.customModelsCache.byStep;
|
|
7300
|
-
let byStep = /* @__PURE__ */ new Map();
|
|
7301
|
-
try {
|
|
7302
|
-
const api = this.addonCtx?.api;
|
|
7303
|
-
if (api) {
|
|
7304
|
-
if ((await api.addons.listCapabilityProviders.query({ capName: "custom-model-registry" })).some((p) => p.isActive)) byStep = groupCustomModelsByStep(await api.customModelRegistry.listModels.query());
|
|
7305
|
-
}
|
|
7306
|
-
} catch (err) {
|
|
7307
|
-
this.log.warn("custom-model-registry query failed — using static catalog only", { meta: { error: require_dist.errMsg(err) } });
|
|
7308
|
-
}
|
|
7309
|
-
this.customModelsCache = {
|
|
7310
|
-
at: now,
|
|
7311
|
-
byStep
|
|
7312
|
-
};
|
|
7313
|
-
return byStep;
|
|
8299
|
+
return this.customModels.refresh(this.addonCtx?.api);
|
|
7314
8300
|
}
|
|
7315
8301
|
/**
|
|
8302
|
+
* A step's definition MERGED with its custom models, synchronously, from
|
|
8303
|
+
* the last custom-model read (D657 round 2). Every placement gate here
|
|
8304
|
+
* judges this catalog — the one the orchestrator's schema carries.
|
|
8305
|
+
*/
|
|
8306
|
+
mergedStepDef = (addonId) => this.customModels.stepDefinition(addonId);
|
|
8307
|
+
/** {@link mergedStepDef} as the throwing lookup `collectZeroBuildIssues` takes. */
|
|
8308
|
+
mergedStepDefOrThrow = (addonId) => {
|
|
8309
|
+
const def = this.mergedStepDef(addonId);
|
|
8310
|
+
if (def === null) throw new Error(`Unknown pipeline step: "${addonId}"`);
|
|
8311
|
+
return def;
|
|
8312
|
+
};
|
|
8313
|
+
/**
|
|
7316
8314
|
* Resolve a model id within a step to a catalog entry — static catalog
|
|
7317
8315
|
* first, then the custom registry. Returns undefined if neither has it.
|
|
7318
8316
|
*/
|
|
@@ -7503,7 +8501,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7503
8501
|
const enabledSteps = this.pruneToEnabledSteps(steps);
|
|
7504
8502
|
const substitutions = collectSubstitutions(enabledSteps, format);
|
|
7505
8503
|
const unknownAddonIssues = collectUnknownAddonIssues(enabledSteps);
|
|
7506
|
-
const zeroBuildIssues = collectZeroBuildIssues(this.flattenEnabledVideoStepInputs(steps), format);
|
|
8504
|
+
const zeroBuildIssues = collectZeroBuildIssues(this.flattenEnabledVideoStepInputs(steps), format, this.mergedStepDefOrThrow);
|
|
7507
8505
|
const issues = [...unknownAddonIssues.map((u) => ({
|
|
7508
8506
|
addonId: u.addonId,
|
|
7509
8507
|
kind: "unknown-addon",
|
|
@@ -7820,6 +8818,12 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7820
8818
|
}
|
|
7821
8819
|
async runPipeline(input, onProgress) {
|
|
7822
8820
|
await this.clusterModels.refresh(this.addonCtx?.api);
|
|
8821
|
+
this.clusterUnknownLog.note(input.deviceId ?? 0, input.steps, this.clusterModels.unknownSteps());
|
|
8822
|
+
if (dispatchLoadPolicyOf(input) === "all-or-nothing") {
|
|
8823
|
+
const unknown = unknownStepsNamed(input.steps, this.clusterModels.unknownSteps());
|
|
8824
|
+
if (unknown.length > 0) throw new Error(`cluster-model-unknown: the owner has not answered which model the cluster runs for ${unknown.join(", ")} — the call runs all of its steps or none`);
|
|
8825
|
+
}
|
|
8826
|
+
await this.getCustomModels();
|
|
7823
8827
|
const nodeId = this.addonCtx?.kernel?.localNodeId ?? "hub";
|
|
7824
8828
|
const sessionId = input.sessionId ?? `run-${Date.now()}-${Math.random().toString(36).slice(2, 10)}`;
|
|
7825
8829
|
const isRuntime = Boolean(input.frame || input.frameRef);
|
|
@@ -7984,11 +8988,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7984
8988
|
}
|
|
7985
8989
|
const enabledSteps = flattenSteps(benchmarkSteps);
|
|
7986
8990
|
emit(`Pipeline: ${enabledSteps.length} step(s) — ${enabledSteps.map((s) => s.addonId).join(" → ")}`);
|
|
7987
|
-
const rootModelId = resolveRootModelId(benchmarkSteps);
|
|
7988
|
-
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
7989
|
-
...r,
|
|
7990
|
-
rootModelId
|
|
7991
|
-
};
|
|
7992
8991
|
if (input.replay === true) {
|
|
7993
8992
|
const verdict = checkReplayPin({
|
|
7994
8993
|
requested: input.steps,
|
|
@@ -8030,21 +9029,51 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8030
9029
|
const dispatchEngine = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey) : runEngine;
|
|
8031
9030
|
const dispatchFactory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8032
9031
|
const needed = enabledSteps.filter((s) => dispatchFactory.needsPoolUpdate(s));
|
|
9032
|
+
let runSteps = benchmarkSteps;
|
|
9033
|
+
const loadPolicy = dispatchLoadPolicyOf(input);
|
|
8033
9034
|
if (needed.length > 0) {
|
|
8034
|
-
|
|
8035
|
-
|
|
8036
|
-
|
|
9035
|
+
const dispatchDeviceKeyLabel = deviceKeyOf(dispatchEngine);
|
|
9036
|
+
const toIssue = needed.filter((s) => this.loadGovernor.check(dispatchFactory, dispatchDeviceKeyLabel, [stepModelKey(s)]).kind === "go");
|
|
9037
|
+
if (toIssue.length > 0) this.log.info("Benchmark: models to load", { meta: {
|
|
9038
|
+
count: toIssue.length,
|
|
9039
|
+
models: toIssue.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8037
9040
|
} });
|
|
8038
|
-
for (const s of
|
|
9041
|
+
for (const s of toIssue) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8039
9042
|
step: s.addonId,
|
|
8040
9043
|
addonId: s.addonId,
|
|
8041
9044
|
modelId: s.modelId
|
|
8042
9045
|
});
|
|
8043
|
-
await this.
|
|
8044
|
-
|
|
9046
|
+
const failures = await this.loadDispatchModels(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
|
|
9047
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
9048
|
+
deviceKey: dispatchDeviceKeyLabel
|
|
9049
|
+
}, loadPolicy);
|
|
9050
|
+
runSteps = stepsSurvivingLoad(benchmarkSteps, failures, loadPolicy);
|
|
9051
|
+
emit(failures.length === 0 ? `All models loaded` : `Models loaded (${failures.length} failed)`);
|
|
8045
9052
|
}
|
|
8046
9053
|
emit("Running inference...");
|
|
8047
|
-
const
|
|
9054
|
+
const notWarm = [];
|
|
9055
|
+
const tree = buildExecutableTree(runSteps, (stepId, modelId) => dispatchFactory.getStepVariant(stepId, modelId), this.customModelResolver, (drop) => notWarm.push({
|
|
9056
|
+
step: drop.step,
|
|
9057
|
+
error: new Error(drop.error)
|
|
9058
|
+
}));
|
|
9059
|
+
if (notWarm.length > 0) {
|
|
9060
|
+
if (loadPolicy === "all-or-nothing") throw new DispatchStepsNotLoadedError(notWarm);
|
|
9061
|
+
const dispatchDeviceKeyLabel = deviceKeyOf(dispatchEngine);
|
|
9062
|
+
for (const failure of notWarm) {
|
|
9063
|
+
const model = stepModelKey(failure.step);
|
|
9064
|
+
this.reportDispatchLoadLoss({
|
|
9065
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
9066
|
+
deviceKey: dispatchDeviceKeyLabel,
|
|
9067
|
+
outcome: "step-dropped"
|
|
9068
|
+
}, dispatchDeviceKeyLabel, [model], model, "not-warm", require_dist.errMsg(failure.error));
|
|
9069
|
+
}
|
|
9070
|
+
}
|
|
9071
|
+
const rootModelId = resolveTreeRootModelId(tree.roots);
|
|
9072
|
+
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
9073
|
+
...r,
|
|
9074
|
+
rootModelId
|
|
9075
|
+
};
|
|
9076
|
+
if (input.replay !== true) this.clusterModelSwitchLog.note(input.deviceId ?? 0, tree);
|
|
8048
9077
|
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8049
9078
|
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8050
9079
|
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
@@ -8150,7 +9179,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8150
9179
|
backend: this.currentEngine.backend,
|
|
8151
9180
|
device: this.currentEngine.device ?? null
|
|
8152
9181
|
}, clusterRefusalsOut) {
|
|
8153
|
-
const { steps: resolvedSteps, diagnostics } = resolveInputSteps(steps, format, engine, this.clusterModels.current(), this.clusterModels.currentSettings());
|
|
9182
|
+
const { steps: resolvedSteps, diagnostics } = resolveInputSteps(steps, format, engine, this.clusterModels.current(), this.clusterModels.currentSettings(), this.clusterModels.unknownSteps());
|
|
8154
9183
|
if (clusterRefusalsOut) clusterRefusalsOut.push(...diagnostics.clusterRefusals);
|
|
8155
9184
|
for (const addonId of diagnostics.unknownAddonIds) this.logLiveDispatchIssueOnce(`unknown:${addonId}`, () => this.log.warn("Live pipeline step references an unknown addon — dropping step", { meta: { addonId } }));
|
|
8156
9185
|
for (const sub of diagnostics.substitutions) this.logLiveDispatchIssueOnce(`${sub.addonId}|${sub.chosen}|${sub.running}|${sub.format}`, () => this.log.info("Live pipeline step model substituted for engine format (node-local)", { meta: {
|
|
@@ -8263,7 +9292,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8263
9292
|
const deviceRefusals = [];
|
|
8264
9293
|
const steps = this.inputStepsToPipelineSteps(args.steps, deviceFormat, deviceStepEngine, deviceRefusals);
|
|
8265
9294
|
const framePlane = args.plane === "frame";
|
|
8266
|
-
const zeroBuild = collectZeroBuildIssues(framePlane ? collectFramePlaneSteps(steps, isDetailPlaneStep) : flattenSteps(steps), deviceFormat);
|
|
9295
|
+
const zeroBuild = collectZeroBuildIssues(framePlane ? collectFramePlaneSteps(steps, isDetailPlaneStep) : flattenSteps(steps), deviceFormat, this.mergedStepDefOrThrow);
|
|
8267
9296
|
if (zeroBuild.length === 0) {
|
|
8268
9297
|
if (!framePlane) {
|
|
8269
9298
|
this.logClusterRefusals(deviceRefusals, deviceId, `${deviceKey} (${deviceFormat})`);
|
|
@@ -8272,7 +9301,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8272
9301
|
steps
|
|
8273
9302
|
};
|
|
8274
9303
|
}
|
|
8275
|
-
const pruneResult = pruneUnrunnableDetailSteps(steps, isDetailPlaneStep, (
|
|
9304
|
+
const pruneResult = pruneUnrunnableDetailSteps(steps, isDetailPlaneStep, (subtree) => !require_default_detection_model.buildSubtreeCanRun(subtree, this.mergedStepDef)(deviceFormat));
|
|
8276
9305
|
if (pruneResult.prunedAddonIds.length > 0) {
|
|
8277
9306
|
const prunedList = pruneResult.prunedAddonIds.join(",");
|
|
8278
9307
|
this.logLiveDispatchIssueOnce(`detail-defer:${deviceKey}:${deviceFormat}:${prunedList}`, () => this.log.info("Detail steps have no model build for the dispatch device format — deferred to the detail-plane device jump", {
|
|
@@ -8284,7 +9313,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8284
9313
|
}
|
|
8285
9314
|
}));
|
|
8286
9315
|
}
|
|
8287
|
-
this.logClusterRefusals(deviceRefusals.filter((r) => !pruneResult.
|
|
9316
|
+
this.logClusterRefusals(deviceRefusals.filter((r) => !pruneResult.removedAddonIds.includes(r.stepId)), deviceId, `${deviceKey} (${deviceFormat})`);
|
|
8288
9317
|
return {
|
|
8289
9318
|
deviceKey,
|
|
8290
9319
|
steps: pruneResult.steps
|
|
@@ -8318,7 +9347,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8318
9347
|
* microseconds instead of a 3-retry download backoff).
|
|
8319
9348
|
*/
|
|
8320
9349
|
assertStepsRunnable(steps, format, deviceId, enginesTried) {
|
|
8321
|
-
const zeroBuild = collectZeroBuildIssues(flattenSteps(steps), format);
|
|
9350
|
+
const zeroBuild = collectZeroBuildIssues(flattenSteps(steps), format, this.mergedStepDefOrThrow);
|
|
8322
9351
|
if (zeroBuild.length === 0) return;
|
|
8323
9352
|
const detail = zeroBuild.map((issue) => {
|
|
8324
9353
|
const step = flattenSteps(steps).find((s) => s.addonId === issue.addonId);
|
|
@@ -8337,53 +9366,235 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8337
9366
|
throw new Error(message);
|
|
8338
9367
|
}
|
|
8339
9368
|
/**
|
|
8340
|
-
*
|
|
8341
|
-
*
|
|
8342
|
-
*
|
|
8343
|
-
*
|
|
8344
|
-
*
|
|
8345
|
-
*
|
|
8346
|
-
*
|
|
8347
|
-
*/
|
|
8348
|
-
modelLoadInFlight =
|
|
9369
|
+
* In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
|
|
9370
|
+
* needing the same models on the same pool share ONE load and its outcome;
|
|
9371
|
+
* a dispatch needing a different set is never handed another set's failure
|
|
9372
|
+
* — it queues behind the pool's current load ({@link modelLoadTail}) and
|
|
9373
|
+
* then decides for itself. Keyed by pool, not node-wide: one GPU compile
|
|
9374
|
+
* that never returned held EVERY load on the node — the NPU's and the CPU's
|
|
9375
|
+
* too — behind it.
|
|
9376
|
+
*/
|
|
9377
|
+
modelLoadInFlight = /* @__PURE__ */ new Map();
|
|
9378
|
+
/** The last load issued on each pool — loads into one pool run one at a time. */
|
|
9379
|
+
modelLoadTail = /* @__PURE__ */ new Map();
|
|
9380
|
+
/** Stands in for "the node-default pool" before it exists. */
|
|
9381
|
+
nullPoolKey = {};
|
|
9382
|
+
/** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
|
|
9383
|
+
loadGovernor = new ModelLoadGovernor();
|
|
9384
|
+
/**
|
|
9385
|
+
* Load a dispatch's models ONE STEP AT A TIME (D657), and answer which steps
|
|
9386
|
+
* did not load. Never throws: each step is its own load, joined, backed off
|
|
9387
|
+
* and charged on its own model, so a step whose model cannot load on this
|
|
9388
|
+
* pool — no build for its format, a compile that failed — costs that step,
|
|
9389
|
+
* never the root or a sibling. The caller decides what the dispatch runs
|
|
9390
|
+
* without it ({@link stepsSurvivingLoad}).
|
|
9391
|
+
*/
|
|
9392
|
+
async loadDispatchModels(steps, factory, format, context, policy) {
|
|
9393
|
+
const failures = [];
|
|
9394
|
+
const walk = async (nodes, root) => {
|
|
9395
|
+
for (const step of nodes) {
|
|
9396
|
+
if (!step.enabled || step.slot === "audio-classifier") continue;
|
|
9397
|
+
if (factory === null || factory.needsPoolUpdate(step)) try {
|
|
9398
|
+
await this.ensureModelsForSteps([{
|
|
9399
|
+
...step,
|
|
9400
|
+
children: []
|
|
9401
|
+
}], factory, format, {
|
|
9402
|
+
...context,
|
|
9403
|
+
outcome: policy === "all-or-nothing" ? "call-failed" : root ? "root-lost" : "step-dropped"
|
|
9404
|
+
});
|
|
9405
|
+
} catch (err) {
|
|
9406
|
+
failures.push({
|
|
9407
|
+
step,
|
|
9408
|
+
error: err
|
|
9409
|
+
});
|
|
9410
|
+
continue;
|
|
9411
|
+
}
|
|
9412
|
+
await walk(step.children ?? [], false);
|
|
9413
|
+
}
|
|
9414
|
+
};
|
|
9415
|
+
await walk(steps, true);
|
|
9416
|
+
return failures;
|
|
9417
|
+
}
|
|
8349
9418
|
/** Ensure all models needed by steps are downloaded and loaded in the engine pool.
|
|
8350
9419
|
* `factory`/`format` default to the node's engine; a per-device dispatch passes
|
|
8351
|
-
* that device's factory + format (Phase 2 multi-device).
|
|
8352
|
-
|
|
8353
|
-
|
|
8354
|
-
const
|
|
9420
|
+
* that device's factory + format (Phase 2 multi-device). `context` names the
|
|
9421
|
+
* camera and the accelerator on the line a rejected load writes. */
|
|
9422
|
+
async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
|
|
9423
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9424
|
+
if (needed.length === 0) return;
|
|
9425
|
+
const pool = factory ?? this.nullPoolKey;
|
|
9426
|
+
const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
|
|
9427
|
+
const models = needed.map(stepModelKey);
|
|
9428
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9429
|
+
if (refusal !== null) {
|
|
9430
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
|
|
9431
|
+
throw refusal;
|
|
9432
|
+
}
|
|
9433
|
+
const setKey = [...models].sort().join(",");
|
|
9434
|
+
let perPool = this.modelLoadInFlight.get(pool);
|
|
9435
|
+
if (perPool === void 0) {
|
|
9436
|
+
perPool = /* @__PURE__ */ new Map();
|
|
9437
|
+
this.modelLoadInFlight.set(pool, perPool);
|
|
9438
|
+
}
|
|
9439
|
+
const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
|
|
9440
|
+
try {
|
|
9441
|
+
await flight.work;
|
|
9442
|
+
} catch (err) {
|
|
9443
|
+
if (err instanceof ModelLoadRefusedError) {
|
|
9444
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
|
|
9445
|
+
throw err;
|
|
9446
|
+
}
|
|
9447
|
+
const failed = failedModelOf(err) ?? models[0] ?? "";
|
|
9448
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, require_dist.errMsg(err));
|
|
9449
|
+
throw err;
|
|
9450
|
+
}
|
|
9451
|
+
}
|
|
9452
|
+
/**
|
|
9453
|
+
* Why `models` may not be loaded on `pool` right now — a back-off after a
|
|
9454
|
+
* failure, or a refusal by name after repeated compile timeouts — or `null`.
|
|
9455
|
+
*/
|
|
9456
|
+
loadRefusal(pool, deviceKey, models) {
|
|
9457
|
+
const verdict = this.loadGovernor.check(pool, deviceKey, models);
|
|
9458
|
+
if (verdict.kind === "go") return null;
|
|
9459
|
+
return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
|
|
9460
|
+
}
|
|
9461
|
+
/**
|
|
9462
|
+
* A compile that outlived its soft bound came back on `factory`'s pool
|
|
9463
|
+
* (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
|
|
9464
|
+
* again, so that model's back-off is lifted at once rather than waited out;
|
|
9465
|
+
* no other model's is. A FAILURE is one more failure of that model, and its
|
|
9466
|
+
* back-off grows.
|
|
9467
|
+
*/
|
|
9468
|
+
noteLateCompile(factory, deviceKey, event) {
|
|
9469
|
+
const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
|
|
9470
|
+
if (event.ok) {
|
|
9471
|
+
this.log.info("model loadable again after late compile", { meta: {
|
|
9472
|
+
deviceKey,
|
|
9473
|
+
model: event.model,
|
|
9474
|
+
backoffLifted: models
|
|
9475
|
+
} });
|
|
9476
|
+
return;
|
|
9477
|
+
}
|
|
9478
|
+
this.log.warn("late compile failed — the model stays backed off", { meta: {
|
|
9479
|
+
deviceKey,
|
|
9480
|
+
model: event.model,
|
|
9481
|
+
backoffExtended: models,
|
|
9482
|
+
error: event.error ?? null
|
|
9483
|
+
} });
|
|
9484
|
+
}
|
|
9485
|
+
/** The enabled steps the factory's pool does not reflect yet. */
|
|
9486
|
+
stepsNeedingLoad(steps, factory) {
|
|
8355
9487
|
const needed = [];
|
|
8356
|
-
for (const step of
|
|
9488
|
+
for (const step of flattenSteps(steps)) {
|
|
8357
9489
|
if (!step.enabled) continue;
|
|
8358
9490
|
if (factory !== null && !factory.needsPoolUpdate(step)) continue;
|
|
8359
9491
|
needed.push(step);
|
|
8360
9492
|
}
|
|
8361
|
-
|
|
8362
|
-
|
|
8363
|
-
|
|
8364
|
-
|
|
8365
|
-
|
|
8366
|
-
|
|
8367
|
-
|
|
8368
|
-
|
|
8369
|
-
|
|
8370
|
-
|
|
8371
|
-
|
|
8372
|
-
|
|
8373
|
-
|
|
8374
|
-
|
|
9493
|
+
return needed;
|
|
9494
|
+
}
|
|
9495
|
+
/**
|
|
9496
|
+
* Issue ONE load for a model set on a pool, queued behind the pool's
|
|
9497
|
+
* previous load, and record its outcome with the governor. The set is
|
|
9498
|
+
* re-derived once the queue reaches it: the load ahead may have brought
|
|
9499
|
+
* some of it in.
|
|
9500
|
+
*/
|
|
9501
|
+
issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
|
|
9502
|
+
const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
|
|
9503
|
+
const flight = {
|
|
9504
|
+
work: Promise.resolve(),
|
|
9505
|
+
generation: null
|
|
9506
|
+
};
|
|
9507
|
+
flight.work = (async () => {
|
|
9508
|
+
await previous;
|
|
9509
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9510
|
+
if (needed.length === 0) return;
|
|
9511
|
+
const models = needed.map(stepModelKey);
|
|
9512
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9513
|
+
if (refusal !== null) throw refusal;
|
|
9514
|
+
try {
|
|
9515
|
+
await this.downloadNeededModels(needed, format);
|
|
9516
|
+
this.log.info("Loading additional models for benchmark", { meta: {
|
|
9517
|
+
count: needed.length,
|
|
9518
|
+
models,
|
|
9519
|
+
deviceKey
|
|
9520
|
+
} });
|
|
9521
|
+
await factory.loadAdditional(needed);
|
|
9522
|
+
} catch (err) {
|
|
9523
|
+
flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
|
|
9524
|
+
throw err;
|
|
8375
9525
|
}
|
|
8376
|
-
this.
|
|
8377
|
-
count: needed.length,
|
|
8378
|
-
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8379
|
-
} });
|
|
8380
|
-
await factory.loadAdditional(needed);
|
|
9526
|
+
this.loadGovernor.recordSuccess(pool, deviceKey, models);
|
|
8381
9527
|
})();
|
|
8382
|
-
|
|
8383
|
-
|
|
8384
|
-
|
|
8385
|
-
|
|
8386
|
-
if (
|
|
9528
|
+
perPool.set(setKey, flight);
|
|
9529
|
+
const tail = flight.work.catch(() => void 0);
|
|
9530
|
+
this.modelLoadTail.set(pool, tail);
|
|
9531
|
+
tail.then(() => {
|
|
9532
|
+
if (perPool.get(setKey) === flight) perPool.delete(setKey);
|
|
9533
|
+
if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
|
|
9534
|
+
});
|
|
9535
|
+
return flight;
|
|
9536
|
+
}
|
|
9537
|
+
/** Charge a failed load to the model that failed; say so once if it is now refused. */
|
|
9538
|
+
recordModelLoadFailure(pool, deviceKey, models, err) {
|
|
9539
|
+
const named = failedModelOf(err);
|
|
9540
|
+
const failedModels = named !== null ? [named] : models;
|
|
9541
|
+
const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
|
|
9542
|
+
const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
|
|
9543
|
+
let generation = 0;
|
|
9544
|
+
for (const model of failedModels) {
|
|
9545
|
+
const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, require_dist.errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
|
|
9546
|
+
generation = outcome.generation;
|
|
9547
|
+
if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
|
|
9548
|
+
deviceKey,
|
|
9549
|
+
model,
|
|
9550
|
+
compileTimeouts: outcome.compileTimeouts,
|
|
9551
|
+
error: require_dist.errMsg(err),
|
|
9552
|
+
device: "keeps serving every other model",
|
|
9553
|
+
rearm: "pipelineExecutor.rearmInferenceDevice"
|
|
9554
|
+
} });
|
|
9555
|
+
}
|
|
9556
|
+
return generation;
|
|
9557
|
+
}
|
|
9558
|
+
/**
|
|
9559
|
+
* One WARN per camera per state change of the model that cost it its frame.
|
|
9560
|
+
* The ERROR for the failed load itself is written once, where the load
|
|
9561
|
+
* failed (`Step variant load failed`); this line answers "which cameras
|
|
9562
|
+
* did it cost?", tagged so it can be counted per camera.
|
|
9563
|
+
*/
|
|
9564
|
+
reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
|
|
9565
|
+
if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
|
|
9566
|
+
this.log.warn("model load for dispatch failed", {
|
|
9567
|
+
...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
|
|
9568
|
+
meta: {
|
|
9569
|
+
deviceKey,
|
|
9570
|
+
models,
|
|
9571
|
+
model,
|
|
9572
|
+
state,
|
|
9573
|
+
error,
|
|
9574
|
+
...context.outcome !== void 0 ? { outcome: context.outcome } : {}
|
|
9575
|
+
}
|
|
9576
|
+
});
|
|
9577
|
+
}
|
|
9578
|
+
/** Download any model of `needed` missing in `format` (the device's). */
|
|
9579
|
+
async downloadNeededModels(needed, format) {
|
|
9580
|
+
for (const step of needed) try {
|
|
9581
|
+
await this.ensureStepModelArtifact(step, format);
|
|
9582
|
+
} catch (err) {
|
|
9583
|
+
throw new StepModelArtifactError(step.addonId, step.modelId, err);
|
|
9584
|
+
}
|
|
9585
|
+
}
|
|
9586
|
+
/** Make one step's model artifact present on disk for `format`, or say why it cannot be. */
|
|
9587
|
+
async ensureStepModelArtifact(step, format) {
|
|
9588
|
+
const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
|
|
9589
|
+
if (modelEntry && !(0, _camstack_system_addon_utils.isModelDownloaded)(this.modelsDir, modelEntry, format)) {
|
|
9590
|
+
if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
|
|
9591
|
+
if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
|
|
9592
|
+
this.log.info("Downloading model for step", { meta: {
|
|
9593
|
+
modelId: step.modelId,
|
|
9594
|
+
format,
|
|
9595
|
+
step: step.addonId
|
|
9596
|
+
} });
|
|
9597
|
+
await this.downloadWithRetry(modelEntry, format, 3);
|
|
8387
9598
|
}
|
|
8388
9599
|
}
|
|
8389
9600
|
/** Download a model with retry + exponential backoff */
|
|
@@ -8481,6 +9692,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8481
9692
|
}
|
|
8482
9693
|
async runPipelineBatchImpl(input) {
|
|
8483
9694
|
if (input.frames.length === 0) return { results: [] };
|
|
9695
|
+
await this.getCustomModels();
|
|
8484
9696
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
8485
9697
|
steps: input.steps,
|
|
8486
9698
|
deviceKey: input.deviceKey,
|
|
@@ -8511,7 +9723,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8511
9723
|
await this.ensureEngineFactory();
|
|
8512
9724
|
const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8513
9725
|
if (!factory) throw new Error("runPipelineBatch: factory not initialised");
|
|
8514
|
-
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat);
|
|
9726
|
+
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
|
|
8515
9727
|
const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
|
|
8516
9728
|
this.log.info("runPipelineBatch path decision", { meta: {
|
|
8517
9729
|
phase: "batch",
|
|
@@ -8723,9 +9935,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8723
9935
|
return this.engineFactory.cacheFrameInPool(buf, input.width, input.height, input.format);
|
|
8724
9936
|
}
|
|
8725
9937
|
async inferCached(input) {
|
|
9938
|
+
if (typeof input.modelId !== "string" || input.modelId === "") throw new Error(`inferCached: no modelId for step "${input.stepId}" — refusing rather than running the step's active variant`);
|
|
8726
9939
|
await this.ensureEngineFactory();
|
|
8727
9940
|
if (!this.engineFactory) throw new Error("inferCached: factory not initialized");
|
|
8728
|
-
return this.engineFactory.inferCached(input.stepId, input.frameId);
|
|
9941
|
+
return this.engineFactory.inferCached(input.stepId, input.frameId, input.modelId);
|
|
8729
9942
|
}
|
|
8730
9943
|
async uncacheFrame(input) {
|
|
8731
9944
|
if (!this.engineFactory) return;
|
|
@@ -8766,7 +9979,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8766
9979
|
if (this.engineFactory.isReady()) return;
|
|
8767
9980
|
const dead = this.engineFactory;
|
|
8768
9981
|
this.engineFactory = null;
|
|
8769
|
-
this.
|
|
9982
|
+
this.noteFactoryDeath(defaultDeviceKey, dead);
|
|
8770
9983
|
await dead.dispose().catch(() => void 0);
|
|
8771
9984
|
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8772
9985
|
}
|
|
@@ -8784,7 +9997,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8784
9997
|
logger: this.log.child("engine"),
|
|
8785
9998
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8786
9999
|
provisioning: this.executorOptions.provisioning,
|
|
8787
|
-
resolveCustomModel: this.customModelResolver
|
|
10000
|
+
resolveCustomModel: this.customModelResolver,
|
|
10001
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
|
|
8788
10002
|
});
|
|
8789
10003
|
try {
|
|
8790
10004
|
await factory.initialize([]);
|
|
@@ -8904,15 +10118,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8904
10118
|
* dispose — which rejects whatever was still in flight on it with a reason
|
|
8905
10119
|
* rather than letting those requests sit until their own deadlines.
|
|
8906
10120
|
*/
|
|
8907
|
-
async condemnDeviceFactory(deviceKey, factory
|
|
10121
|
+
async condemnDeviceFactory(deviceKey, factory) {
|
|
8908
10122
|
if (this.factoriesByDevice.get(deviceKey) === factory) {
|
|
8909
10123
|
this.factoriesByDevice.delete(deviceKey);
|
|
8910
10124
|
this.deviceReaper.cancel(deviceKey);
|
|
8911
|
-
this.
|
|
10125
|
+
this.noteFactoryDeath(deviceKey, factory);
|
|
8912
10126
|
}
|
|
8913
10127
|
await factory.dispose().catch(() => void 0);
|
|
8914
10128
|
}
|
|
8915
10129
|
/**
|
|
10130
|
+
* Charge one dead pool to its device — unless it died of a compile that was
|
|
10131
|
+
* still running at its hard bound (D653, fix round 1). That death is not a
|
|
10132
|
+
* crash of the DEVICE: the model that hung was already counted when its
|
|
10133
|
+
* load answered `compile-timeout`, and the same model timing out again on
|
|
10134
|
+
* the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
|
|
10135
|
+
* the respawns. A crash stays a crash.
|
|
10136
|
+
*/
|
|
10137
|
+
noteFactoryDeath(deviceKey, factory) {
|
|
10138
|
+
const cause = factory.getDeathCause();
|
|
10139
|
+
if (cause !== null && !cause.chargesDeviceBudget) {
|
|
10140
|
+
this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
|
|
10141
|
+
deviceKey,
|
|
10142
|
+
reason: cause.reason,
|
|
10143
|
+
cause: cause.message
|
|
10144
|
+
} });
|
|
10145
|
+
return;
|
|
10146
|
+
}
|
|
10147
|
+
this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
|
|
10148
|
+
}
|
|
10149
|
+
/**
|
|
8916
10150
|
* Every inference device this node currently refuses, and why.
|
|
8917
10151
|
*
|
|
8918
10152
|
* The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
|
|
@@ -8947,7 +10181,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8947
10181
|
state: "backoff",
|
|
8948
10182
|
since: Date.now(),
|
|
8949
10183
|
deaths: 0,
|
|
8950
|
-
lastError:
|
|
10184
|
+
lastError: deathReasonOf(factory)
|
|
8951
10185
|
});
|
|
8952
10186
|
}
|
|
8953
10187
|
return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
|
|
@@ -8964,10 +10198,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8964
10198
|
* no-op, reported as such.
|
|
8965
10199
|
*/
|
|
8966
10200
|
async rearmInferenceDevice(input) {
|
|
8967
|
-
const
|
|
10201
|
+
const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
|
|
10202
|
+
const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
|
|
10203
|
+
const rearmed = rearmedDevice || rearmedModels > 0;
|
|
8968
10204
|
this.log.info("inference device re-armed by operator", { meta: {
|
|
8969
10205
|
deviceKey: input.deviceKey,
|
|
8970
|
-
rearmed
|
|
10206
|
+
rearmed,
|
|
10207
|
+
rearmedModels
|
|
8971
10208
|
} });
|
|
8972
10209
|
return { rearmed };
|
|
8973
10210
|
}
|
|
@@ -8984,7 +10221,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8984
10221
|
this.deviceReaper.touch(deviceKey);
|
|
8985
10222
|
return existing;
|
|
8986
10223
|
}
|
|
8987
|
-
await this.condemnDeviceFactory(deviceKey, existing
|
|
10224
|
+
await this.condemnDeviceFactory(deviceKey, existing);
|
|
8988
10225
|
this.refuseIfDeviceUnusable(deviceKey);
|
|
8989
10226
|
}
|
|
8990
10227
|
const inflight = this.deviceFactoryInflight.get(deviceKey);
|
|
@@ -8997,7 +10234,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8997
10234
|
logger: this.log.child(`engine:${deviceKey}`),
|
|
8998
10235
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8999
10236
|
provisioning: this.executorOptions.provisioning,
|
|
9000
|
-
resolveCustomModel: this.customModelResolver
|
|
10237
|
+
resolveCustomModel: this.customModelResolver,
|
|
10238
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
|
|
9001
10239
|
});
|
|
9002
10240
|
try {
|
|
9003
10241
|
await factory.initialize([]);
|
|
@@ -9340,7 +10578,7 @@ function buildSchemaSlots(formats, modelsDir, customByStep) {
|
|
|
9340
10578
|
const formatList = [...formats];
|
|
9341
10579
|
for (const pipelineStep of require_default_detection_model.ALL_PIPELINE_STEPS) {
|
|
9342
10580
|
const step = pipelineStep.definition;
|
|
9343
|
-
const availableModels = mergeCustomModels(step.models, customByStep?.get(step.id) ?? []).filter((m) => m.legacy !== true && formatList.some((f) => m.formats[f]));
|
|
10581
|
+
const availableModels = require_default_detection_model.mergeCustomModels(step.models, customByStep?.get(step.id) ?? []).filter((m) => m.legacy !== true && formatList.some((f) => m.formats[f]));
|
|
9344
10582
|
if (availableModels.length === 0) continue;
|
|
9345
10583
|
const slot = step.slot;
|
|
9346
10584
|
if (!slotMap.has(slot)) slotMap.set(slot, []);
|
|
@@ -9504,13 +10742,22 @@ function isDetailPlaneStep(addonId) {
|
|
|
9504
10742
|
}
|
|
9505
10743
|
}
|
|
9506
10744
|
/**
|
|
9507
|
-
*
|
|
9508
|
-
*
|
|
9509
|
-
*
|
|
9510
|
-
*
|
|
9511
|
-
|
|
9512
|
-
|
|
9513
|
-
|
|
10745
|
+
* What a dispatch runs after its per-step loads (D657).
|
|
10746
|
+
*
|
|
10747
|
+
* An ALL-OR-NOTHING call (a benchmark, a replay — D56 — a capacity ramp) fails
|
|
10748
|
+
* on any failure, naming every step it lost ({@link DispatchStepsNotLoadedError}):
|
|
10749
|
+
* it measures or reproduces exactly what it names. A camera's LIVE dispatch
|
|
10750
|
+
* runs the tree without the steps whose model did not load, and fails — with
|
|
10751
|
+
* the first root's own named error — only when no root survives: a dispatch
|
|
10752
|
+
* with no root has nothing to produce.
|
|
10753
|
+
*/
|
|
10754
|
+
function stepsSurvivingLoad(steps, failures, policy) {
|
|
10755
|
+
if (failures.length === 0) return steps;
|
|
10756
|
+
if (policy === "all-or-nothing") throw new DispatchStepsNotLoadedError(failures);
|
|
10757
|
+
const kept = withoutFailedSteps(steps, new Set(failures.map((f) => f.step)));
|
|
10758
|
+
if (kept.some((s) => s.enabled)) return kept;
|
|
10759
|
+
const roots = new Set(steps);
|
|
10760
|
+
throw (failures.find((f) => roots.has(f.step)) ?? failures[0]).error;
|
|
9514
10761
|
}
|
|
9515
10762
|
/**
|
|
9516
10763
|
* The union of model formats a step's catalog ships ANY build for — the
|