@camstack/addon-pipeline 1.2.295 → 1.2.297
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD_PARTY_MODELS.md +8 -0
- package/dist/audio-analyzer/index.js +2 -2
- package/dist/audio-analyzer/index.mjs +2 -2
- package/dist/{default-detection-model-wNEISVA9.js → default-detection-model-B2n_SFXt.js} +442 -50
- package/dist/{default-detection-model-CAUBVbgK.mjs → default-detection-model-D6DypCkI.mjs} +413 -51
- package/dist/detection-pipeline/index.js +1516 -269
- package/dist/detection-pipeline/index.mjs +1515 -268
- package/dist/{dist-CuxSNLKW.mjs → dist-CxHIpsQs.mjs} +834 -398
- package/dist/{dist-BsVcf5wO.js → dist-DUqr1zyq.js} +845 -409
- package/dist/motion-wasm/index.js +1 -1
- package/dist/motion-wasm/index.mjs +1 -1
- package/dist/{node-UFk6I2f6.js → node-Dx6DLQ1j.js} +1 -1
- package/dist/{node-Cqb1QiXk.mjs → node-DyeWu78a.mjs} +1 -1
- package/dist/pipeline-runner/index.js +906 -310
- package/dist/pipeline-runner/index.mjs +906 -310
- package/dist/{process-memory-D0Nvs9rr.mjs → process-memory-BVR4592X.mjs} +1 -1
- package/dist/{process-memory-CJV29sPf.js → process-memory-CMJ3NY-s.js} +1 -1
- package/dist/recorder/index.js +4 -6
- package/dist/recorder/index.mjs +4 -6
- package/dist/{segment-demux-js-DGwmf5hu.js → segment-demux-js-CmGIF_uK.js} +1 -1
- package/dist/{segment-demux-js-DIDWw1iE.mjs → segment-demux-js-Dzga-XgE.mjs} +1 -1
- package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
- package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
- package/dist/stream-broker/_stub.js +2 -2
- package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-B7nBFqva.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-P71ze8cu.mjs} +2 -2
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CZkpFU-J.mjs +26 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-Dti8Ex88.mjs +26 -0
- package/dist/stream-broker/demux-worker-child.js +1 -1
- package/dist/stream-broker/demux-worker-child.mjs +1 -1
- package/dist/stream-broker/{hostInit-BKlb3qac.mjs → hostInit-CRlStbz4.mjs} +2 -2
- package/dist/stream-broker/index.js +4 -4
- package/dist/stream-broker/index.mjs +4 -4
- package/dist/stream-broker/remoteEntry.js +1 -1
- package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
- package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
- package/package.json +1 -1
- package/python/inference_pool.py +422 -64
- package/python/test_inference_pool_compile_off_loop.py +414 -0
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CCIyBvRa.mjs +0 -26
- package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DeD_UFdb.mjs +0 -26
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { n as __require } from "../chunk-DnnnRqeS.mjs";
|
|
2
|
-
import {
|
|
3
|
-
import { s as readProcessCost } from "../node-
|
|
4
|
-
import { a as
|
|
2
|
+
import { Dn as array, Dt as overlayClusterStepSettings, Gt as runtimeDevices$1, H as YAMNET_TO_MACRO, In as string, Kt as stepModelRunsOnFormat, Ln as union, Mt as pipelineExecutorCapability, Nn as object, Rn as EventCategory, Sn as parseJsonUnknown, Tn as sleep, Ut as resolvePoolMemoryPolicy, Zt as supportedRuntimes$1, _n as hydrateSchema, en as errMsg, et as defaultDeviceFor$1, it as detectionPipelineCapability, j as PoolMemoryWatchdog, kt as pickClusterStepSettings, lt as evaluateZoneRules, m as DEFAULT_CLUSTER_STEP_SETTINGS, pn as createEvent, sn as BaseAddon, st as enumerateInferenceDevices, t as APPLE_SA_TO_MACRO, u as CLUSTER_MODEL_SCOPED_STEPS, vt as loadContributionCapability, xn as nodePin, y as DEVICE_BACKEND_TO_FORMAT, zt as resolveClusterStepModelId } from "../dist-CxHIpsQs.mjs";
|
|
3
|
+
import { s as readProcessCost } from "../node-DyeWu78a.mjs";
|
|
4
|
+
import { _ as resolveModelForFormat, a as buildSubtreeCanRun, c as macroGateVerdict, f as ALL_PIPELINE_STEPS, g as getStepDefinition, h as getStep, i as mergeCustomModels, m as getDefaultModelForFormat, n as resolveDefaultDetectionModel, p as ALL_STEPS, r as CustomModelCatalog, u as localFrameRegistry, y as landmarkPrecisionVerdict } from "../default-detection-model-D6DypCkI.mjs";
|
|
5
5
|
import { t as getSharp } from "../lazy-sharp-DXsqwpph.mjs";
|
|
6
|
-
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-
|
|
6
|
+
import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-BVR4592X.mjs";
|
|
7
7
|
import * as os from "node:os";
|
|
8
8
|
import * as fs from "node:fs";
|
|
9
9
|
import * as path$1 from "node:path";
|
|
@@ -360,12 +360,14 @@ function normalizeEngineNodeId(rawNodeId) {
|
|
|
360
360
|
* builders eventually reaches ten of them, and the cameras attached by the
|
|
361
361
|
* forgotten one quietly write a foreign feature space into the shared index.
|
|
362
362
|
*
|
|
363
|
-
* ## Why the failure mode keeps the last value
|
|
363
|
+
* ## Why the failure mode keeps the last value — and refuses before the first
|
|
364
364
|
*
|
|
365
365
|
* A failed read must never revert to the registry default MID-PASS: that would
|
|
366
366
|
* silently change what every remaining vector of that dispatch means, which is
|
|
367
367
|
* the exact class of harm the scope exists to prevent. So a failure keeps the
|
|
368
|
-
* last known row and says so,
|
|
368
|
+
* last known row and says so, once per change. Before the FIRST answer there is
|
|
369
|
+
* no row to keep: the step is refused by name (D658) — a default run then made
|
|
370
|
+
* the default model the pool's active variant for the process lifetime.
|
|
369
371
|
*/
|
|
370
372
|
/**
|
|
371
373
|
* The addon whose GLOBAL settings own the cluster row. Same owner as the crop
|
|
@@ -374,35 +376,58 @@ function normalizeEngineNodeId(rawNodeId) {
|
|
|
374
376
|
* enforces it live — neither should own a value the other depends on).
|
|
375
377
|
*/
|
|
376
378
|
var CLUSTER_MODEL_OWNER_ADDON_ID = "pipeline-orchestrator";
|
|
377
|
-
/**
|
|
379
|
+
/**
|
|
380
|
+
* TTL-cached read of the cluster model row.
|
|
381
|
+
*
|
|
382
|
+
* A step's model is KNOWN only once the owner has answered for it (`stored` or
|
|
383
|
+
* `absent`). Until then it is UNKNOWN, and a cluster-scoped step is refused by
|
|
384
|
+
* name rather than run with the registry default (D49, D658): a default run
|
|
385
|
+
* during the owner's boot window made that model the pool's active variant for
|
|
386
|
+
* the whole process, and the face default was ArcFace until D651. Once known,
|
|
387
|
+
* an unanswered read keeps the LAST KNOWN model — never reverts to a default.
|
|
388
|
+
*/
|
|
378
389
|
var ClusterModelSource = class {
|
|
379
|
-
models =
|
|
390
|
+
models = {};
|
|
391
|
+
unknown = new Map(CLUSTER_MODEL_SCOPED_STEPS.map((s) => [s.stepId, "not-yet-asked"]));
|
|
380
392
|
settings = DEFAULT_CLUSTER_STEP_SETTINGS;
|
|
381
393
|
lastReadAtMs = Number.NEGATIVE_INFINITY;
|
|
382
394
|
inFlight = null;
|
|
383
395
|
logger;
|
|
384
396
|
now;
|
|
385
397
|
ttlMs;
|
|
398
|
+
unknownRetryMs;
|
|
399
|
+
/** The last "not known" state logged, so it is said once per change. */
|
|
400
|
+
lastUnknownLogKey = "";
|
|
386
401
|
constructor(options) {
|
|
387
402
|
this.logger = options.logger;
|
|
388
403
|
this.now = options.now ?? (() => Date.now());
|
|
389
404
|
this.ttlMs = options.ttlMs ?? 6e4;
|
|
405
|
+
this.unknownRetryMs = options.unknownRetryMs ?? 5e3;
|
|
390
406
|
}
|
|
391
|
-
/**
|
|
407
|
+
/**
|
|
408
|
+
* The KNOWN models, `stepId → modelId`. A step absent here is unknown (see
|
|
409
|
+
* {@link unknownSteps}) and must be refused, never defaulted.
|
|
410
|
+
*/
|
|
392
411
|
current() {
|
|
393
412
|
return this.models;
|
|
394
413
|
}
|
|
414
|
+
/** Cluster-scoped steps whose model the owner has not answered yet, with why. */
|
|
415
|
+
unknownSteps() {
|
|
416
|
+
return this.unknown;
|
|
417
|
+
}
|
|
395
418
|
/** Cluster-scoped step knobs in force right now (admission floors, …). */
|
|
396
419
|
currentSettings() {
|
|
397
420
|
return this.settings;
|
|
398
421
|
}
|
|
399
422
|
/**
|
|
400
|
-
* Re-read if the cached row has expired
|
|
401
|
-
* in-flight read; a failure keeps the LAST
|
|
423
|
+
* Re-read if the cached row has expired (sooner while a step is unknown).
|
|
424
|
+
* Concurrent callers share one in-flight read; a failure keeps the LAST
|
|
425
|
+
* KNOWN row.
|
|
402
426
|
*/
|
|
403
427
|
async refresh(api) {
|
|
404
428
|
if (api === void 0) return this.models;
|
|
405
|
-
|
|
429
|
+
const ttl = this.unknown.size > 0 ? Math.min(this.ttlMs, this.unknownRetryMs) : this.ttlMs;
|
|
430
|
+
if (this.now() - this.lastReadAtMs < ttl) return this.models;
|
|
406
431
|
this.inFlight ??= this.read(api).finally(() => {
|
|
407
432
|
this.inFlight = null;
|
|
408
433
|
});
|
|
@@ -410,38 +435,186 @@ var ClusterModelSource = class {
|
|
|
410
435
|
return this.models;
|
|
411
436
|
}
|
|
412
437
|
async read(api) {
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
to: next[stepId]
|
|
429
|
-
})) } });
|
|
430
|
-
this.models = next;
|
|
431
|
-
}
|
|
432
|
-
if (settingsChanged) {
|
|
433
|
-
this.logger.info("cluster step settings changed", { meta: {
|
|
434
|
-
from: { ...this.settings },
|
|
435
|
-
to: { ...nextSettings }
|
|
436
|
-
} });
|
|
437
|
-
this.settings = nextSettings;
|
|
438
|
+
this.lastReadAtMs = this.now();
|
|
439
|
+
await Promise.all([this.readModels(api), this.readSettings(api)]);
|
|
440
|
+
}
|
|
441
|
+
async readModels(api) {
|
|
442
|
+
const answers = await Promise.all(CLUSTER_MODEL_SCOPED_STEPS.map(async (step) => ({
|
|
443
|
+
stepId: step.stepId,
|
|
444
|
+
answer: await askOwner(api, step.stepId)
|
|
445
|
+
})));
|
|
446
|
+
const nextModels = { ...this.models };
|
|
447
|
+
const nextUnknown = /* @__PURE__ */ new Map();
|
|
448
|
+
const unanswered = [];
|
|
449
|
+
for (const { stepId, answer } of answers) {
|
|
450
|
+
if (answer.kind === "model") {
|
|
451
|
+
nextModels[stepId] = answer.modelId;
|
|
452
|
+
continue;
|
|
438
453
|
}
|
|
454
|
+
const keeping = this.models[stepId] ?? null;
|
|
455
|
+
unanswered.push({
|
|
456
|
+
stepId,
|
|
457
|
+
reason: answer.reason,
|
|
458
|
+
keeping
|
|
459
|
+
});
|
|
460
|
+
if (keeping === null) nextUnknown.set(stepId, answer.reason);
|
|
461
|
+
}
|
|
462
|
+
const changed = Object.keys(nextModels).filter((stepId) => this.models[stepId] !== void 0 && nextModels[stepId] !== this.models[stepId]);
|
|
463
|
+
if (changed.length > 0) this.logger.info("cluster step model changed — vectors already stored by the previous model need a re-embed pass", { meta: { changed: changed.map((stepId) => ({
|
|
464
|
+
stepId,
|
|
465
|
+
from: this.models[stepId],
|
|
466
|
+
to: nextModels[stepId]
|
|
467
|
+
})) } });
|
|
468
|
+
const learned = Object.keys(nextModels).filter((stepId) => this.models[stepId] === void 0);
|
|
469
|
+
if (learned.length > 0) this.logger.info("cluster step model known — cluster-scoped steps run it", { meta: { known: Object.fromEntries(learned.map((id) => [id, nextModels[id]])) } });
|
|
470
|
+
this.models = nextModels;
|
|
471
|
+
this.unknown = nextUnknown;
|
|
472
|
+
this.reportUnanswered(unanswered);
|
|
473
|
+
}
|
|
474
|
+
/** Once per change of the unanswered set — never once per retry. */
|
|
475
|
+
reportUnanswered(unanswered) {
|
|
476
|
+
const key = JSON.stringify(unanswered);
|
|
477
|
+
if (key === this.lastUnknownLogKey) return;
|
|
478
|
+
this.lastUnknownLogKey = key;
|
|
479
|
+
if (unanswered.length === 0) return;
|
|
480
|
+
this.logger.warn("cluster step model not answered by its owner — a step never known is REFUSED, a known one keeps its last model", { meta: {
|
|
481
|
+
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
482
|
+
unanswered: unanswered.map((u) => ({
|
|
483
|
+
stepId: u.stepId,
|
|
484
|
+
reason: u.reason,
|
|
485
|
+
...u.keeping !== null ? { keeping: u.keeping } : { refused: true }
|
|
486
|
+
}))
|
|
487
|
+
} });
|
|
488
|
+
}
|
|
489
|
+
async readSettings(api) {
|
|
490
|
+
let view;
|
|
491
|
+
try {
|
|
492
|
+
view = await api.addonSettings.getGlobalSettings.query({ addonId: CLUSTER_MODEL_OWNER_ADDON_ID });
|
|
439
493
|
} catch (err) {
|
|
440
|
-
this.logger.warn("cluster step
|
|
494
|
+
this.logger.warn("cluster step settings read failed — keeping the last known knobs", { meta: {
|
|
441
495
|
owner: CLUSTER_MODEL_OWNER_ADDON_ID,
|
|
442
|
-
inUse: { ...this.models },
|
|
443
496
|
error: err instanceof Error ? err.message : String(err)
|
|
444
497
|
} });
|
|
498
|
+
return;
|
|
499
|
+
}
|
|
500
|
+
if (view === null) return;
|
|
501
|
+
const nextSettings = pickClusterStepSettings(view);
|
|
502
|
+
if (Object.keys(nextSettings).some((stepId) => {
|
|
503
|
+
const from = this.settings[stepId] ?? {};
|
|
504
|
+
const to = nextSettings[stepId] ?? {};
|
|
505
|
+
return [...new Set([...Object.keys(from), ...Object.keys(to)])].some((key) => from[key] !== to[key]);
|
|
506
|
+
})) {
|
|
507
|
+
this.logger.info("cluster step settings changed", { meta: {
|
|
508
|
+
from: { ...this.settings },
|
|
509
|
+
to: { ...nextSettings }
|
|
510
|
+
} });
|
|
511
|
+
this.settings = nextSettings;
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
};
|
|
515
|
+
/** One step's model from the owner's three-way answer; a throw is no answer. */
|
|
516
|
+
async function askOwner(api, stepId) {
|
|
517
|
+
try {
|
|
518
|
+
const choice = await api.pipelineOrchestrator.getClusterModelChoice.query({ stepId });
|
|
519
|
+
switch (choice.kind) {
|
|
520
|
+
case "stored": return {
|
|
521
|
+
kind: "model",
|
|
522
|
+
modelId: choice.modelId
|
|
523
|
+
};
|
|
524
|
+
case "absent": return {
|
|
525
|
+
kind: "model",
|
|
526
|
+
modelId: choice.defaultModelId
|
|
527
|
+
};
|
|
528
|
+
case "unreadable": return {
|
|
529
|
+
kind: "none",
|
|
530
|
+
reason: choice.reason
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
} catch (err) {
|
|
534
|
+
return {
|
|
535
|
+
kind: "none",
|
|
536
|
+
reason: `read-failed: ${err instanceof Error ? err.message : String(err)}`
|
|
537
|
+
};
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
//#endregion
|
|
541
|
+
//#region src/detection-pipeline/cluster-model-switch-log.ts
|
|
542
|
+
function* clusterScopedSteps(steps) {
|
|
543
|
+
for (const step of steps) {
|
|
544
|
+
if (step.definition.modelScope === "cluster") yield step;
|
|
545
|
+
yield* clusterScopedSteps(step.children);
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
var ClusterModelSwitchLog = class {
|
|
549
|
+
logger;
|
|
550
|
+
/** `deviceId|stepId` → the model that step last ran for that camera. */
|
|
551
|
+
lastRun = /* @__PURE__ */ new Map();
|
|
552
|
+
constructor(logger) {
|
|
553
|
+
this.logger = logger;
|
|
554
|
+
}
|
|
555
|
+
/** Record what `tree` runs for `deviceId`; say so when a cluster step changed model. */
|
|
556
|
+
note(deviceId, tree) {
|
|
557
|
+
if (deviceId <= 0) return;
|
|
558
|
+
for (const step of clusterScopedSteps(tree.roots)) {
|
|
559
|
+
const key = `${String(deviceId)}|${step.stepId}`;
|
|
560
|
+
const previous = this.lastRun.get(key);
|
|
561
|
+
if (previous === step.modelId) continue;
|
|
562
|
+
this.lastRun.set(key, step.modelId);
|
|
563
|
+
if (previous === void 0) continue;
|
|
564
|
+
this.logger.info("cluster-scoped step re-provisioned for this camera — now runs the new model", {
|
|
565
|
+
tags: { deviceId },
|
|
566
|
+
meta: {
|
|
567
|
+
stepId: step.stepId,
|
|
568
|
+
from: previous,
|
|
569
|
+
to: step.modelId
|
|
570
|
+
}
|
|
571
|
+
});
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
};
|
|
575
|
+
function* enabledStepIds(steps) {
|
|
576
|
+
for (const step of steps) {
|
|
577
|
+
if (!step.enabled) continue;
|
|
578
|
+
yield step.addonId;
|
|
579
|
+
if (step.children) yield* enabledStepIds(step.children);
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
/** The enabled steps of `steps` (any depth) whose cluster model is unknown. */
|
|
583
|
+
function unknownStepsNamed(steps, unknown) {
|
|
584
|
+
if (unknown.size === 0) return [];
|
|
585
|
+
return [...enabledStepIds(steps)].filter((id) => unknown.has(id));
|
|
586
|
+
}
|
|
587
|
+
/**
|
|
588
|
+
* One line per (camera, step) while a cluster-scoped step is REFUSED because
|
|
589
|
+
* the owner has not answered which model the cluster runs (D658). Said once per
|
|
590
|
+
* episode: when every step becomes known the memory is cleared, so the next
|
|
591
|
+
* episode is said again.
|
|
592
|
+
*/
|
|
593
|
+
var ClusterUnknownRefusalLog = class {
|
|
594
|
+
logger;
|
|
595
|
+
said = /* @__PURE__ */ new Set();
|
|
596
|
+
constructor(logger) {
|
|
597
|
+
this.logger = logger;
|
|
598
|
+
}
|
|
599
|
+
note(deviceId, steps, unknown) {
|
|
600
|
+
if (unknown.size === 0) {
|
|
601
|
+
if (this.said.size > 0) this.said = /* @__PURE__ */ new Set();
|
|
602
|
+
return;
|
|
603
|
+
}
|
|
604
|
+
if (deviceId <= 0) return;
|
|
605
|
+
for (const stepId of enabledStepIds(steps)) {
|
|
606
|
+
const reason = unknown.get(stepId);
|
|
607
|
+
if (reason === void 0) continue;
|
|
608
|
+
const key = `${String(deviceId)}|${stepId}`;
|
|
609
|
+
if (this.said.has(key)) continue;
|
|
610
|
+
this.said = new Set(this.said).add(key);
|
|
611
|
+
this.logger.warn("cluster-scoped step refused for this camera — the owner has not answered which model the cluster runs", {
|
|
612
|
+
tags: { deviceId },
|
|
613
|
+
meta: {
|
|
614
|
+
stepId,
|
|
615
|
+
reason
|
|
616
|
+
}
|
|
617
|
+
});
|
|
445
618
|
}
|
|
446
619
|
}
|
|
447
620
|
};
|
|
@@ -562,6 +735,98 @@ var DeviceOverrideMirror = class {
|
|
|
562
735
|
}
|
|
563
736
|
};
|
|
564
737
|
//#endregion
|
|
738
|
+
//#region src/detection-pipeline/engine/pool-worker-health.ts
|
|
739
|
+
/**
|
|
740
|
+
* The one table of load bounds (D653, fix round 1).
|
|
741
|
+
*
|
|
742
|
+
* The soft bound decides how soon the operator HEARS about a slow compile; the
|
|
743
|
+
* hard bound decides when a compile is given up for hung. A single 120 s bound
|
|
744
|
+
* that also killed the compile was wrong twice: a slow but finite compile was
|
|
745
|
+
* killed before it could write its cache, so every respawn faced the same cold
|
|
746
|
+
* compile and three of them walked the device to `failed`; and on CoreML the
|
|
747
|
+
* bound was shorter than the cross-process compile-cache lock wait alone.
|
|
748
|
+
*/
|
|
749
|
+
var POOL_MODEL_LOAD_BOUNDS = {
|
|
750
|
+
openvino: {
|
|
751
|
+
softMs: 12e4,
|
|
752
|
+
hardMs: 6e5,
|
|
753
|
+
why: "A cold iGPU compile of the WHOLE model set measured ~25 s on the N100; one model is 1-3 s (YuNet 1.4 s on a fresh worker, 2026-09-26). 120 s is ~5x the worst measured set. No evidence of a legitimate compile past it; the hard bound gives one 5x more before the worker is recycled.",
|
|
754
|
+
gilMayBeHeldDuringCompile: false,
|
|
755
|
+
gilWhy: "The OpenVINO Python binding is believed to release the GIL around compile_model (gil_scoped_release). Not measured on the hub; the passive recipe in D653 checks it."
|
|
756
|
+
},
|
|
757
|
+
coreml: {
|
|
758
|
+
softMs: 3e5,
|
|
759
|
+
hardMs: 9e5,
|
|
760
|
+
why: "A load may first WAIT up to 180 s for another worker holding the compile-cache lock (`COMPILE_CACHE_LOCK_TIMEOUT_SEC` in inference_pool.py), then compile itself: 180 s of lock plus 120 s of compile. The old 120 s bound was shorter than the lock wait alone.",
|
|
761
|
+
gilMayBeHeldDuringCompile: true,
|
|
762
|
+
gilWhy: "Unverified whether coremltools releases the GIL while it compiles an mlpackage. If it does not, the loop freezes and nothing — not the soft timer, not mem_stats — can answer."
|
|
763
|
+
},
|
|
764
|
+
onnxruntime: {
|
|
765
|
+
softMs: 12e4,
|
|
766
|
+
hardMs: 6e5,
|
|
767
|
+
why: "Session creation, no persistent compile cache; CUDA/CoreML EP init is the slow case. Same numbers as OpenVINO for lack of any measurement saying otherwise.",
|
|
768
|
+
gilMayBeHeldDuringCompile: false,
|
|
769
|
+
gilWhy: "Not known to hold the GIL through session creation, and no frozen loop has been observed. Claiming the grace would blind stall detection during every load."
|
|
770
|
+
},
|
|
771
|
+
edgetpu: {
|
|
772
|
+
softMs: 12e4,
|
|
773
|
+
hardMs: 6e5,
|
|
774
|
+
why: "An Edge TPU model is precompiled; a load is an interpreter + delegate bind of seconds. Kept at the shared default rather than tightened without a measurement.",
|
|
775
|
+
gilMayBeHeldDuringCompile: false,
|
|
776
|
+
gilWhy: "Nothing is compiled: the load is an interpreter + delegate bind of seconds."
|
|
777
|
+
}
|
|
778
|
+
};
|
|
779
|
+
/**
|
|
780
|
+
* The host-side deadline of a `load` / `replace`: the soft bound plus a margin,
|
|
781
|
+
* so the worker's NAMED answer always lands first.
|
|
782
|
+
*/
|
|
783
|
+
var POOL_MODEL_LOAD_REPLY_MARGIN_MS = 3e4;
|
|
784
|
+
function chargesDeviceBudget(reason) {
|
|
785
|
+
return reason !== "compile-hung";
|
|
786
|
+
}
|
|
787
|
+
/**
|
|
788
|
+
* Tracks live inference outcomes on one worker and says when its executor has
|
|
789
|
+
* produced nothing for too long. Consulted on demand — no timer.
|
|
790
|
+
*
|
|
791
|
+
* A SHED (`dropped`) does not count as a result: the Python loop answers sheds
|
|
792
|
+
* itself, so a worker whose executor threads are hung keeps shedding happily.
|
|
793
|
+
*/
|
|
794
|
+
var InferStallDetector = class {
|
|
795
|
+
unanswered = 0;
|
|
796
|
+
lastResultAt;
|
|
797
|
+
/**
|
|
798
|
+
* When the current streak of unanswered requests began — the first one after
|
|
799
|
+
* a result. IDLE time is not silence: quiet cameras send nothing, and a
|
|
800
|
+
* worker that was asked nothing has failed nothing. Measuring from the last
|
|
801
|
+
* result let a 60 s lull plus the first 20 sheds of a burst (~8 s with six
|
|
802
|
+
* cameras) poison a healthy worker and charge the device.
|
|
803
|
+
*/
|
|
804
|
+
streakStartedAt = null;
|
|
805
|
+
constructor(now) {
|
|
806
|
+
this.lastResultAt = now;
|
|
807
|
+
}
|
|
808
|
+
/** A reply that carried a real result (or a real error) from the executor. */
|
|
809
|
+
noteResult(now) {
|
|
810
|
+
this.unanswered = 0;
|
|
811
|
+
this.lastResultAt = now;
|
|
812
|
+
this.streakStartedAt = null;
|
|
813
|
+
}
|
|
814
|
+
/**
|
|
815
|
+
* A live request that ended without a result (deadline expiry or a shed).
|
|
816
|
+
* Returns the stall when the worker has crossed both thresholds.
|
|
817
|
+
*/
|
|
818
|
+
noteUnanswered(now) {
|
|
819
|
+
this.unanswered += 1;
|
|
820
|
+
if (this.streakStartedAt === null) this.streakStartedAt = now;
|
|
821
|
+
const silentMs = now - Math.max(this.lastResultAt, this.streakStartedAt);
|
|
822
|
+
if (this.unanswered >= 20 && silentMs >= 3e4) return {
|
|
823
|
+
unanswered: this.unanswered,
|
|
824
|
+
silentMs
|
|
825
|
+
};
|
|
826
|
+
return null;
|
|
827
|
+
}
|
|
828
|
+
};
|
|
829
|
+
//#endregion
|
|
565
830
|
//#region src/detection-pipeline/engine/shared-inference-pool.ts
|
|
566
831
|
/**
|
|
567
832
|
* SharedInferencePool — TypeScript wrapper for inference_pool.py.
|
|
@@ -603,6 +868,27 @@ var RAW_FMT_CODE = {
|
|
|
603
868
|
gray: 2
|
|
604
869
|
};
|
|
605
870
|
/**
|
|
871
|
+
* A load the worker answered with a machine-readable failure class. The
|
|
872
|
+
* provider tells a `compile-timeout` (counted against THAT MODEL on the
|
|
873
|
+
* device) from an ordinary failure by `reason`, never by parsing text.
|
|
874
|
+
*/
|
|
875
|
+
var PoolModelLoadError = class extends Error {
|
|
876
|
+
reason;
|
|
877
|
+
/** The model's file stem, as the worker named it. */
|
|
878
|
+
model;
|
|
879
|
+
constructor(message, reason, model) {
|
|
880
|
+
super(message);
|
|
881
|
+
this.name = "PoolModelLoadError";
|
|
882
|
+
this.reason = reason;
|
|
883
|
+
this.model = model;
|
|
884
|
+
}
|
|
885
|
+
};
|
|
886
|
+
/**
|
|
887
|
+
* Request id the worker uses for UNSOLICITED events — a reply to no request
|
|
888
|
+
* (`compile-finished-late`). Never allocated to a request.
|
|
889
|
+
*/
|
|
890
|
+
var WORKER_EVENT_REQ_ID = 4294967295;
|
|
891
|
+
/**
|
|
606
892
|
* Per-inference-request reply timeout (ms). Turns a wedged request (worker
|
|
607
893
|
* alive but no reply) into a rejection so every caller settles — runtime frame
|
|
608
894
|
* dispatch drops the frame; the benchmark full-tree run rejects the wedged
|
|
@@ -611,6 +897,22 @@ var RAW_FMT_CODE = {
|
|
|
611
897
|
* large frame) never trips; override via env for constrained hardware.
|
|
612
898
|
*/
|
|
613
899
|
var POOL_INFER_TIMEOUT_MS = Math.max(1e3, Number(process.env["CAMSTACK_POOL_INFER_TIMEOUT_MS"]) || 6e4);
|
|
900
|
+
/** Commands that compile a model — the only ones that get the load deadline. */
|
|
901
|
+
var MODEL_LOAD_COMMANDS = new Set(["load", "replace"]);
|
|
902
|
+
/**
|
|
903
|
+
* Deadline of the liveness probe (`mem_stats`) sent after a command times out.
|
|
904
|
+
* The worker answers `mem_stats` on its event loop, never behind a compile
|
|
905
|
+
* (D653), so a healthy worker replies in milliseconds even mid-compile; ten
|
|
906
|
+
* seconds only has to outlast a GC pause or a burst of inference replies.
|
|
907
|
+
*/
|
|
908
|
+
var POOL_LIVENESS_PROBE_TIMEOUT_MS = 1e4;
|
|
909
|
+
/**
|
|
910
|
+
* How long a poisoned worker may keep its in-flight LIVE requests before it is
|
|
911
|
+
* killed anyway. A poisoned worker takes no new work, and every live request
|
|
912
|
+
* carries a {@link POOL_LIVE_INFER_TIMEOUT_MS} deadline, so the drain is over
|
|
913
|
+
* by then; the second of slack keeps the backstop from racing the last reply.
|
|
914
|
+
*/
|
|
915
|
+
var POOL_POISON_DRAIN_SLACK_MS = 1e3;
|
|
614
916
|
/**
|
|
615
917
|
* Max inference requests outstanding to ONE worker before new ones are SHED.
|
|
616
918
|
*
|
|
@@ -798,6 +1100,27 @@ var PoolWorker = class {
|
|
|
798
1100
|
wireVersion = 1;
|
|
799
1101
|
nextRequestId = 1;
|
|
800
1102
|
ready = false;
|
|
1103
|
+
/** Set once this worker is declared unusable while ALIVE (D653). Never cleared:
|
|
1104
|
+
* a poisoned worker is recycled, not revived. */
|
|
1105
|
+
poisonVerdict = null;
|
|
1106
|
+
/** The kill of a poisoned worker has been started. */
|
|
1107
|
+
recycling = false;
|
|
1108
|
+
/** Kills a poisoned worker whose drain did not finish in time. */
|
|
1109
|
+
recycleBackstop = null;
|
|
1110
|
+
/** A liveness probe is in flight — one at a time, never a probe storm. */
|
|
1111
|
+
probeInFlight = false;
|
|
1112
|
+
/** The child has exited (any cause). */
|
|
1113
|
+
exited = false;
|
|
1114
|
+
/** A compile past its soft bound, still running in the worker (D653). */
|
|
1115
|
+
abandonedCompile = null;
|
|
1116
|
+
/** Says when live inference has produced nothing for too long (D653). */
|
|
1117
|
+
inferStall = new InferStallDetector(Date.now());
|
|
1118
|
+
/**
|
|
1119
|
+
* A load's HOST deadline expired with no reply since the last reply of any
|
|
1120
|
+
* kind (D653 round 3): the loop missed its own soft bound, so the GIL grace
|
|
1121
|
+
* is over for this worker until something — anything — comes back.
|
|
1122
|
+
*/
|
|
1123
|
+
loadDeadlineMissed = false;
|
|
801
1124
|
log;
|
|
802
1125
|
opts;
|
|
803
1126
|
constructor(opts) {
|
|
@@ -822,6 +1145,24 @@ var PoolWorker = class {
|
|
|
822
1145
|
isReady() {
|
|
823
1146
|
return this.ready;
|
|
824
1147
|
}
|
|
1148
|
+
/** Why this worker was recycled, typed for the provider's budget, or `null`. */
|
|
1149
|
+
getDeathCause() {
|
|
1150
|
+
const v = this.poisonVerdict;
|
|
1151
|
+
const message = this.getPoisonDescription();
|
|
1152
|
+
if (v === null || message === null) return null;
|
|
1153
|
+
return {
|
|
1154
|
+
reason: v.reason,
|
|
1155
|
+
chargesDeviceBudget: chargesDeviceBudget(v.reason),
|
|
1156
|
+
message
|
|
1157
|
+
};
|
|
1158
|
+
}
|
|
1159
|
+
/** Why this live worker was declared unusable, or `null` (D653). */
|
|
1160
|
+
getPoisonDescription() {
|
|
1161
|
+
const v = this.poisonVerdict;
|
|
1162
|
+
if (v === null) return null;
|
|
1163
|
+
const target = v.command.model !== null ? ` of ${v.command.model}` : "";
|
|
1164
|
+
return `${v.reason} on ${v.command.cmd}${target} (pid ${v.pid ?? "unknown"})`;
|
|
1165
|
+
}
|
|
825
1166
|
async initialize(initialModels) {
|
|
826
1167
|
this.process = spawn(this.opts.pythonPath, [this.opts.scriptPath], { stdio: [
|
|
827
1168
|
"pipe",
|
|
@@ -845,7 +1186,11 @@ var PoolWorker = class {
|
|
|
845
1186
|
const spawnedProcess = this.process;
|
|
846
1187
|
spawnedProcess.on("exit", (code, signal) => {
|
|
847
1188
|
this.ready = false;
|
|
1189
|
+
this.exited = true;
|
|
1190
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1191
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
848
1192
|
if (this.process !== spawnedProcess) return;
|
|
1193
|
+
const poisonedBy = this.getPoisonDescription();
|
|
849
1194
|
this.log.error("Worker process exited", { meta: {
|
|
850
1195
|
worker: this.opts.workerLabel,
|
|
851
1196
|
pid: spawnedProcess.pid ?? null,
|
|
@@ -853,9 +1198,10 @@ var PoolWorker = class {
|
|
|
853
1198
|
device: this.opts.device ?? "default",
|
|
854
1199
|
code,
|
|
855
1200
|
signal,
|
|
856
|
-
inFlight: this.pending.size
|
|
1201
|
+
inFlight: this.pending.size,
|
|
1202
|
+
...poisonedBy !== null ? { poisonedBy } : {}
|
|
857
1203
|
} });
|
|
858
|
-
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})`));
|
|
1204
|
+
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: worker process exited (code=${code ?? "null"}, signal=${signal ?? "none"})` + (poisonedBy !== null ? ` — recycled, poisoned by ${poisonedBy}` : "")));
|
|
859
1205
|
});
|
|
860
1206
|
this.process.stdout.on("data", (chunk) => {
|
|
861
1207
|
const t0 = Date.now();
|
|
@@ -869,6 +1215,7 @@ var PoolWorker = class {
|
|
|
869
1215
|
runtime: this.opts.poolRuntime,
|
|
870
1216
|
concurrency: this.opts.concurrency,
|
|
871
1217
|
protocolVersion: 2,
|
|
1218
|
+
modelLoadTimeoutMs: POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs,
|
|
872
1219
|
models: initialModels.map((m) => serializeModelConfig(m))
|
|
873
1220
|
};
|
|
874
1221
|
if (this.opts.device) config["device"] = this.opts.device;
|
|
@@ -891,6 +1238,7 @@ var PoolWorker = class {
|
|
|
891
1238
|
clearTimeout(timeout);
|
|
892
1239
|
if (result["status"] === "ready") {
|
|
893
1240
|
this.ready = true;
|
|
1241
|
+
this.inferStall.noteResult(Date.now());
|
|
894
1242
|
const loadedCount = result["models"];
|
|
895
1243
|
const startupMs = result["startupMs"];
|
|
896
1244
|
const workers = result["workers"] ?? 1;
|
|
@@ -913,7 +1261,7 @@ var PoolWorker = class {
|
|
|
913
1261
|
async infer(modelByte, jpeg, deviceId) {
|
|
914
1262
|
this.ensureReady();
|
|
915
1263
|
const payload = Buffer.concat([Buffer.from([modelByte]), jpeg]);
|
|
916
|
-
return this.dispatch(MSG_INFER_JPEG, payload, deviceId);
|
|
1264
|
+
return this.dispatch(MSG_INFER_JPEG, payload, { deviceId });
|
|
917
1265
|
}
|
|
918
1266
|
async inferRaw(modelByte, raw, width, height, format, deviceId) {
|
|
919
1267
|
this.ensureReady();
|
|
@@ -966,12 +1314,18 @@ var PoolWorker = class {
|
|
|
966
1314
|
const payload = Buffer.allocUnsafe(5);
|
|
967
1315
|
payload[0] = modelByte;
|
|
968
1316
|
payload.writeUInt32LE(frameId, 1);
|
|
969
|
-
return this.dispatch(MSG_INFER_CACHED, payload, deviceId);
|
|
1317
|
+
return this.dispatch(MSG_INFER_CACHED, payload, { deviceId });
|
|
970
1318
|
}
|
|
971
1319
|
async sendCommand(cmd) {
|
|
972
1320
|
this.ensureReady();
|
|
1321
|
+
const command = describeCommand(cmd);
|
|
973
1322
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
974
|
-
|
|
1323
|
+
const sentAt = Date.now();
|
|
1324
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.warnIfLoadQueued(command);
|
|
1325
|
+
const raw = await this.dispatch(MSG_COMMAND, payload, { command });
|
|
1326
|
+
if (MODEL_LOAD_COMMANDS.has(command.cmd)) this.inferStall.noteResult(Date.now());
|
|
1327
|
+
this.inspectCommandReply(raw, command, sentAt);
|
|
1328
|
+
return raw;
|
|
975
1329
|
}
|
|
976
1330
|
/** Command whose reply shape is NOT the load/unload/replace/status envelope
|
|
977
1331
|
* (e.g. `mem_stats`) — returns the raw JSON record, so no cast is needed
|
|
@@ -979,13 +1333,15 @@ var PoolWorker = class {
|
|
|
979
1333
|
async sendRawCommand(cmd) {
|
|
980
1334
|
this.ensureReady();
|
|
981
1335
|
const payload = Buffer.from(JSON.stringify(cmd), "utf8");
|
|
982
|
-
return this.dispatch(MSG_COMMAND, payload);
|
|
1336
|
+
return this.dispatch(MSG_COMMAND, payload, { command: describeCommand(cmd) });
|
|
983
1337
|
}
|
|
984
1338
|
async dispose() {
|
|
985
1339
|
const proc = this.process;
|
|
986
1340
|
if (!proc) return;
|
|
987
1341
|
this.process = null;
|
|
988
1342
|
this.ready = false;
|
|
1343
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1344
|
+
if (this.abandonedCompile) clearTimeout(this.abandonedCompile.hardTimer);
|
|
989
1345
|
await terminateChild(proc, POOL_WORKER_TERM_GRACE_MS);
|
|
990
1346
|
this.rejectAll(/* @__PURE__ */ new Error(`PoolWorker[${this.opts.workerLabel}]: pool disposed while the request was in flight`));
|
|
991
1347
|
}
|
|
@@ -1028,8 +1384,11 @@ var PoolWorker = class {
|
|
|
1028
1384
|
}
|
|
1029
1385
|
/** Live inference gets seconds; commands and model loads keep the long
|
|
1030
1386
|
* timeout — see {@link POOL_LIVE_INFER_TIMEOUT_MS}. */
|
|
1031
|
-
deadlineFor(msgType) {
|
|
1032
|
-
|
|
1387
|
+
deadlineFor(msgType, opts = {}) {
|
|
1388
|
+
if (opts.timeoutMs !== void 0) return opts.timeoutMs;
|
|
1389
|
+
if (SHEDDABLE_MSG_TYPES.has(msgType)) return POOL_LIVE_INFER_TIMEOUT_MS;
|
|
1390
|
+
if (opts.command !== void 0 && MODEL_LOAD_COMMANDS.has(opts.command.cmd)) return POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].softMs + POOL_MODEL_LOAD_REPLY_MARGIN_MS;
|
|
1391
|
+
return POOL_INFER_TIMEOUT_MS;
|
|
1033
1392
|
}
|
|
1034
1393
|
/**
|
|
1035
1394
|
* The request's ABSOLUTE deadline for the wire (D350), or `null` when this
|
|
@@ -1046,16 +1405,18 @@ var PoolWorker = class {
|
|
|
1046
1405
|
stamp.writeBigUInt64LE(BigInt(Date.now() + this.deadlineFor(msgType)), 0);
|
|
1047
1406
|
return stamp;
|
|
1048
1407
|
}
|
|
1049
|
-
dispatch(msgType, payload,
|
|
1050
|
-
const shed = this.shedIfSaturated(msgType, deviceId);
|
|
1408
|
+
dispatch(msgType, payload, opts = {}) {
|
|
1409
|
+
const shed = this.shedIfSaturated(msgType, opts.deviceId);
|
|
1051
1410
|
if (shed) return Promise.resolve(shed);
|
|
1052
1411
|
const reqId = this.allocRequestId();
|
|
1053
1412
|
return new Promise((resolve, reject) => {
|
|
1054
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1413
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, opts);
|
|
1055
1414
|
this.pending.set(reqId, {
|
|
1056
1415
|
resolve,
|
|
1057
1416
|
reject,
|
|
1058
|
-
timer
|
|
1417
|
+
timer,
|
|
1418
|
+
...opts.command !== void 0 ? { command: opts.command } : {},
|
|
1419
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1059
1420
|
});
|
|
1060
1421
|
try {
|
|
1061
1422
|
const stamp = this.deadlineStamp(msgType);
|
|
@@ -1086,8 +1447,9 @@ var PoolWorker = class {
|
|
|
1086
1447
|
* - Anything else (commands, model loads) still REJECTS with the error the
|
|
1087
1448
|
* dashboards grep for — a lost command is a fault, not flow control.
|
|
1088
1449
|
*/
|
|
1089
|
-
armRequestTimeout(reqId, msgType, resolve, reject,
|
|
1090
|
-
const
|
|
1450
|
+
armRequestTimeout(reqId, msgType, resolve, reject, opts = {}) {
|
|
1451
|
+
const { deviceId, command } = opts;
|
|
1452
|
+
const timeoutMs = this.deadlineFor(msgType, opts);
|
|
1091
1453
|
const timer = setTimeout(() => {
|
|
1092
1454
|
if (this.pending.delete(reqId)) {
|
|
1093
1455
|
if (SHEDDABLE_MSG_TYPES.has(msgType)) {
|
|
@@ -1111,6 +1473,7 @@ var PoolWorker = class {
|
|
|
1111
1473
|
dropped: true,
|
|
1112
1474
|
shedReason: "deadline-expired"
|
|
1113
1475
|
});
|
|
1476
|
+
this.noteInferUnanswered();
|
|
1114
1477
|
return;
|
|
1115
1478
|
}
|
|
1116
1479
|
this.timedOutCount++;
|
|
@@ -1123,10 +1486,20 @@ var PoolWorker = class {
|
|
|
1123
1486
|
device: this.opts.device ?? "default",
|
|
1124
1487
|
inFlight: this.pending.size,
|
|
1125
1488
|
reqId,
|
|
1126
|
-
timeoutMs
|
|
1489
|
+
timeoutMs,
|
|
1490
|
+
...command !== void 0 ? {
|
|
1491
|
+
command: command.cmd,
|
|
1492
|
+
modelIndex: command.modelIndex,
|
|
1493
|
+
model: command.model
|
|
1494
|
+
} : {}
|
|
1127
1495
|
}
|
|
1128
1496
|
});
|
|
1129
|
-
|
|
1497
|
+
const message = `PoolWorker[${this.opts.workerLabel}]: inference request ${reqId} timed out after ${timeoutMs}ms on ${this.opts.poolRuntime}:${this.opts.device ?? "default"} (worker alive, no reply)`;
|
|
1498
|
+
if (command !== void 0 && MODEL_LOAD_COMMANDS.has(command.cmd)) {
|
|
1499
|
+
this.loadDeadlineMissed = true;
|
|
1500
|
+
reject(new PoolModelLoadError(`compile-timeout: ${message}`, "compile-timeout", command.model));
|
|
1501
|
+
} else reject(new Error(message));
|
|
1502
|
+
if (opts.probe !== true && command !== void 0) this.probeLiveness(command);
|
|
1130
1503
|
}
|
|
1131
1504
|
}, timeoutMs);
|
|
1132
1505
|
timer.unref?.();
|
|
@@ -1137,11 +1510,12 @@ var PoolWorker = class {
|
|
|
1137
1510
|
if (shed) return Promise.resolve(shed);
|
|
1138
1511
|
const reqId = this.allocRequestId();
|
|
1139
1512
|
return new Promise((resolve, reject) => {
|
|
1140
|
-
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, deviceId);
|
|
1513
|
+
const timer = this.armRequestTimeout(reqId, msgType, resolve, reject, { deviceId });
|
|
1141
1514
|
this.pending.set(reqId, {
|
|
1142
1515
|
resolve,
|
|
1143
1516
|
reject,
|
|
1144
|
-
timer
|
|
1517
|
+
timer,
|
|
1518
|
+
...SHEDDABLE_MSG_TYPES.has(msgType) ? { live: true } : {}
|
|
1145
1519
|
});
|
|
1146
1520
|
try {
|
|
1147
1521
|
if (!this.process?.stdin) throw new Error("PoolWorker: not initialized");
|
|
@@ -1163,10 +1537,10 @@ var PoolWorker = class {
|
|
|
1163
1537
|
}
|
|
1164
1538
|
allocRequestId() {
|
|
1165
1539
|
let id = this.nextRequestId;
|
|
1166
|
-
this.nextRequestId = id >=
|
|
1540
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1167
1541
|
while (this.pending.has(id)) {
|
|
1168
1542
|
id = this.nextRequestId;
|
|
1169
|
-
this.nextRequestId = id >=
|
|
1543
|
+
this.nextRequestId = id >= WORKER_EVENT_REQ_ID - 1 ? 1 : id + 1;
|
|
1170
1544
|
}
|
|
1171
1545
|
return id;
|
|
1172
1546
|
}
|
|
@@ -1181,6 +1555,8 @@ var PoolWorker = class {
|
|
|
1181
1555
|
this.process.stdin.write(payload);
|
|
1182
1556
|
}
|
|
1183
1557
|
ensureReady() {
|
|
1558
|
+
const poisonedBy = this.getPoisonDescription();
|
|
1559
|
+
if (poisonedBy !== null) throw new Error(`PoolWorker[${this.opts.workerLabel}]: poisoned (${poisonedBy}) — recycling, not taking work`);
|
|
1184
1560
|
if (!this.ready || !this.process?.stdin) throw new Error(`PoolWorker[${this.opts.workerLabel}]: not initialized`);
|
|
1185
1561
|
}
|
|
1186
1562
|
/** Time spent in `Buffer.concat` since the last report. */
|
|
@@ -1220,23 +1596,273 @@ var PoolWorker = class {
|
|
|
1220
1596
|
const reqId = this.receiveBuffer.readUInt32LE(4);
|
|
1221
1597
|
const jsonBytes = this.receiveBuffer.subarray(8, 4 + totalLen);
|
|
1222
1598
|
this.receiveBuffer = this.receiveBuffer.subarray(4 + totalLen);
|
|
1599
|
+
if (reqId === WORKER_EVENT_REQ_ID) {
|
|
1600
|
+
this.handleWorkerEvent(jsonBytes);
|
|
1601
|
+
continue;
|
|
1602
|
+
}
|
|
1223
1603
|
const entry = this.pending.get(reqId);
|
|
1224
1604
|
if (!entry) {
|
|
1225
1605
|
this.log.warn("Response for unknown request id", { meta: {
|
|
1226
1606
|
worker: this.opts.workerLabel,
|
|
1227
1607
|
reqId
|
|
1228
1608
|
} });
|
|
1609
|
+
this.noteLateReply(jsonBytes);
|
|
1229
1610
|
continue;
|
|
1230
1611
|
}
|
|
1231
1612
|
this.pending.delete(reqId);
|
|
1232
1613
|
if (entry.timer) clearTimeout(entry.timer);
|
|
1233
1614
|
try {
|
|
1234
1615
|
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1616
|
+
if (entry.live === true) if (parsed["dropped"] === true) this.noteInferUnanswered();
|
|
1617
|
+
else this.inferStall.noteResult(Date.now());
|
|
1235
1618
|
entry.resolve(parsed);
|
|
1236
1619
|
} catch (err) {
|
|
1237
1620
|
entry.reject(err instanceof Error ? err : new Error(String(err)));
|
|
1238
1621
|
}
|
|
1239
1622
|
}
|
|
1623
|
+
this.noteAnyReply();
|
|
1624
|
+
if (this.poisonVerdict !== null && this.pending.size === 0) this.recycle();
|
|
1625
|
+
}
|
|
1626
|
+
/**
|
|
1627
|
+
* Something came back from the worker: its loop was alive just now, so the
|
|
1628
|
+
* missed-deadline verdict is lifted.
|
|
1629
|
+
*/
|
|
1630
|
+
noteAnyReply() {
|
|
1631
|
+
this.loadDeadlineMissed = false;
|
|
1632
|
+
}
|
|
1633
|
+
/**
|
|
1634
|
+
* The invariant the load deadline depends on (D653 § 2): loads into one
|
|
1635
|
+
* pool are serialised, so at most ONE load is outstanding per worker. The
|
|
1636
|
+
* worker runs model commands in order and starts a load's soft bound only
|
|
1637
|
+
* when it begins, while the host's deadline for it runs from the SEND. A
|
|
1638
|
+
* load queued behind another could therefore see its host deadline fire
|
|
1639
|
+
* before the worker's named answer. Nothing in the provider issues that; if
|
|
1640
|
+
* anything ever does, this says so.
|
|
1641
|
+
*/
|
|
1642
|
+
warnIfLoadQueued(next) {
|
|
1643
|
+
const ahead = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0 && MODEL_LOAD_COMMANDS.has(c.cmd));
|
|
1644
|
+
if (ahead.length === 0) return;
|
|
1645
|
+
this.log.warn("model load queued behind another on one worker — its deadline may fire before the worker answers", { meta: {
|
|
1646
|
+
worker: this.opts.workerLabel,
|
|
1647
|
+
pid: this.getPid(),
|
|
1648
|
+
runtime: this.opts.poolRuntime,
|
|
1649
|
+
device: this.opts.device ?? "default",
|
|
1650
|
+
model: next.model,
|
|
1651
|
+
queuedBehind: ahead.map((c) => c.model)
|
|
1652
|
+
} });
|
|
1653
|
+
}
|
|
1654
|
+
hasPendingLoad() {
|
|
1655
|
+
for (const p of this.pending.values()) if (p.command !== void 0 && MODEL_LOAD_COMMANDS.has(p.command.cmd)) return true;
|
|
1656
|
+
return false;
|
|
1657
|
+
}
|
|
1658
|
+
/**
|
|
1659
|
+
* A load reply that says a compile outlived its SOFT bound (`compile-timeout`)
|
|
1660
|
+
* or that an earlier one still has (`worker-poisoned`). The worker is not
|
|
1661
|
+
* recycled for it (fix round 1): the compile keeps running so a slow but
|
|
1662
|
+
* finite one writes its cache, the loaded models keep serving, and the worker
|
|
1663
|
+
* starts no new load. Only a compile still running at the HARD bound costs
|
|
1664
|
+
* the process.
|
|
1665
|
+
*/
|
|
1666
|
+
inspectCommandReply(raw, command, sentAt) {
|
|
1667
|
+
const reason = raw["reason"];
|
|
1668
|
+
if (reason !== "compile-timeout" && reason !== "worker-poisoned") return;
|
|
1669
|
+
if (this.abandonedCompile !== null || this.poisonVerdict !== null || this.exited) return;
|
|
1670
|
+
const model = raw["modelId"];
|
|
1671
|
+
const abandoned = {
|
|
1672
|
+
...command,
|
|
1673
|
+
model: typeof model === "string" ? model : command.model
|
|
1674
|
+
};
|
|
1675
|
+
const bounds = POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime];
|
|
1676
|
+
const reported = raw["compileElapsedMs"];
|
|
1677
|
+
const elapsedMs = typeof reported === "number" && Number.isFinite(reported) && reported >= 0 ? reported : Date.now() - sentAt;
|
|
1678
|
+
const hardTimer = setTimeout(() => this.poison("compile-hung", abandoned, `the compile of ${abandoned.model ?? "unknown"} was still running at its hard bound (${bounds.hardMs}ms)`), Math.max(0, bounds.hardMs - elapsedMs));
|
|
1679
|
+
hardTimer.unref?.();
|
|
1680
|
+
this.abandonedCompile = {
|
|
1681
|
+
command: abandoned,
|
|
1682
|
+
since: Date.now(),
|
|
1683
|
+
hardTimer
|
|
1684
|
+
};
|
|
1685
|
+
this.log.warn("model compile past its soft bound — left running so its cache can be written", { meta: {
|
|
1686
|
+
worker: this.opts.workerLabel,
|
|
1687
|
+
pid: this.getPid(),
|
|
1688
|
+
runtime: this.opts.poolRuntime,
|
|
1689
|
+
device: this.opts.device ?? "default",
|
|
1690
|
+
model: abandoned.model,
|
|
1691
|
+
modelIndex: abandoned.modelIndex,
|
|
1692
|
+
softMs: bounds.softMs,
|
|
1693
|
+
hardMs: bounds.hardMs,
|
|
1694
|
+
loads: "refused until it returns"
|
|
1695
|
+
} });
|
|
1696
|
+
}
|
|
1697
|
+
/** An unsolicited worker event (`compile-finished-late`). */
|
|
1698
|
+
handleWorkerEvent(jsonBytes) {
|
|
1699
|
+
let event;
|
|
1700
|
+
try {
|
|
1701
|
+
event = JSON.parse(jsonBytes.toString("utf8"));
|
|
1702
|
+
} catch {
|
|
1703
|
+
return;
|
|
1704
|
+
}
|
|
1705
|
+
if (event["event"] !== "compile-finished-late") return;
|
|
1706
|
+
const abandoned = this.abandonedCompile;
|
|
1707
|
+
if (abandoned !== null) clearTimeout(abandoned.hardTimer);
|
|
1708
|
+
this.abandonedCompile = null;
|
|
1709
|
+
this.inferStall.noteResult(Date.now());
|
|
1710
|
+
const ok = event["ok"] === true;
|
|
1711
|
+
const meta = {
|
|
1712
|
+
worker: this.opts.workerLabel,
|
|
1713
|
+
pid: this.getPid(),
|
|
1714
|
+
runtime: this.opts.poolRuntime,
|
|
1715
|
+
device: this.opts.device ?? "default",
|
|
1716
|
+
model: typeof event["modelId"] === "string" ? event["modelId"] : null,
|
|
1717
|
+
elapsedMs: typeof event["elapsedMs"] === "number" ? event["elapsedMs"] : null,
|
|
1718
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1719
|
+
};
|
|
1720
|
+
if (ok) this.log.info("slow model compile finished after its soft bound — cache written, loads re-enabled", { meta });
|
|
1721
|
+
else this.log.warn("slow model compile failed after its soft bound — loads re-enabled", { meta });
|
|
1722
|
+
this.opts.onCompileFinishedLate?.({
|
|
1723
|
+
model: meta.model,
|
|
1724
|
+
ok,
|
|
1725
|
+
...typeof event["error"] === "string" ? { error: event["error"] } : {}
|
|
1726
|
+
});
|
|
1727
|
+
}
|
|
1728
|
+
/**
|
|
1729
|
+
* The GIL grace (D653 round 2): may a silent loop be a compile holding the
|
|
1730
|
+
* GIL rather than a wedge? Only while a load is SENT AND UNANSWERED, and only
|
|
1731
|
+
* on a runtime whose compile may hold the GIL (`POOL_MODEL_LOAD_BOUNDS`).
|
|
1732
|
+
*
|
|
1733
|
+
* NOT while a compile runs past its soft bound: the worker REPLIED
|
|
1734
|
+
* `compile-timeout`, so its loop is proven alive, and an inference hang
|
|
1735
|
+
* behind that compile — the 2026-09-26 incident exactly — must be seen on
|
|
1736
|
+
* the normal rule, not ~600-900 s later at the hard bound.
|
|
1737
|
+
*/
|
|
1738
|
+
gilGraceActive() {
|
|
1739
|
+
if (!POOL_MODEL_LOAD_BOUNDS[this.opts.poolRuntime].gilMayBeHeldDuringCompile) return false;
|
|
1740
|
+
if (this.loadDeadlineMissed) return false;
|
|
1741
|
+
return this.hasPendingLoad();
|
|
1742
|
+
}
|
|
1743
|
+
/**
|
|
1744
|
+
* The reason a silent worker is recycled with: a freeze that began with a
|
|
1745
|
+
* load missing its deadline is that compile's doing (`compile-hung`, charged
|
|
1746
|
+
* to the model, not the device); anything else is the device's.
|
|
1747
|
+
*/
|
|
1748
|
+
silenceReason(fallback) {
|
|
1749
|
+
return this.loadDeadlineMissed ? "compile-hung" : fallback;
|
|
1750
|
+
}
|
|
1751
|
+
/** A live request ended with no result: judge the executor (D653, item 4). */
|
|
1752
|
+
noteInferUnanswered() {
|
|
1753
|
+
const stall = this.inferStall.noteUnanswered(Date.now());
|
|
1754
|
+
if (stall === null || this.gilGraceActive()) return;
|
|
1755
|
+
this.poison(this.silenceReason("infer-unresponsive"), {
|
|
1756
|
+
cmd: "infer",
|
|
1757
|
+
modelIndex: null,
|
|
1758
|
+
model: null
|
|
1759
|
+
}, `${stall.unanswered} live requests in a row ended without a result, none for ${stall.silentMs}ms`);
|
|
1760
|
+
}
|
|
1761
|
+
/** A reply whose request was already abandoned: a real result still proves the executor runs. */
|
|
1762
|
+
noteLateReply(jsonBytes) {
|
|
1763
|
+
try {
|
|
1764
|
+
const parsed = JSON.parse(jsonBytes.toString("utf8"));
|
|
1765
|
+
if (parsed["dropped"] !== true && parsed["cmd"] === void 0) this.inferStall.noteResult(Date.now());
|
|
1766
|
+
} catch {}
|
|
1767
|
+
}
|
|
1768
|
+
/**
|
|
1769
|
+
* Ask a worker whose command just timed out whether its loop still answers.
|
|
1770
|
+
* `mem_stats` is served on the loop, never behind a compile, so a worker
|
|
1771
|
+
* that cannot answer it inside {@link POOL_LIVENESS_PROBE_TIMEOUT_MS} is not
|
|
1772
|
+
* slow — it is wedged. That is the 2026-09-26 shape exactly: 123 of 123
|
|
1773
|
+
* `mem_stats` lost while the process looked alive.
|
|
1774
|
+
*/
|
|
1775
|
+
probeLiveness(timedOut) {
|
|
1776
|
+
if (this.probeInFlight || this.poisonVerdict !== null || this.exited) return;
|
|
1777
|
+
this.probeInFlight = true;
|
|
1778
|
+
this.dispatch(MSG_COMMAND, Buffer.from(JSON.stringify({ cmd: "mem_stats" }), "utf8"), {
|
|
1779
|
+
command: {
|
|
1780
|
+
cmd: "mem_stats",
|
|
1781
|
+
modelIndex: null,
|
|
1782
|
+
model: null
|
|
1783
|
+
},
|
|
1784
|
+
timeoutMs: POOL_LIVENESS_PROBE_TIMEOUT_MS,
|
|
1785
|
+
probe: true
|
|
1786
|
+
}).then(() => {
|
|
1787
|
+
this.log.warn("pool command timed out but the worker answers its liveness probe — slow, not wedged", { meta: {
|
|
1788
|
+
worker: this.opts.workerLabel,
|
|
1789
|
+
pid: this.getPid(),
|
|
1790
|
+
runtime: this.opts.poolRuntime,
|
|
1791
|
+
device: this.opts.device ?? "default",
|
|
1792
|
+
command: timedOut.cmd,
|
|
1793
|
+
modelIndex: timedOut.modelIndex,
|
|
1794
|
+
model: timedOut.model
|
|
1795
|
+
} });
|
|
1796
|
+
}, (err) => {
|
|
1797
|
+
if (this.gilGraceActive()) {
|
|
1798
|
+
this.log.warn("liveness probe unanswered while a model load is outstanding — not poisoning yet", { meta: {
|
|
1799
|
+
worker: this.opts.workerLabel,
|
|
1800
|
+
pid: this.getPid(),
|
|
1801
|
+
runtime: this.opts.poolRuntime,
|
|
1802
|
+
device: this.opts.device ?? "default",
|
|
1803
|
+
command: timedOut.cmd,
|
|
1804
|
+
model: timedOut.model
|
|
1805
|
+
} });
|
|
1806
|
+
return;
|
|
1807
|
+
}
|
|
1808
|
+
this.poison(this.silenceReason("unresponsive"), timedOut, `${timedOut.cmd} timed out, then the liveness probe failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1809
|
+
}).finally(() => {
|
|
1810
|
+
this.probeInFlight = false;
|
|
1811
|
+
});
|
|
1812
|
+
}
|
|
1813
|
+
/**
|
|
1814
|
+
* Declare this LIVE worker unusable, say so once at ERROR, stop taking work,
|
|
1815
|
+
* and recycle it once it has drained.
|
|
1816
|
+
*
|
|
1817
|
+
* `ready = false` is what hands the pool back to the provider: its next
|
|
1818
|
+
* dispatch finds the factory not ready and condemns it under the per-device
|
|
1819
|
+
* restart budget (3 deaths in 10 min, then a terminal `failed` the balancer
|
|
1820
|
+
* excludes) — the same path a crashed worker takes, except that a
|
|
1821
|
+
* `compile-hung` death is not charged to the device (see PoolPoisonReason).
|
|
1822
|
+
* An unresponsive worker is killed at once (it will answer nothing it
|
|
1823
|
+
* holds); a compile-hung one keeps serving its loaded models until its
|
|
1824
|
+
* in-flight requests are answered, bounded by the live deadline.
|
|
1825
|
+
*/
|
|
1826
|
+
poison(reason, command, detail) {
|
|
1827
|
+
if (this.poisonVerdict !== null || this.exited || this.process === null) return;
|
|
1828
|
+
this.poisonVerdict = {
|
|
1829
|
+
reason,
|
|
1830
|
+
command,
|
|
1831
|
+
detail,
|
|
1832
|
+
pid: this.getPid()
|
|
1833
|
+
};
|
|
1834
|
+
this.ready = false;
|
|
1835
|
+
const inFlightCommands = [...this.pending.values()].map((p) => p.command).filter((c) => c !== void 0).map((c) => c.model !== null ? `${c.cmd}:${c.model}` : c.cmd);
|
|
1836
|
+
this.log.error("pool worker POISONED — recycling it", { meta: {
|
|
1837
|
+
worker: this.opts.workerLabel,
|
|
1838
|
+
pid: this.getPid(),
|
|
1839
|
+
runtime: this.opts.poolRuntime,
|
|
1840
|
+
device: this.opts.device ?? "default",
|
|
1841
|
+
reason,
|
|
1842
|
+
command: command.cmd,
|
|
1843
|
+
modelIndex: command.modelIndex,
|
|
1844
|
+
model: command.model,
|
|
1845
|
+
detail,
|
|
1846
|
+
inFlight: this.pending.size,
|
|
1847
|
+
inFlightCommands
|
|
1848
|
+
} });
|
|
1849
|
+
if (reason !== "compile-hung" || this.pending.size === 0) {
|
|
1850
|
+
this.recycle();
|
|
1851
|
+
return;
|
|
1852
|
+
}
|
|
1853
|
+
this.recycleBackstop = setTimeout(() => this.recycle(), POOL_LIVE_INFER_TIMEOUT_MS + POOL_POISON_DRAIN_SLACK_MS);
|
|
1854
|
+
this.recycleBackstop.unref?.();
|
|
1855
|
+
}
|
|
1856
|
+
/**
|
|
1857
|
+
* Kill a poisoned worker WITHOUT nulling `this.process`, so its `exit` is
|
|
1858
|
+
* reported and rejects whatever it still held — a deliberate dispose would
|
|
1859
|
+
* silence both, and this is not one.
|
|
1860
|
+
*/
|
|
1861
|
+
recycle() {
|
|
1862
|
+
if (this.recycling || this.exited || this.process === null) return;
|
|
1863
|
+
this.recycling = true;
|
|
1864
|
+
if (this.recycleBackstop) clearTimeout(this.recycleBackstop);
|
|
1865
|
+
terminateChild(this.process, POOL_WORKER_TERM_GRACE_MS);
|
|
1240
1866
|
}
|
|
1241
1867
|
rejectAll(err) {
|
|
1242
1868
|
const entries = [...this.pending.values()];
|
|
@@ -1283,8 +1909,10 @@ var SharedInferencePool = class {
|
|
|
1283
1909
|
this.tuning = options.tuning ?? null;
|
|
1284
1910
|
this.numWorkers = Math.max(1, options.numWorkers ?? 1);
|
|
1285
1911
|
this.device = options.device;
|
|
1912
|
+
this.onCompileFinishedLate = options.onCompileFinishedLate;
|
|
1286
1913
|
}
|
|
1287
1914
|
device;
|
|
1915
|
+
onCompileFinishedLate;
|
|
1288
1916
|
/** Pid of the first worker (for legacy callers). Use `getPids()` for all. */
|
|
1289
1917
|
/** Summed backlog and shed count across the pool's workers. A rising
|
|
1290
1918
|
* `inFlight` with a rising `shed` is a worker falling behind; a rising
|
|
@@ -1323,7 +1951,8 @@ var SharedInferencePool = class {
|
|
|
1323
1951
|
tuning: this.tuning,
|
|
1324
1952
|
logger: this.log,
|
|
1325
1953
|
workerLabel: `w${i}`,
|
|
1326
|
-
...this.device ? { device: this.device } : {}
|
|
1954
|
+
...this.device ? { device: this.device } : {},
|
|
1955
|
+
...this.onCompileFinishedLate ? { onCompileFinishedLate: this.onCompileFinishedLate } : {}
|
|
1327
1956
|
}));
|
|
1328
1957
|
const t0 = performance.now();
|
|
1329
1958
|
const results = await Promise.all(this.workers.map((w) => w.initialize(initialModels)));
|
|
@@ -1418,7 +2047,7 @@ var SharedInferencePool = class {
|
|
|
1418
2047
|
index,
|
|
1419
2048
|
config: serializeModelConfig(config)
|
|
1420
2049
|
})));
|
|
1421
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
2050
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to load model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1422
2051
|
if (index >= this.nextFreeIndex) this.nextFreeIndex = index + 1;
|
|
1423
2052
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1424
2053
|
}
|
|
@@ -1454,7 +2083,7 @@ var SharedInferencePool = class {
|
|
|
1454
2083
|
index,
|
|
1455
2084
|
config: serializeModelConfig(config)
|
|
1456
2085
|
})));
|
|
1457
|
-
for (const resp of responses) if (resp.status !== "ok") throw new
|
|
2086
|
+
for (const resp of responses) if (resp.status !== "ok") throw new PoolModelLoadError(`Failed to replace model at index ${index}: ${describeCommandFailure(resp)}`, resp.reason ?? null, resp.modelId ?? null);
|
|
1458
2087
|
return { loadMs: Math.max(...responses.map((r) => r.loadMs ?? 0)) };
|
|
1459
2088
|
}
|
|
1460
2089
|
/**
|
|
@@ -1504,9 +2133,35 @@ var SharedInferencePool = class {
|
|
|
1504
2133
|
allocateIndex() {
|
|
1505
2134
|
return this.nextFreeIndex++;
|
|
1506
2135
|
}
|
|
2136
|
+
/**
|
|
2137
|
+
* Give back an index whose load FAILED, so the retry reuses it. Only the most
|
|
2138
|
+
* recent allocation can be returned. Loads into one pool are serialised by
|
|
2139
|
+
* the provider, so a failed load is normally the latest one; anything else
|
|
2140
|
+
* is left allocated rather than risk handing out a live slot twice.
|
|
2141
|
+
*/
|
|
2142
|
+
releaseIndex(index) {
|
|
2143
|
+
if (index === this.nextFreeIndex - 1) this.nextFreeIndex = index;
|
|
2144
|
+
}
|
|
1507
2145
|
isReady() {
|
|
1508
2146
|
return this.workers.length > 0 && this.workers.every((w) => w.isReady());
|
|
1509
2147
|
}
|
|
2148
|
+
/**
|
|
2149
|
+
* Why a worker of this pool was declared unusable while alive (D653), or
|
|
2150
|
+
* `null`. The provider charges the restart budget with THIS — or, for a
|
|
2151
|
+
* `compile-hung` death, does not charge the device at all — so the
|
|
2152
|
+
* `inference device FAILED` line names the cause instead of "pool worker is
|
|
2153
|
+
* not ready".
|
|
2154
|
+
*/
|
|
2155
|
+
getDeathCause() {
|
|
2156
|
+
for (const w of this.workers) {
|
|
2157
|
+
const cause = w.getDeathCause();
|
|
2158
|
+
if (cause !== null) return {
|
|
2159
|
+
...cause,
|
|
2160
|
+
message: `worker ${cause.message}`
|
|
2161
|
+
};
|
|
2162
|
+
}
|
|
2163
|
+
return null;
|
|
2164
|
+
}
|
|
1510
2165
|
async dispose() {
|
|
1511
2166
|
await Promise.all(this.workers.map((w) => w.dispose()));
|
|
1512
2167
|
this.workers.length = 0;
|
|
@@ -1571,6 +2226,23 @@ var SharedInferencePool = class {
|
|
|
1571
2226
|
return found;
|
|
1572
2227
|
}
|
|
1573
2228
|
};
|
|
2229
|
+
/** A failed command's reply as one message: its reason class first, when it has one. */
|
|
2230
|
+
function describeCommandFailure(resp) {
|
|
2231
|
+
const error = resp.error ?? "unknown";
|
|
2232
|
+
return resp.reason !== void 0 ? `${resp.reason}: ${error}` : error;
|
|
2233
|
+
}
|
|
2234
|
+
/** Name a command for the lines that must say which one hung (D653). */
|
|
2235
|
+
function describeCommand(cmd) {
|
|
2236
|
+
const name = typeof cmd["cmd"] === "string" ? cmd["cmd"] : "unknown";
|
|
2237
|
+
const index = cmd["index"];
|
|
2238
|
+
const config = cmd["config"];
|
|
2239
|
+
const modelPath = typeof config === "object" && config !== null && "path" in config ? config.path : void 0;
|
|
2240
|
+
return {
|
|
2241
|
+
cmd: name,
|
|
2242
|
+
modelIndex: typeof index === "number" ? index : null,
|
|
2243
|
+
model: typeof modelPath === "string" && modelPath.length > 0 ? path$1.basename(modelPath, path$1.extname(modelPath)) : null
|
|
2244
|
+
};
|
|
2245
|
+
}
|
|
1574
2246
|
function serializeModelConfig(config) {
|
|
1575
2247
|
const result = {
|
|
1576
2248
|
path: config.path,
|
|
@@ -1642,7 +2314,8 @@ function collectFramePlaneSteps(steps, isDetailStep) {
|
|
|
1642
2314
|
}
|
|
1643
2315
|
/**
|
|
1644
2316
|
* Immutably remove from a frame-plane dispatch tree every enabled DETAIL
|
|
1645
|
-
* subtree (at any depth below the top level)
|
|
2317
|
+
* subtree (at any depth below the top level) that `isUnrunnable` — judged
|
|
2318
|
+
* WHOLE at its top, because its children ride its call (D657).
|
|
1646
2319
|
* Those steps can never load into the dispatch device's pool — leaving them
|
|
1647
2320
|
* in the tree makes `ensureModelsForSteps` fail the whole call on a format
|
|
1648
2321
|
* build that does not exist. They are NOT dropped work: each runs later as
|
|
@@ -1654,6 +2327,11 @@ function collectFramePlaneSteps(steps, isDetailStep) {
|
|
|
1654
2327
|
*/
|
|
1655
2328
|
function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
1656
2329
|
const prunedAddonIds = [];
|
|
2330
|
+
const removedAddonIds = [];
|
|
2331
|
+
const collect = (step) => {
|
|
2332
|
+
removedAddonIds.push(step.addonId);
|
|
2333
|
+
for (const child of step.children ?? []) collect(child);
|
|
2334
|
+
};
|
|
1657
2335
|
const walk = (nodes, topLevel) => {
|
|
1658
2336
|
const kept = [];
|
|
1659
2337
|
for (const step of nodes) {
|
|
@@ -1661,8 +2339,9 @@ function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
|
1661
2339
|
kept.push(step);
|
|
1662
2340
|
continue;
|
|
1663
2341
|
}
|
|
1664
|
-
if (!topLevel && isDetailStep(step.addonId) && isUnrunnable(step
|
|
2342
|
+
if (!topLevel && isDetailStep(step.addonId) && isUnrunnable(step)) {
|
|
1665
2343
|
prunedAddonIds.push(step.addonId);
|
|
2344
|
+
collect(step);
|
|
1666
2345
|
continue;
|
|
1667
2346
|
}
|
|
1668
2347
|
kept.push(step.children?.length ? {
|
|
@@ -1674,7 +2353,8 @@ function pruneUnrunnableDetailSteps(steps, isDetailStep, isUnrunnable) {
|
|
|
1674
2353
|
};
|
|
1675
2354
|
return {
|
|
1676
2355
|
steps: walk(steps, true),
|
|
1677
|
-
prunedAddonIds
|
|
2356
|
+
prunedAddonIds,
|
|
2357
|
+
removedAddonIds
|
|
1678
2358
|
};
|
|
1679
2359
|
}
|
|
1680
2360
|
//#endregion
|
|
@@ -1858,6 +2538,28 @@ function poolDecodeSettingsEqual(a, b) {
|
|
|
1858
2538
|
}
|
|
1859
2539
|
//#endregion
|
|
1860
2540
|
//#region src/detection-pipeline/engine/pipeline-model-manager.ts
|
|
2541
|
+
/**
|
|
2542
|
+
* A (step, model) load the pool rejected — carrying the pool's failure class
|
|
2543
|
+
* (`compile-timeout`, …) so the provider can count a timeout against THAT
|
|
2544
|
+
* model without parsing a message (D653).
|
|
2545
|
+
*/
|
|
2546
|
+
var StepVariantLoadError = class extends Error {
|
|
2547
|
+
stepId;
|
|
2548
|
+
modelId;
|
|
2549
|
+
poolIndex;
|
|
2550
|
+
reason;
|
|
2551
|
+
/** The model's file stem as the pool names it (`camstack-yunet-2023mar`). */
|
|
2552
|
+
poolModel;
|
|
2553
|
+
constructor(stepId, modelId, poolIndex, cause) {
|
|
2554
|
+
super(cause instanceof Error ? cause.message : String(cause));
|
|
2555
|
+
this.name = "StepVariantLoadError";
|
|
2556
|
+
this.stepId = stepId;
|
|
2557
|
+
this.modelId = modelId;
|
|
2558
|
+
this.poolIndex = poolIndex;
|
|
2559
|
+
this.reason = cause instanceof PoolModelLoadError ? cause.reason : null;
|
|
2560
|
+
this.poolModel = cause instanceof PoolModelLoadError ? cause.model : null;
|
|
2561
|
+
}
|
|
2562
|
+
};
|
|
1861
2563
|
var PipelineModelManager = class {
|
|
1862
2564
|
pool;
|
|
1863
2565
|
source;
|
|
@@ -1869,11 +2571,20 @@ var PipelineModelManager = class {
|
|
|
1869
2571
|
lruClock = 0;
|
|
1870
2572
|
log;
|
|
1871
2573
|
maxModelsPerStep;
|
|
2574
|
+
deviceKey;
|
|
2575
|
+
/**
|
|
2576
|
+
* Loads issued and not yet answered, keyed `stepId::modelId`. A second
|
|
2577
|
+
* caller for the same pair JOINS the pending load instead of issuing its own
|
|
2578
|
+
* (D653): on 2026-09-26 43 identical YuNet loads queued behind one hung GPU
|
|
2579
|
+
* compile, each allocating a pool index of its own.
|
|
2580
|
+
*/
|
|
2581
|
+
pendingLoads = /* @__PURE__ */ new Map();
|
|
1872
2582
|
constructor(pool, source, logger, options) {
|
|
1873
2583
|
this.pool = pool;
|
|
1874
2584
|
this.source = source;
|
|
1875
2585
|
this.log = logger;
|
|
1876
2586
|
this.maxModelsPerStep = options?.maxModelsPerStep ?? 4;
|
|
2587
|
+
this.deviceKey = options?.deviceKey ?? null;
|
|
1877
2588
|
}
|
|
1878
2589
|
/**
|
|
1879
2590
|
* Apply a new pipeline configuration — driven by the runtime config
|
|
@@ -1916,16 +2627,26 @@ var PipelineModelManager = class {
|
|
|
1916
2627
|
for (const { entry, step } of diff.unchanged) await this.reconcileDecode(entry, step.settings);
|
|
1917
2628
|
}
|
|
1918
2629
|
/**
|
|
1919
|
-
*
|
|
1920
|
-
*
|
|
1921
|
-
*
|
|
1922
|
-
*
|
|
1923
|
-
*
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
2630
|
+
* The warm variant that runs `(stepId, modelId)` — its handle AND the model
|
|
2631
|
+
* loaded at that pool index, which is the only honest answer to "which model
|
|
2632
|
+
* produced this output" (D658).
|
|
2633
|
+
*
|
|
2634
|
+
* No active-variant fallback, deliberately. The step-only lookup this
|
|
2635
|
+
* replaces (`getHandle(stepId)`, deleted) answered the step's ACTIVE
|
|
2636
|
+
* variant — the first one loaded after boot — and a tree that paired that
|
|
2637
|
+
* handle with the step's REQUESTED id stamped SigLIP2 on MobileCLIP vectors
|
|
2638
|
+
* for twenty minutes on 2026-09-27. A pair that is not resident throws,
|
|
2639
|
+
* naming both.
|
|
2640
|
+
*/
|
|
2641
|
+
getVariant(stepId, modelId) {
|
|
2642
|
+
const entry = this.loaded.get(stepId)?.get(modelId);
|
|
2643
|
+
if (entry === void 0) throw new Error(`Step "${stepId}" has no warm variant for model "${modelId}" in the inference pool — refusing rather than running another variant`);
|
|
1927
2644
|
this.touch(entry);
|
|
1928
|
-
return
|
|
2645
|
+
return {
|
|
2646
|
+
engine: this.pool.getHandle(entry.poolIndex),
|
|
2647
|
+
modelId: entry.modelId,
|
|
2648
|
+
poolIndex: entry.poolIndex
|
|
2649
|
+
};
|
|
1929
2650
|
}
|
|
1930
2651
|
/** True iff the step has any model loaded. */
|
|
1931
2652
|
isLoaded(stepId) {
|
|
@@ -1956,13 +2677,14 @@ var PipelineModelManager = class {
|
|
|
1956
2677
|
return this.activeByStep.get(stepId);
|
|
1957
2678
|
}
|
|
1958
2679
|
/**
|
|
1959
|
-
* Pool index for
|
|
1960
|
-
*
|
|
1961
|
-
*
|
|
2680
|
+
* Pool index for exactly `(stepId, modelId)`, or `null` when that pair is not
|
|
2681
|
+
* warm. Used by the inference fast paths that bypass `getVariant` and call
|
|
2682
|
+
* `pool.inferBatch` directly. The model is REQUIRED: the active-variant
|
|
2683
|
+
* fallback this had timed a benchmark on whichever variant was active (D658).
|
|
1962
2684
|
*/
|
|
1963
2685
|
getPoolIndex(stepId, modelId) {
|
|
1964
|
-
const entry = this.
|
|
1965
|
-
if (
|
|
2686
|
+
const entry = this.loaded.get(stepId)?.get(modelId);
|
|
2687
|
+
if (entry === void 0) return null;
|
|
1966
2688
|
this.touch(entry);
|
|
1967
2689
|
return entry.poolIndex;
|
|
1968
2690
|
}
|
|
@@ -2023,6 +2745,23 @@ var PipelineModelManager = class {
|
|
|
2023
2745
|
await this.reconcileDecode(existing, settings);
|
|
2024
2746
|
return existing;
|
|
2025
2747
|
}
|
|
2748
|
+
const key = `${stepId}::${modelId}`;
|
|
2749
|
+
const pending = this.pendingLoads.get(key);
|
|
2750
|
+
if (pending) {
|
|
2751
|
+
const joined = await pending;
|
|
2752
|
+
await this.reconcileDecode(joined, settings);
|
|
2753
|
+
return joined;
|
|
2754
|
+
}
|
|
2755
|
+
const load = this.issueLoad(perStep, stepId, modelId, settings);
|
|
2756
|
+
this.pendingLoads.set(key, load);
|
|
2757
|
+
try {
|
|
2758
|
+
return await load;
|
|
2759
|
+
} finally {
|
|
2760
|
+
this.pendingLoads.delete(key);
|
|
2761
|
+
}
|
|
2762
|
+
}
|
|
2763
|
+
/** Evict if at capacity, then send ONE load command and record the result. */
|
|
2764
|
+
async issueLoad(perStep, stepId, modelId, settings) {
|
|
2026
2765
|
while (perStep.size >= this.maxModelsPerStep) {
|
|
2027
2766
|
const evicted = this.pickEvictionTarget(stepId);
|
|
2028
2767
|
if (!evicted) break;
|
|
@@ -2034,16 +2773,31 @@ var PipelineModelManager = class {
|
|
|
2034
2773
|
cap: this.maxModelsPerStep
|
|
2035
2774
|
} });
|
|
2036
2775
|
}
|
|
2037
|
-
const index = this.pool.allocateIndex();
|
|
2038
2776
|
const config = this.source.buildConfig(stepId, modelId, settings);
|
|
2039
2777
|
const decode = poolDecodeSettingsOf(config);
|
|
2778
|
+
const index = this.pool.allocateIndex();
|
|
2040
2779
|
this.log.info("Loading step variant", { meta: {
|
|
2041
2780
|
step: stepId,
|
|
2042
2781
|
modelId,
|
|
2043
2782
|
poolIndex: index,
|
|
2783
|
+
deviceKey: this.deviceKey,
|
|
2044
2784
|
...decode
|
|
2045
2785
|
} });
|
|
2046
|
-
|
|
2786
|
+
let loadMs;
|
|
2787
|
+
try {
|
|
2788
|
+
({loadMs} = await this.pool.loadModel(index, config));
|
|
2789
|
+
} catch (err) {
|
|
2790
|
+
this.pool.releaseIndex(index);
|
|
2791
|
+
this.log.error("Step variant load failed", { meta: {
|
|
2792
|
+
step: stepId,
|
|
2793
|
+
modelId,
|
|
2794
|
+
poolIndex: index,
|
|
2795
|
+
deviceKey: this.deviceKey,
|
|
2796
|
+
reason: err instanceof PoolModelLoadError ? err.reason : null,
|
|
2797
|
+
error: err instanceof Error ? err.message : String(err)
|
|
2798
|
+
} });
|
|
2799
|
+
throw new StepVariantLoadError(stepId, modelId, index, err);
|
|
2800
|
+
}
|
|
2047
2801
|
this.log.info("Step variant loaded", { meta: {
|
|
2048
2802
|
step: stepId,
|
|
2049
2803
|
modelId,
|
|
@@ -2150,18 +2904,6 @@ var PipelineModelManager = class {
|
|
|
2150
2904
|
await this.unloadEntry(evicted);
|
|
2151
2905
|
}
|
|
2152
2906
|
}
|
|
2153
|
-
resolve(stepId, modelId) {
|
|
2154
|
-
const perStep = this.loaded.get(stepId);
|
|
2155
|
-
if (!perStep) return null;
|
|
2156
|
-
const targetModelId = modelId ?? this.activeByStep.get(stepId);
|
|
2157
|
-
if (!targetModelId) return null;
|
|
2158
|
-
return perStep.get(targetModelId) ?? null;
|
|
2159
|
-
}
|
|
2160
|
-
resolveOrThrow(stepId, modelId) {
|
|
2161
|
-
const entry = this.resolve(stepId, modelId);
|
|
2162
|
-
if (!entry) throw new Error(`Step "${stepId}"${modelId ? ` (model "${modelId}")` : ""} is not loaded in the inference pool`);
|
|
2163
|
-
return entry;
|
|
2164
|
-
}
|
|
2165
2907
|
touch(entry) {
|
|
2166
2908
|
entry.lruTick = ++this.lruClock;
|
|
2167
2909
|
}
|
|
@@ -2357,16 +3099,15 @@ var EngineFactory = class {
|
|
|
2357
3099
|
await this.poolManager.applyConfig(steps);
|
|
2358
3100
|
}
|
|
2359
3101
|
/**
|
|
2360
|
-
*
|
|
2361
|
-
*
|
|
2362
|
-
*
|
|
2363
|
-
* paths picking a non-runtime model).
|
|
3102
|
+
* The resident variant for exactly `(stepId, modelId)`, with the model it
|
|
3103
|
+
* runs — never the step's active variant in its place (D658). The executable
|
|
3104
|
+
* tree is built from this, and a result's model stamp comes from it.
|
|
2364
3105
|
*
|
|
2365
|
-
* @throws if
|
|
3106
|
+
* @throws if that exact pair is not warm.
|
|
2366
3107
|
*/
|
|
2367
|
-
|
|
3108
|
+
getStepVariant(stepId, modelId) {
|
|
2368
3109
|
if (!this.poolManager) throw new Error("EngineFactory not initialized");
|
|
2369
|
-
return this.poolManager.
|
|
3110
|
+
return this.poolManager.getVariant(stepId, modelId);
|
|
2370
3111
|
}
|
|
2371
3112
|
/** Check if a step is loaded (any variant). */
|
|
2372
3113
|
isLoaded(stepId) {
|
|
@@ -2413,6 +3154,15 @@ var EngineFactory = class {
|
|
|
2413
3154
|
isReady() {
|
|
2414
3155
|
return this.pool?.isReady() ?? false;
|
|
2415
3156
|
}
|
|
3157
|
+
/**
|
|
3158
|
+
* Why this factory's pool stopped being usable while its process was alive —
|
|
3159
|
+
* a compile that never returned, or a worker that answered nothing (D653) —
|
|
3160
|
+
* or `null`. A pool that merely crashed says nothing here; its `Worker
|
|
3161
|
+
* process exited` line already names the signal.
|
|
3162
|
+
*/
|
|
3163
|
+
getDeathCause() {
|
|
3164
|
+
return this.pool?.getDeathCause() ?? null;
|
|
3165
|
+
}
|
|
2416
3166
|
/** Native pid of the underlying Python pool, if any. */
|
|
2417
3167
|
getPoolPid() {
|
|
2418
3168
|
return this.pool?.getPid() ?? null;
|
|
@@ -2461,7 +3211,7 @@ var EngineFactory = class {
|
|
|
2461
3211
|
async batchInferRaw(stepId, items, modelId, frameId = 0) {
|
|
2462
3212
|
if (!this.poolManager) throw new Error("EngineFactory.batchInferRaw: pool not initialized");
|
|
2463
3213
|
const idx = this.poolManager.getPoolIndex(stepId, modelId);
|
|
2464
|
-
if (idx === null) throw new Error(`EngineFactory.batchInferRaw: step "${stepId}"
|
|
3214
|
+
if (idx === null) throw new Error(`EngineFactory.batchInferRaw: step "${stepId}" (model "${modelId}") is not loaded`);
|
|
2465
3215
|
return this.poolManager.getPool().inferBatch(idx, items, frameId);
|
|
2466
3216
|
}
|
|
2467
3217
|
async cacheFrameInPool(raw, width, height, format) {
|
|
@@ -2471,7 +3221,7 @@ var EngineFactory = class {
|
|
|
2471
3221
|
async inferCached(stepId, frameId, modelId) {
|
|
2472
3222
|
if (!this.poolManager) throw new Error("inferCached: pool not available");
|
|
2473
3223
|
const idx = this.poolManager.getPoolIndex(stepId, modelId);
|
|
2474
|
-
if (idx === null) throw new Error(`inferCached: step "${stepId}"
|
|
3224
|
+
if (idx === null) throw new Error(`inferCached: step "${stepId}" (model "${modelId}") not loaded`);
|
|
2475
3225
|
return this.poolManager.getPool().inferCached(idx, frameId);
|
|
2476
3226
|
}
|
|
2477
3227
|
async uncacheFrame(frameId) {
|
|
@@ -2539,12 +3289,13 @@ var EngineFactory = class {
|
|
|
2539
3289
|
concurrency,
|
|
2540
3290
|
tuning: resolvedTuning,
|
|
2541
3291
|
numWorkers,
|
|
2542
|
-
...this.opts.engine.device ? { device: this.opts.engine.device } : {}
|
|
3292
|
+
...this.opts.engine.device ? { device: this.opts.engine.device } : {},
|
|
3293
|
+
...this.opts.onCompileFinishedLate ? { onCompileFinishedLate: this.opts.onCompileFinishedLate } : {}
|
|
2543
3294
|
});
|
|
2544
3295
|
this.poolManager = new PipelineModelManager(this.pool, {
|
|
2545
3296
|
buildConfig: (stepId, modelId, settings) => this.buildPoolModelConfig(stepId, modelId, poolRuntime, settings),
|
|
2546
3297
|
resolveDecode: (stepId, modelId, settings) => this.resolveDecode(stepId, modelId, settings)
|
|
2547
|
-
}, this.log.child("model-mgr"));
|
|
3298
|
+
}, this.log.child("model-mgr"), { deviceKey: this.deviceKey });
|
|
2548
3299
|
await this.pool.initialize([]);
|
|
2549
3300
|
await this.poolManager.applyConfig(steps);
|
|
2550
3301
|
}
|
|
@@ -2660,6 +3411,175 @@ function buildPoolModelConfigForStep(inputs) {
|
|
|
2660
3411
|
};
|
|
2661
3412
|
}
|
|
2662
3413
|
//#endregion
|
|
3414
|
+
//#region src/detection-pipeline/engine/model-load-governor.ts
|
|
3415
|
+
/**
|
|
3416
|
+
* What the dispatch path may ask of a pool's model loads, and what it must say
|
|
3417
|
+
* when it may not (D653, fix round 1).
|
|
3418
|
+
*
|
|
3419
|
+
* Three pieces of state, consulted ON DEMAND by the dispatch that needs a
|
|
3420
|
+
* model — there is no timer anywhere in here:
|
|
3421
|
+
*
|
|
3422
|
+
* 1. A NEGATIVE CACHE per (pool, model). `needsPoolUpdate` stays true after a
|
|
3423
|
+
* failed load, so without it every frame of every camera re-issued the
|
|
3424
|
+
* failing load and wrote two ERRORs. After a failure the model is not
|
|
3425
|
+
* re-issued on that pool for an interval that doubles per consecutive
|
|
3426
|
+
* failure up to a cap; a success clears it.
|
|
3427
|
+
* 2. COMPILE TIMEOUTS per (device, model). A load that outlived its soft
|
|
3428
|
+
* bound is not a crash of the device. The SAME model timing out
|
|
3429
|
+
* {@link COMPILE_TIMEOUTS_BEFORE_REFUSAL} times on one device — which,
|
|
3430
|
+
* since a slow compile now runs on and writes its cache, only a compile
|
|
3431
|
+
* that never finishes can do — refuses THAT MODEL there, by name. The
|
|
3432
|
+
* device keeps serving every other model.
|
|
3433
|
+
* 3. Which camera has already been TOLD about the current state of a model,
|
|
3434
|
+
* so each camera gets one line per state change, never one per frame.
|
|
3435
|
+
*/
|
|
3436
|
+
/** The first back-off after a failed load. */
|
|
3437
|
+
var MODEL_LOAD_BACKOFF_INITIAL_MS = 5e3;
|
|
3438
|
+
/** The back-off doubles per consecutive failure up to this. */
|
|
3439
|
+
var MODEL_LOAD_BACKOFF_MAX_MS = 5 * 6e4;
|
|
3440
|
+
/** Timeouts older than this no longer count toward a refusal. */
|
|
3441
|
+
var COMPILE_TIMEOUT_WINDOW_MS = 60 * 6e4;
|
|
3442
|
+
var ModelLoadGovernor = class {
|
|
3443
|
+
/** Keyed by the pool OBJECT, so a respawned pool starts clean. */
|
|
3444
|
+
backoff = /* @__PURE__ */ new WeakMap();
|
|
3445
|
+
timeouts = /* @__PURE__ */ new Map();
|
|
3446
|
+
reported = /* @__PURE__ */ new Map();
|
|
3447
|
+
generation = 0;
|
|
3448
|
+
/** May this dispatch load `models` on `pool` (a device `deviceKey`) now? */
|
|
3449
|
+
check(pool, deviceKey, models, now = Date.now()) {
|
|
3450
|
+
for (const model of models) {
|
|
3451
|
+
const t = this.timeouts.get(timeoutKey(deviceKey, model));
|
|
3452
|
+
if (t?.refused === true) return {
|
|
3453
|
+
kind: "refused",
|
|
3454
|
+
model,
|
|
3455
|
+
timeouts: t.at.length,
|
|
3456
|
+
error: t.lastError
|
|
3457
|
+
};
|
|
3458
|
+
}
|
|
3459
|
+
const perPool = this.backoff.get(pool);
|
|
3460
|
+
if (perPool === void 0) return { kind: "go" };
|
|
3461
|
+
for (const model of models) {
|
|
3462
|
+
const b = perPool.get(model);
|
|
3463
|
+
if (b !== void 0 && now < b.retryAtMs) return {
|
|
3464
|
+
kind: "backoff",
|
|
3465
|
+
model,
|
|
3466
|
+
retryInMs: b.retryAtMs - now,
|
|
3467
|
+
failures: b.failures,
|
|
3468
|
+
error: b.error,
|
|
3469
|
+
generation: b.generation
|
|
3470
|
+
};
|
|
3471
|
+
}
|
|
3472
|
+
return { kind: "go" };
|
|
3473
|
+
}
|
|
3474
|
+
/** One load of `model` on `pool` failed. */
|
|
3475
|
+
recordFailure(pool, deviceKey, model, error, compileTimeout, now = Date.now(), poolModel = null, reason = null) {
|
|
3476
|
+
let perPool = this.backoff.get(pool);
|
|
3477
|
+
if (perPool === void 0) {
|
|
3478
|
+
perPool = /* @__PURE__ */ new Map();
|
|
3479
|
+
this.backoff.set(pool, perPool);
|
|
3480
|
+
}
|
|
3481
|
+
const previousEntry = perPool.get(model);
|
|
3482
|
+
const failures = (previousEntry?.failures ?? 0) + 1;
|
|
3483
|
+
const retryInMs = Math.min(MODEL_LOAD_BACKOFF_MAX_MS, MODEL_LOAD_BACKOFF_INITIAL_MS * 2 ** (failures - 1));
|
|
3484
|
+
this.generation += 1;
|
|
3485
|
+
const generation = this.generation;
|
|
3486
|
+
perPool.set(model, {
|
|
3487
|
+
failures,
|
|
3488
|
+
retryAtMs: now + retryInMs,
|
|
3489
|
+
error,
|
|
3490
|
+
generation,
|
|
3491
|
+
poolModel: poolModel ?? previousEntry?.poolModel ?? null,
|
|
3492
|
+
reason
|
|
3493
|
+
});
|
|
3494
|
+
if (!compileTimeout) return {
|
|
3495
|
+
generation,
|
|
3496
|
+
retryInMs,
|
|
3497
|
+
refusedNow: false,
|
|
3498
|
+
compileTimeouts: 0
|
|
3499
|
+
};
|
|
3500
|
+
const key = timeoutKey(deviceKey, model);
|
|
3501
|
+
const previous = this.timeouts.get(key);
|
|
3502
|
+
const at = [...(previous?.at ?? []).filter((t) => t >= now - COMPILE_TIMEOUT_WINDOW_MS), now];
|
|
3503
|
+
const refused = at.length >= 2;
|
|
3504
|
+
this.timeouts.set(key, {
|
|
3505
|
+
at,
|
|
3506
|
+
refused,
|
|
3507
|
+
lastError: error
|
|
3508
|
+
});
|
|
3509
|
+
return {
|
|
3510
|
+
generation,
|
|
3511
|
+
retryInMs,
|
|
3512
|
+
refusedNow: refused && previous?.refused !== true,
|
|
3513
|
+
compileTimeouts: at.length
|
|
3514
|
+
};
|
|
3515
|
+
}
|
|
3516
|
+
/**
|
|
3517
|
+
* A compile that outlived its bound came back on `pool` (D653 round 3).
|
|
3518
|
+
*
|
|
3519
|
+
* `ok`: THAT model's cache is written and the worker loads again, so its
|
|
3520
|
+
* back-off is lifted — and only its: another model's failure on the same
|
|
3521
|
+
* pool is not the compile that just finished. Not ok: it is one more
|
|
3522
|
+
* failure of that model, and its back-off grows. Refusals by name are never
|
|
3523
|
+
* lifted here; a model refused for timing out twice stays refused until the
|
|
3524
|
+
* operator re-arms. Returns the governor keys it touched.
|
|
3525
|
+
*/
|
|
3526
|
+
settleLateCompile(pool, deviceKey, poolModel, ok, error, now = Date.now()) {
|
|
3527
|
+
const perPool = this.backoff.get(pool);
|
|
3528
|
+
if (perPool === void 0) return [];
|
|
3529
|
+
const refusedMeanwhile = [...perPool].filter(([, b]) => b.reason === "worker-poisoned" && b.poolModel !== poolModel).map(([key]) => key);
|
|
3530
|
+
for (const key of refusedMeanwhile) perPool.delete(key);
|
|
3531
|
+
const own = poolModel === null ? [] : [...perPool].filter(([, b]) => b.poolModel === poolModel).map(([key]) => key);
|
|
3532
|
+
for (const key of own) if (ok) perPool.delete(key);
|
|
3533
|
+
else this.recordFailure(pool, deviceKey, key, error, false, now, poolModel);
|
|
3534
|
+
return [...refusedMeanwhile, ...own];
|
|
3535
|
+
}
|
|
3536
|
+
/** `models` loaded on `pool`: forget their failures there. */
|
|
3537
|
+
recordSuccess(pool, deviceKey, models) {
|
|
3538
|
+
const perPool = this.backoff.get(pool);
|
|
3539
|
+
for (const model of models) {
|
|
3540
|
+
perPool?.delete(model);
|
|
3541
|
+
this.timeouts.delete(timeoutKey(deviceKey, model));
|
|
3542
|
+
}
|
|
3543
|
+
}
|
|
3544
|
+
/**
|
|
3545
|
+
* Has `camera` been told about this `state` of `model` on `deviceKey` yet?
|
|
3546
|
+
* Returns `true` exactly once per state change.
|
|
3547
|
+
*/
|
|
3548
|
+
shouldReport(camera, deviceKey, model, state) {
|
|
3549
|
+
const key = `${camera ?? "-"}|${deviceKey}|${model}`;
|
|
3550
|
+
if (this.reported.get(key) === state) return false;
|
|
3551
|
+
this.reported.set(key, state);
|
|
3552
|
+
return true;
|
|
3553
|
+
}
|
|
3554
|
+
/** Every model refused on some device. */
|
|
3555
|
+
refused() {
|
|
3556
|
+
const out = [];
|
|
3557
|
+
for (const [key, t] of this.timeouts) {
|
|
3558
|
+
if (!t.refused) continue;
|
|
3559
|
+
const [deviceKey = "", model = ""] = key.split("\0");
|
|
3560
|
+
out.push({
|
|
3561
|
+
deviceKey,
|
|
3562
|
+
model,
|
|
3563
|
+
timeouts: t.at.length,
|
|
3564
|
+
lastError: t.lastError
|
|
3565
|
+
});
|
|
3566
|
+
}
|
|
3567
|
+
return out;
|
|
3568
|
+
}
|
|
3569
|
+
/** Operator re-arm: forget the refusals on `deviceKey`. Returns how many. */
|
|
3570
|
+
rearm(deviceKey) {
|
|
3571
|
+
let cleared = 0;
|
|
3572
|
+
for (const key of [...this.timeouts.keys()]) if (key.startsWith(`${deviceKey}\u0000`)) {
|
|
3573
|
+
this.timeouts.delete(key);
|
|
3574
|
+
cleared += 1;
|
|
3575
|
+
}
|
|
3576
|
+
return cleared;
|
|
3577
|
+
}
|
|
3578
|
+
};
|
|
3579
|
+
function timeoutKey(deviceKey, model) {
|
|
3580
|
+
return `${deviceKey}\u0000${model}`;
|
|
3581
|
+
}
|
|
3582
|
+
//#endregion
|
|
2663
3583
|
//#region src/detection-pipeline/engine/idle-pool-reaper.ts
|
|
2664
3584
|
var IdlePoolReaper = class {
|
|
2665
3585
|
lastUsed = /* @__PURE__ */ new Map();
|
|
@@ -3196,13 +4116,16 @@ function checkReplayPin(input) {
|
|
|
3196
4116
|
}
|
|
3197
4117
|
var PRIMARY_ROOT_STEP = "object-detection";
|
|
3198
4118
|
/**
|
|
3199
|
-
* The
|
|
3200
|
-
*
|
|
3201
|
-
*
|
|
4119
|
+
* The same rule over an EXECUTABLE tree's roots (D658): their model ids are the
|
|
4120
|
+
* pool variants that will run — after resolution, per-step loads and drops —
|
|
4121
|
+
* so this is what a live `runPipeline` stamps. The tree holds only enabled
|
|
4122
|
+
* video roots already.
|
|
3202
4123
|
*/
|
|
3203
|
-
function
|
|
3204
|
-
|
|
3205
|
-
|
|
4124
|
+
function resolveTreeRootModelId(roots) {
|
|
4125
|
+
return primaryRootModelId(roots);
|
|
4126
|
+
}
|
|
4127
|
+
function primaryRootModelId(roots) {
|
|
4128
|
+
return (roots.find((s) => s.stepId === PRIMARY_ROOT_STEP) ?? roots[0])?.modelId;
|
|
3206
4129
|
}
|
|
3207
4130
|
//#endregion
|
|
3208
4131
|
//#region src/detection-pipeline/engine-provisioner.ts
|
|
@@ -5995,26 +6918,47 @@ function resolveEffectiveClassMap(definition, modelId, resolveModel) {
|
|
|
5995
6918
|
* Build an executable tree from user config.
|
|
5996
6919
|
*
|
|
5997
6920
|
* @param steps - User-configured pipeline steps (from PipelineDefaultStep[])
|
|
5998
|
-
* @param getEngine -
|
|
5999
|
-
* Throws if
|
|
6921
|
+
* @param getEngine - Returns the engine that runs exactly `(stepId, modelId)`
|
|
6922
|
+
* and the model it runs. Throws if that pair is not loaded.
|
|
6000
6923
|
*/
|
|
6001
|
-
function buildExecutableTree(steps, getEngine, resolveModel) {
|
|
6002
|
-
return { roots: steps.filter((s) => s.enabled).filter((s) => s.slot !== "audio-classifier").map((s) => buildNode(s, getEngine, resolveModel)) };
|
|
6924
|
+
function buildExecutableTree(steps, getEngine, resolveModel, onMissingVariant) {
|
|
6925
|
+
return { roots: steps.filter((s) => s.enabled).filter((s) => s.slot !== "audio-classifier").map((s) => buildNode(s, getEngine, resolveModel, onMissingVariant)) };
|
|
6926
|
+
}
|
|
6927
|
+
function subtreeIds(step) {
|
|
6928
|
+
return [step.addonId, ...(step.children ?? []).filter((c) => c.enabled).flatMap((c) => subtreeIds(c))];
|
|
6003
6929
|
}
|
|
6004
|
-
function
|
|
6930
|
+
function buildChild(child, parentStepId, getEngine, resolveModel, onMissingVariant) {
|
|
6931
|
+
if (onMissingVariant === void 0) return buildNode(child, getEngine, resolveModel, onMissingVariant);
|
|
6932
|
+
let variant;
|
|
6933
|
+
try {
|
|
6934
|
+
variant = getEngine(child.addonId, child.modelId);
|
|
6935
|
+
} catch (err) {
|
|
6936
|
+
onMissingVariant({
|
|
6937
|
+
step: child,
|
|
6938
|
+
stepId: child.addonId,
|
|
6939
|
+
modelId: child.modelId,
|
|
6940
|
+
parentStepId,
|
|
6941
|
+
droppedSubtree: subtreeIds(child),
|
|
6942
|
+
error: err instanceof Error ? err.message : String(err)
|
|
6943
|
+
});
|
|
6944
|
+
return null;
|
|
6945
|
+
}
|
|
6946
|
+
return buildNode(child, getEngine, resolveModel, onMissingVariant, variant);
|
|
6947
|
+
}
|
|
6948
|
+
function buildNode(step, getEngine, resolveModel, onMissingVariant, resolved) {
|
|
6005
6949
|
const definition = getStepDefinition(step.addonId);
|
|
6006
|
-
const
|
|
6007
|
-
const children = (step.children ?? []).filter((c) => c.enabled).map((c) =>
|
|
6950
|
+
const variant = resolved ?? getEngine(step.addonId, step.modelId);
|
|
6951
|
+
const children = (step.children ?? []).filter((c) => c.enabled).map((c) => buildChild(c, step.addonId, getEngine, resolveModel, onMissingVariant)).filter((c) => c !== null);
|
|
6008
6952
|
const mergedSettings = {
|
|
6009
6953
|
...collectSchemaDefaults(step.addonId),
|
|
6010
6954
|
...step.settings
|
|
6011
6955
|
};
|
|
6012
|
-
const effectiveClassMap = resolveEffectiveClassMap(definition,
|
|
6956
|
+
const effectiveClassMap = resolveEffectiveClassMap(definition, variant.modelId, resolveModel);
|
|
6013
6957
|
return {
|
|
6014
6958
|
stepId: step.addonId,
|
|
6015
6959
|
definition,
|
|
6016
|
-
engine,
|
|
6017
|
-
modelId:
|
|
6960
|
+
engine: variant.engine,
|
|
6961
|
+
modelId: variant.modelId,
|
|
6018
6962
|
inputClasses: definition.inputClasses ?? [],
|
|
6019
6963
|
enabled: step.enabled,
|
|
6020
6964
|
children,
|
|
@@ -6044,43 +6988,6 @@ function walkFieldsForDefaults(fields, out) {
|
|
|
6044
6988
|
}
|
|
6045
6989
|
}
|
|
6046
6990
|
//#endregion
|
|
6047
|
-
//#region src/detection-pipeline/registry/custom-models.ts
|
|
6048
|
-
/**
|
|
6049
|
-
* Group a flat list of custom-model descriptors (as returned by the
|
|
6050
|
-
* `custom-model-registry` collection cap) into a `stepId → entries` map for
|
|
6051
|
-
* the picker / resolution union. Pure; order within a step preserved.
|
|
6052
|
-
*/
|
|
6053
|
-
function groupCustomModelsByStep(descriptors) {
|
|
6054
|
-
const byStep = /* @__PURE__ */ new Map();
|
|
6055
|
-
for (const d of descriptors) {
|
|
6056
|
-
const arr = byStep.get(d.stepId) ?? [];
|
|
6057
|
-
arr.push(d.entry);
|
|
6058
|
-
byStep.set(d.stepId, arr);
|
|
6059
|
-
}
|
|
6060
|
-
return byStep;
|
|
6061
|
-
}
|
|
6062
|
-
/**
|
|
6063
|
-
* Union a step's static catalog models with operator-registered custom
|
|
6064
|
-
* models. On an `id` collision the static catalog entry wins — a custom
|
|
6065
|
-
* model can never shadow a built-in one.
|
|
6066
|
-
*
|
|
6067
|
-
* Pure + side-effect-free so it can be unit-tested in isolation and called
|
|
6068
|
-
* from the (free) `buildSchemaSlots` builder without any addon context.
|
|
6069
|
-
*/
|
|
6070
|
-
function mergeCustomModels(staticModels, customModels) {
|
|
6071
|
-
const seen = new Set(staticModels.map((m) => m.id));
|
|
6072
|
-
const merged = [...staticModels];
|
|
6073
|
-
for (const m of customModels) {
|
|
6074
|
-
if (seen.has(m.id)) continue;
|
|
6075
|
-
seen.add(m.id);
|
|
6076
|
-
merged.push({
|
|
6077
|
-
...m,
|
|
6078
|
-
provider: inferModelProvider(m)
|
|
6079
|
-
});
|
|
6080
|
-
}
|
|
6081
|
-
return merged;
|
|
6082
|
-
}
|
|
6083
|
-
//#endregion
|
|
6084
6991
|
//#region src/detection-pipeline/cluster-model-resolution.ts
|
|
6085
6992
|
/**
|
|
6086
6993
|
* The cluster model rule, enforced on the node that has to obey it.
|
|
@@ -6185,8 +7092,10 @@ function collectSubstitutions(steps, format) {
|
|
|
6185
7092
|
return result;
|
|
6186
7093
|
}
|
|
6187
7094
|
/**
|
|
6188
|
-
* Return one {@link ZeroBuildIssue} per step in `steps`
|
|
6189
|
-
*
|
|
7095
|
+
* Return one {@link ZeroBuildIssue} per step in `steps` that a pool of `format`
|
|
7096
|
+
* cannot run: neither the model it names nor any NON-legacy model of its
|
|
7097
|
+
* catalog (via `getStepDef`) ships a build for `format`
|
|
7098
|
+
* (`stepModelRunsOnFormat`, D657). `steps` is
|
|
6190
7099
|
* expected to already be flattened + filtered to the set the caller cares
|
|
6191
7100
|
* about (e.g. enabled video steps) — this does not recurse into children or
|
|
6192
7101
|
* filter by `enabled`, unlike {@link collectSubstitutions}.
|
|
@@ -6207,7 +7116,7 @@ function collectZeroBuildIssues(steps, format, getStepDef = getStepDefinition) {
|
|
|
6207
7116
|
} catch {
|
|
6208
7117
|
continue;
|
|
6209
7118
|
}
|
|
6210
|
-
if (!def.models.
|
|
7119
|
+
if (!stepModelRunsOnFormat(def.models, step.modelId, format)) result.push({
|
|
6211
7120
|
addonId: step.addonId,
|
|
6212
7121
|
format
|
|
6213
7122
|
});
|
|
@@ -6257,11 +7166,20 @@ function collectUnknownAddonIssues(steps, getStepDef = getStepDefinition) {
|
|
|
6257
7166
|
* `diagnostics.substitutions` instead of happening silently. The caller
|
|
6258
7167
|
* (the provider) owns logging + dedup for both.
|
|
6259
7168
|
*/
|
|
6260
|
-
function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings) {
|
|
7169
|
+
function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings, unknownClusterSteps) {
|
|
6261
7170
|
const resolvedSteps = [];
|
|
6262
7171
|
const unknownAddonIds = [];
|
|
6263
7172
|
const substitutions = [];
|
|
6264
7173
|
const clusterRefusals = [];
|
|
7174
|
+
const clusterUnknown = [];
|
|
7175
|
+
const recurse = (children) => {
|
|
7176
|
+
const r = resolveInputSteps(children, format, engine, clusterModels, clusterSettings, unknownClusterSteps);
|
|
7177
|
+
unknownAddonIds.push(...r.diagnostics.unknownAddonIds);
|
|
7178
|
+
substitutions.push(...r.diagnostics.substitutions);
|
|
7179
|
+
clusterRefusals.push(...r.diagnostics.clusterRefusals);
|
|
7180
|
+
clusterUnknown.push(...r.diagnostics.clusterUnknown);
|
|
7181
|
+
return r;
|
|
7182
|
+
};
|
|
6265
7183
|
for (const s of steps) {
|
|
6266
7184
|
let def;
|
|
6267
7185
|
try {
|
|
@@ -6270,6 +7188,10 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6270
7188
|
unknownAddonIds.push(s.addonId);
|
|
6271
7189
|
continue;
|
|
6272
7190
|
}
|
|
7191
|
+
if (def.modelScope === "cluster" && unknownClusterSteps?.has(s.addonId) === true) {
|
|
7192
|
+
clusterUnknown.push(s.addonId);
|
|
7193
|
+
continue;
|
|
7194
|
+
}
|
|
6273
7195
|
if (clusterModels !== void 0) {
|
|
6274
7196
|
const verdict = resolveClusterStepModel(s.addonId, clusterModels, format, () => def);
|
|
6275
7197
|
if (verdict.kind !== "node-scoped") {
|
|
@@ -6279,12 +7201,7 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6279
7201
|
format: verdict.format,
|
|
6280
7202
|
formatsShipped: verdict.formatsShipped
|
|
6281
7203
|
});
|
|
6282
|
-
const childResult = s.children ?
|
|
6283
|
-
if (childResult) {
|
|
6284
|
-
unknownAddonIds.push(...childResult.diagnostics.unknownAddonIds);
|
|
6285
|
-
substitutions.push(...childResult.diagnostics.substitutions);
|
|
6286
|
-
clusterRefusals.push(...childResult.diagnostics.clusterRefusals);
|
|
6287
|
-
}
|
|
7204
|
+
const childResult = s.children ? recurse(s.children) : null;
|
|
6288
7205
|
const settings = overlayClusterStepSettings(s.addonId, s.settings, clusterSettings ?? {});
|
|
6289
7206
|
resolvedSteps.push({
|
|
6290
7207
|
addonId: s.addonId,
|
|
@@ -6308,12 +7225,7 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6308
7225
|
running: runningModelId,
|
|
6309
7226
|
format
|
|
6310
7227
|
});
|
|
6311
|
-
const childResult = s.children ?
|
|
6312
|
-
if (childResult) {
|
|
6313
|
-
unknownAddonIds.push(...childResult.diagnostics.unknownAddonIds);
|
|
6314
|
-
substitutions.push(...childResult.diagnostics.substitutions);
|
|
6315
|
-
clusterRefusals.push(...childResult.diagnostics.clusterRefusals);
|
|
6316
|
-
}
|
|
7228
|
+
const childResult = s.children ? recurse(s.children) : null;
|
|
6317
7229
|
const settings = overlayClusterStepSettings(s.addonId, s.settings, clusterSettings ?? {});
|
|
6318
7230
|
resolvedSteps.push({
|
|
6319
7231
|
addonId: s.addonId,
|
|
@@ -6332,7 +7244,8 @@ function resolveInputSteps(steps, format, engine, clusterModels, clusterSettings
|
|
|
6332
7244
|
diagnostics: {
|
|
6333
7245
|
unknownAddonIds,
|
|
6334
7246
|
substitutions,
|
|
6335
|
-
clusterRefusals
|
|
7247
|
+
clusterRefusals,
|
|
7248
|
+
clusterUnknown
|
|
6336
7249
|
}
|
|
6337
7250
|
};
|
|
6338
7251
|
}
|
|
@@ -6556,6 +7469,90 @@ function parseWavToAudioChunk(filePath) {
|
|
|
6556
7469
|
function enginesEqual(a, b) {
|
|
6557
7470
|
return a.runtime === b.runtime && a.backend === b.backend && a.format === b.format && (a.device ?? null) === (b.device ?? null);
|
|
6558
7471
|
}
|
|
7472
|
+
/** A dispatch's load refused before it was issued: a back-off, or a refusal by name. */
|
|
7473
|
+
var ModelLoadRefusedError = class extends Error {
|
|
7474
|
+
model;
|
|
7475
|
+
state;
|
|
7476
|
+
constructor(message, model, state) {
|
|
7477
|
+
super(message);
|
|
7478
|
+
this.name = "ModelLoadRefusedError";
|
|
7479
|
+
this.model = model;
|
|
7480
|
+
this.state = state;
|
|
7481
|
+
}
|
|
7482
|
+
};
|
|
7483
|
+
/** `step/model` — how the governor and the log lines name a model. */
|
|
7484
|
+
function stepModelKey(step) {
|
|
7485
|
+
return `${step.addonId}/${step.modelId}`;
|
|
7486
|
+
}
|
|
7487
|
+
/** The reporting state a refusal verdict stands for. */
|
|
7488
|
+
function stateOf(verdict) {
|
|
7489
|
+
return verdict.kind === "refused" ? "refused" : `failed:${verdict.generation}`;
|
|
7490
|
+
}
|
|
7491
|
+
/** The load policy of a `runPipeline` call — see {@link DispatchLoadPolicy}. */
|
|
7492
|
+
function dispatchLoadPolicyOf(input) {
|
|
7493
|
+
return input.deviceId !== void 0 && input.replay !== true ? "partial" : "all-or-nothing";
|
|
7494
|
+
}
|
|
7495
|
+
/**
|
|
7496
|
+
* An all-or-nothing call lost steps to failed loads (D657). Names every one,
|
|
7497
|
+
* so a benchmark never reports timings for a tree it did not run.
|
|
7498
|
+
*/
|
|
7499
|
+
var DispatchStepsNotLoadedError = class extends Error {
|
|
7500
|
+
failedSteps;
|
|
7501
|
+
constructor(failures) {
|
|
7502
|
+
const named = failures.map((f) => `${f.step.addonId}/${f.step.modelId} (${f.error instanceof Error ? f.error.message : String(f.error)})`);
|
|
7503
|
+
super(`pipeline steps failed to load — the call runs all of them or none: ${named.join("; ")}`, { cause: failures[0]?.error });
|
|
7504
|
+
this.name = "DispatchStepsNotLoadedError";
|
|
7505
|
+
this.failedSteps = failures.map((f) => `${f.step.addonId}/${f.step.modelId}`);
|
|
7506
|
+
}
|
|
7507
|
+
};
|
|
7508
|
+
/**
|
|
7509
|
+
* A step's model artifact cannot be made present for the pool's format — no
|
|
7510
|
+
* build for that format, a custom model missing from this node, a download
|
|
7511
|
+
* that failed. Named by step and model so the failure is charged to THAT
|
|
7512
|
+
* model only (D657): a plain `Error` was charged to every model of the set,
|
|
7513
|
+
* and one face model with no tflite build backed the Coral's root detector
|
|
7514
|
+
* off for minutes.
|
|
7515
|
+
*/
|
|
7516
|
+
var StepModelArtifactError = class extends Error {
|
|
7517
|
+
stepId;
|
|
7518
|
+
modelId;
|
|
7519
|
+
constructor(stepId, modelId, cause) {
|
|
7520
|
+
super(cause instanceof Error ? cause.message : String(cause), { cause });
|
|
7521
|
+
this.name = "StepModelArtifactError";
|
|
7522
|
+
this.stepId = stepId;
|
|
7523
|
+
this.modelId = modelId;
|
|
7524
|
+
}
|
|
7525
|
+
};
|
|
7526
|
+
/** The one model a failed load is charged to, when the error names it. */
|
|
7527
|
+
function failedModelOf(err) {
|
|
7528
|
+
if (err instanceof StepVariantLoadError || err instanceof StepModelArtifactError) return `${err.stepId}/${err.modelId}`;
|
|
7529
|
+
return null;
|
|
7530
|
+
}
|
|
7531
|
+
/**
|
|
7532
|
+
* `steps` without the subtrees of the steps whose model did not load (D657).
|
|
7533
|
+
* A step's children consume its output, so they leave with it. Matched by
|
|
7534
|
+
* INSTANCE: two steps of one addon in a tree are two steps.
|
|
7535
|
+
*/
|
|
7536
|
+
function withoutFailedSteps(steps, failed) {
|
|
7537
|
+
const kept = [];
|
|
7538
|
+
for (const step of steps) {
|
|
7539
|
+
if (failed.has(step)) continue;
|
|
7540
|
+
kept.push(step.children && step.children.length > 0 ? {
|
|
7541
|
+
...step,
|
|
7542
|
+
children: withoutFailedSteps(step.children, failed)
|
|
7543
|
+
} : step);
|
|
7544
|
+
}
|
|
7545
|
+
return kept;
|
|
7546
|
+
}
|
|
7547
|
+
/**
|
|
7548
|
+
* The reason a condemned pool is charged against its device's restart budget.
|
|
7549
|
+
* A pool recycled because it HUNG (D653) names the hang — `inference device
|
|
7550
|
+
* FAILED … reason: pool worker is not ready` could not tell a GPU compile that
|
|
7551
|
+
* never returned from a crash.
|
|
7552
|
+
*/
|
|
7553
|
+
function deathReasonOf(factory) {
|
|
7554
|
+
return factory.getDeathCause()?.message ?? "pool worker is not ready";
|
|
7555
|
+
}
|
|
6559
7556
|
/** Build a `RuntimeEnv` from the running process + probed hardware. */
|
|
6560
7557
|
function runtimeEnvFromProcess(hardware) {
|
|
6561
7558
|
return {
|
|
@@ -6654,18 +7651,18 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6654
7651
|
* paths don't issue a cap round-trip on every call. Empty map = no provider
|
|
6655
7652
|
* (or a query failure) → behaviour identical to the static-catalog-only path.
|
|
6656
7653
|
*/
|
|
6657
|
-
|
|
7654
|
+
customModels = new CustomModelCatalog({ warn: (message, extras) => this.log.warn(message, extras) });
|
|
6658
7655
|
/**
|
|
6659
7656
|
* SYNC custom-model lookup handed to every live {@link EngineFactory} so
|
|
6660
7657
|
* `buildPoolModelConfigForStep` can resolve an operator-registered model at
|
|
6661
|
-
* pool-load time. Reads the {@link
|
|
7658
|
+
* pool-load time. Reads the {@link customModels} snapshot — every load
|
|
6662
7659
|
* path (`ensureModelsForSteps`, `ensureModelsForCurrentSteps`) refreshes the
|
|
6663
7660
|
* cache via `resolveModelEntry` BEFORE `loadAdditional`/`applyConfig`, so
|
|
6664
7661
|
* the snapshot is fresh when the pool asks. Without this seam a selected
|
|
6665
7662
|
* custom model failed the WHOLE pool applyConfig with
|
|
6666
7663
|
* `Model "X" not found in step "Y" catalog`.
|
|
6667
7664
|
*/
|
|
6668
|
-
customModelResolver = (stepId, modelId) => this.
|
|
7665
|
+
customModelResolver = (stepId, modelId) => this.customModels.current().get(stepId)?.find((m) => m.id === modelId);
|
|
6669
7666
|
/**
|
|
6670
7667
|
* Per-device {@link DeviceProxy} cache used for zone gating at the
|
|
6671
7668
|
* runtime path. Reads `state.zones.value` + `state.zoneRules.value`
|
|
@@ -6688,6 +7685,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
6688
7685
|
info: (message, extras) => this.log.info(message, extras),
|
|
6689
7686
|
warn: (message, extras) => this.log.warn(message, extras)
|
|
6690
7687
|
} });
|
|
7688
|
+
/** One line per camera when a cluster-scoped step starts running another model (D658). */
|
|
7689
|
+
clusterModelSwitchLog = new ClusterModelSwitchLog({ info: (message, extras) => this.log.info(message, extras) });
|
|
7690
|
+
/** One line per (camera, step) while a cluster step is refused as unknown (D658). */
|
|
7691
|
+
clusterUnknownLog = new ClusterUnknownRefusalLog({ warn: (message, extras) => this.log.warn(message, extras) });
|
|
6691
7692
|
/**
|
|
6692
7693
|
* Last logged "step configured" signature — used to dedupe the
|
|
6693
7694
|
* per-step debug trail so it only fires on actual config CHANGE
|
|
@@ -7154,7 +8155,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7154
8155
|
if (!steps) return;
|
|
7155
8156
|
const format = this.currentEngine.format;
|
|
7156
8157
|
const substitutionIssues = this.getActiveModelSubstitutions().map((s) => `${s.addonId}: chose ${s.chosen}, running ${s.running} (${s.format})`);
|
|
7157
|
-
const zeroBuildIssues = collectZeroBuildIssues(flattenSteps(steps), format).map((i) => `${i.addonId}: no model has a ${i.format} build`);
|
|
8158
|
+
const zeroBuildIssues = collectZeroBuildIssues(flattenSteps(steps), format, this.mergedStepDefOrThrow).map((i) => `${i.addonId}: no model has a ${i.format} build`);
|
|
7158
8159
|
this.configIssues = [...substitutionIssues, ...zeroBuildIssues];
|
|
7159
8160
|
} catch (err) {
|
|
7160
8161
|
this.configIssues = [];
|
|
@@ -7288,24 +8289,21 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7288
8289
|
* a warning — callers then behave exactly like the static-catalog path.
|
|
7289
8290
|
*/
|
|
7290
8291
|
async getCustomModels() {
|
|
7291
|
-
|
|
7292
|
-
if (this.customModelsCache && now - this.customModelsCache.at < 5e3) return this.customModelsCache.byStep;
|
|
7293
|
-
let byStep = /* @__PURE__ */ new Map();
|
|
7294
|
-
try {
|
|
7295
|
-
const api = this.addonCtx?.api;
|
|
7296
|
-
if (api) {
|
|
7297
|
-
if ((await api.addons.listCapabilityProviders.query({ capName: "custom-model-registry" })).some((p) => p.isActive)) byStep = groupCustomModelsByStep(await api.customModelRegistry.listModels.query());
|
|
7298
|
-
}
|
|
7299
|
-
} catch (err) {
|
|
7300
|
-
this.log.warn("custom-model-registry query failed — using static catalog only", { meta: { error: errMsg(err) } });
|
|
7301
|
-
}
|
|
7302
|
-
this.customModelsCache = {
|
|
7303
|
-
at: now,
|
|
7304
|
-
byStep
|
|
7305
|
-
};
|
|
7306
|
-
return byStep;
|
|
8292
|
+
return this.customModels.refresh(this.addonCtx?.api);
|
|
7307
8293
|
}
|
|
7308
8294
|
/**
|
|
8295
|
+
* A step's definition MERGED with its custom models, synchronously, from
|
|
8296
|
+
* the last custom-model read (D657 round 2). Every placement gate here
|
|
8297
|
+
* judges this catalog — the one the orchestrator's schema carries.
|
|
8298
|
+
*/
|
|
8299
|
+
mergedStepDef = (addonId) => this.customModels.stepDefinition(addonId);
|
|
8300
|
+
/** {@link mergedStepDef} as the throwing lookup `collectZeroBuildIssues` takes. */
|
|
8301
|
+
mergedStepDefOrThrow = (addonId) => {
|
|
8302
|
+
const def = this.mergedStepDef(addonId);
|
|
8303
|
+
if (def === null) throw new Error(`Unknown pipeline step: "${addonId}"`);
|
|
8304
|
+
return def;
|
|
8305
|
+
};
|
|
8306
|
+
/**
|
|
7309
8307
|
* Resolve a model id within a step to a catalog entry — static catalog
|
|
7310
8308
|
* first, then the custom registry. Returns undefined if neither has it.
|
|
7311
8309
|
*/
|
|
@@ -7496,7 +8494,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7496
8494
|
const enabledSteps = this.pruneToEnabledSteps(steps);
|
|
7497
8495
|
const substitutions = collectSubstitutions(enabledSteps, format);
|
|
7498
8496
|
const unknownAddonIssues = collectUnknownAddonIssues(enabledSteps);
|
|
7499
|
-
const zeroBuildIssues = collectZeroBuildIssues(this.flattenEnabledVideoStepInputs(steps), format);
|
|
8497
|
+
const zeroBuildIssues = collectZeroBuildIssues(this.flattenEnabledVideoStepInputs(steps), format, this.mergedStepDefOrThrow);
|
|
7500
8498
|
const issues = [...unknownAddonIssues.map((u) => ({
|
|
7501
8499
|
addonId: u.addonId,
|
|
7502
8500
|
kind: "unknown-addon",
|
|
@@ -7813,6 +8811,12 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7813
8811
|
}
|
|
7814
8812
|
async runPipeline(input, onProgress) {
|
|
7815
8813
|
await this.clusterModels.refresh(this.addonCtx?.api);
|
|
8814
|
+
this.clusterUnknownLog.note(input.deviceId ?? 0, input.steps, this.clusterModels.unknownSteps());
|
|
8815
|
+
if (dispatchLoadPolicyOf(input) === "all-or-nothing") {
|
|
8816
|
+
const unknown = unknownStepsNamed(input.steps, this.clusterModels.unknownSteps());
|
|
8817
|
+
if (unknown.length > 0) throw new Error(`cluster-model-unknown: the owner has not answered which model the cluster runs for ${unknown.join(", ")} — the call runs all of its steps or none`);
|
|
8818
|
+
}
|
|
8819
|
+
await this.getCustomModels();
|
|
7816
8820
|
const nodeId = this.addonCtx?.kernel?.localNodeId ?? "hub";
|
|
7817
8821
|
const sessionId = input.sessionId ?? `run-${Date.now()}-${Math.random().toString(36).slice(2, 10)}`;
|
|
7818
8822
|
const isRuntime = Boolean(input.frame || input.frameRef);
|
|
@@ -7977,11 +8981,6 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
7977
8981
|
}
|
|
7978
8982
|
const enabledSteps = flattenSteps(benchmarkSteps);
|
|
7979
8983
|
emit(`Pipeline: ${enabledSteps.length} step(s) — ${enabledSteps.map((s) => s.addonId).join(" → ")}`);
|
|
7980
|
-
const rootModelId = resolveRootModelId(benchmarkSteps);
|
|
7981
|
-
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
7982
|
-
...r,
|
|
7983
|
-
rootModelId
|
|
7984
|
-
};
|
|
7985
8984
|
if (input.replay === true) {
|
|
7986
8985
|
const verdict = checkReplayPin({
|
|
7987
8986
|
requested: input.steps,
|
|
@@ -8023,21 +9022,51 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8023
9022
|
const dispatchEngine = dispatchDeviceKey ? resolveDeviceEngine(dispatchDeviceKey) : runEngine;
|
|
8024
9023
|
const dispatchFactory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8025
9024
|
const needed = enabledSteps.filter((s) => dispatchFactory.needsPoolUpdate(s));
|
|
9025
|
+
let runSteps = benchmarkSteps;
|
|
9026
|
+
const loadPolicy = dispatchLoadPolicyOf(input);
|
|
8026
9027
|
if (needed.length > 0) {
|
|
8027
|
-
|
|
8028
|
-
|
|
8029
|
-
|
|
9028
|
+
const dispatchDeviceKeyLabel = deviceKeyOf(dispatchEngine);
|
|
9029
|
+
const toIssue = needed.filter((s) => this.loadGovernor.check(dispatchFactory, dispatchDeviceKeyLabel, [stepModelKey(s)]).kind === "go");
|
|
9030
|
+
if (toIssue.length > 0) this.log.info("Benchmark: models to load", { meta: {
|
|
9031
|
+
count: toIssue.length,
|
|
9032
|
+
models: toIssue.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8030
9033
|
} });
|
|
8031
|
-
for (const s of
|
|
9034
|
+
for (const s of toIssue) emit(`Loading model: ${s.addonName} (${s.modelId})...`, {
|
|
8032
9035
|
step: s.addonId,
|
|
8033
9036
|
addonId: s.addonId,
|
|
8034
9037
|
modelId: s.modelId
|
|
8035
9038
|
});
|
|
8036
|
-
await this.
|
|
8037
|
-
|
|
9039
|
+
const failures = await this.loadDispatchModels(benchmarkSteps, dispatchFactory, dispatchEngine.format, {
|
|
9040
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
9041
|
+
deviceKey: dispatchDeviceKeyLabel
|
|
9042
|
+
}, loadPolicy);
|
|
9043
|
+
runSteps = stepsSurvivingLoad(benchmarkSteps, failures, loadPolicy);
|
|
9044
|
+
emit(failures.length === 0 ? `All models loaded` : `Models loaded (${failures.length} failed)`);
|
|
8038
9045
|
}
|
|
8039
9046
|
emit("Running inference...");
|
|
8040
|
-
const
|
|
9047
|
+
const notWarm = [];
|
|
9048
|
+
const tree = buildExecutableTree(runSteps, (stepId, modelId) => dispatchFactory.getStepVariant(stepId, modelId), this.customModelResolver, (drop) => notWarm.push({
|
|
9049
|
+
step: drop.step,
|
|
9050
|
+
error: new Error(drop.error)
|
|
9051
|
+
}));
|
|
9052
|
+
if (notWarm.length > 0) {
|
|
9053
|
+
if (loadPolicy === "all-or-nothing") throw new DispatchStepsNotLoadedError(notWarm);
|
|
9054
|
+
const dispatchDeviceKeyLabel = deviceKeyOf(dispatchEngine);
|
|
9055
|
+
for (const failure of notWarm) {
|
|
9056
|
+
const model = stepModelKey(failure.step);
|
|
9057
|
+
this.reportDispatchLoadLoss({
|
|
9058
|
+
...input.deviceId !== void 0 ? { deviceId: input.deviceId } : {},
|
|
9059
|
+
deviceKey: dispatchDeviceKeyLabel,
|
|
9060
|
+
outcome: "step-dropped"
|
|
9061
|
+
}, dispatchDeviceKeyLabel, [model], model, "not-warm", errMsg(failure.error));
|
|
9062
|
+
}
|
|
9063
|
+
}
|
|
9064
|
+
const rootModelId = resolveTreeRootModelId(tree.roots);
|
|
9065
|
+
const stampRoot = (r) => rootModelId === void 0 ? r : {
|
|
9066
|
+
...r,
|
|
9067
|
+
rootModelId
|
|
9068
|
+
};
|
|
9069
|
+
if (input.replay !== true) this.clusterModelSwitchLog.note(input.deviceId ?? 0, tree);
|
|
8041
9070
|
setupMs = performance.now() - wallT0 - decodeMs;
|
|
8042
9071
|
const effectiveDeviceId = input.deviceId ?? 0;
|
|
8043
9072
|
const deviceOverrides = effectiveDeviceId > 0 ? await this.deviceOverrides.resolve(effectiveDeviceId) : {};
|
|
@@ -8143,7 +9172,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8143
9172
|
backend: this.currentEngine.backend,
|
|
8144
9173
|
device: this.currentEngine.device ?? null
|
|
8145
9174
|
}, clusterRefusalsOut) {
|
|
8146
|
-
const { steps: resolvedSteps, diagnostics } = resolveInputSteps(steps, format, engine, this.clusterModels.current(), this.clusterModels.currentSettings());
|
|
9175
|
+
const { steps: resolvedSteps, diagnostics } = resolveInputSteps(steps, format, engine, this.clusterModels.current(), this.clusterModels.currentSettings(), this.clusterModels.unknownSteps());
|
|
8147
9176
|
if (clusterRefusalsOut) clusterRefusalsOut.push(...diagnostics.clusterRefusals);
|
|
8148
9177
|
for (const addonId of diagnostics.unknownAddonIds) this.logLiveDispatchIssueOnce(`unknown:${addonId}`, () => this.log.warn("Live pipeline step references an unknown addon — dropping step", { meta: { addonId } }));
|
|
8149
9178
|
for (const sub of diagnostics.substitutions) this.logLiveDispatchIssueOnce(`${sub.addonId}|${sub.chosen}|${sub.running}|${sub.format}`, () => this.log.info("Live pipeline step model substituted for engine format (node-local)", { meta: {
|
|
@@ -8256,7 +9285,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8256
9285
|
const deviceRefusals = [];
|
|
8257
9286
|
const steps = this.inputStepsToPipelineSteps(args.steps, deviceFormat, deviceStepEngine, deviceRefusals);
|
|
8258
9287
|
const framePlane = args.plane === "frame";
|
|
8259
|
-
const zeroBuild = collectZeroBuildIssues(framePlane ? collectFramePlaneSteps(steps, isDetailPlaneStep) : flattenSteps(steps), deviceFormat);
|
|
9288
|
+
const zeroBuild = collectZeroBuildIssues(framePlane ? collectFramePlaneSteps(steps, isDetailPlaneStep) : flattenSteps(steps), deviceFormat, this.mergedStepDefOrThrow);
|
|
8260
9289
|
if (zeroBuild.length === 0) {
|
|
8261
9290
|
if (!framePlane) {
|
|
8262
9291
|
this.logClusterRefusals(deviceRefusals, deviceId, `${deviceKey} (${deviceFormat})`);
|
|
@@ -8265,7 +9294,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8265
9294
|
steps
|
|
8266
9295
|
};
|
|
8267
9296
|
}
|
|
8268
|
-
const pruneResult = pruneUnrunnableDetailSteps(steps, isDetailPlaneStep, (
|
|
9297
|
+
const pruneResult = pruneUnrunnableDetailSteps(steps, isDetailPlaneStep, (subtree) => !buildSubtreeCanRun(subtree, this.mergedStepDef)(deviceFormat));
|
|
8269
9298
|
if (pruneResult.prunedAddonIds.length > 0) {
|
|
8270
9299
|
const prunedList = pruneResult.prunedAddonIds.join(",");
|
|
8271
9300
|
this.logLiveDispatchIssueOnce(`detail-defer:${deviceKey}:${deviceFormat}:${prunedList}`, () => this.log.info("Detail steps have no model build for the dispatch device format — deferred to the detail-plane device jump", {
|
|
@@ -8277,7 +9306,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8277
9306
|
}
|
|
8278
9307
|
}));
|
|
8279
9308
|
}
|
|
8280
|
-
this.logClusterRefusals(deviceRefusals.filter((r) => !pruneResult.
|
|
9309
|
+
this.logClusterRefusals(deviceRefusals.filter((r) => !pruneResult.removedAddonIds.includes(r.stepId)), deviceId, `${deviceKey} (${deviceFormat})`);
|
|
8281
9310
|
return {
|
|
8282
9311
|
deviceKey,
|
|
8283
9312
|
steps: pruneResult.steps
|
|
@@ -8311,7 +9340,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8311
9340
|
* microseconds instead of a 3-retry download backoff).
|
|
8312
9341
|
*/
|
|
8313
9342
|
assertStepsRunnable(steps, format, deviceId, enginesTried) {
|
|
8314
|
-
const zeroBuild = collectZeroBuildIssues(flattenSteps(steps), format);
|
|
9343
|
+
const zeroBuild = collectZeroBuildIssues(flattenSteps(steps), format, this.mergedStepDefOrThrow);
|
|
8315
9344
|
if (zeroBuild.length === 0) return;
|
|
8316
9345
|
const detail = zeroBuild.map((issue) => {
|
|
8317
9346
|
const step = flattenSteps(steps).find((s) => s.addonId === issue.addonId);
|
|
@@ -8330,53 +9359,235 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8330
9359
|
throw new Error(message);
|
|
8331
9360
|
}
|
|
8332
9361
|
/**
|
|
8333
|
-
*
|
|
8334
|
-
*
|
|
8335
|
-
*
|
|
8336
|
-
*
|
|
8337
|
-
*
|
|
8338
|
-
*
|
|
8339
|
-
*
|
|
8340
|
-
*/
|
|
8341
|
-
modelLoadInFlight =
|
|
9362
|
+
* In-flight model loads, per POOL and per MODEL SET (D653). Two dispatches
|
|
9363
|
+
* needing the same models on the same pool share ONE load and its outcome;
|
|
9364
|
+
* a dispatch needing a different set is never handed another set's failure
|
|
9365
|
+
* — it queues behind the pool's current load ({@link modelLoadTail}) and
|
|
9366
|
+
* then decides for itself. Keyed by pool, not node-wide: one GPU compile
|
|
9367
|
+
* that never returned held EVERY load on the node — the NPU's and the CPU's
|
|
9368
|
+
* too — behind it.
|
|
9369
|
+
*/
|
|
9370
|
+
modelLoadInFlight = /* @__PURE__ */ new Map();
|
|
9371
|
+
/** The last load issued on each pool — loads into one pool run one at a time. */
|
|
9372
|
+
modelLoadTail = /* @__PURE__ */ new Map();
|
|
9373
|
+
/** Stands in for "the node-default pool" before it exists. */
|
|
9374
|
+
nullPoolKey = {};
|
|
9375
|
+
/** Negative cache, per-model compile-timeout refusals, per-camera reporting (D653). */
|
|
9376
|
+
loadGovernor = new ModelLoadGovernor();
|
|
9377
|
+
/**
|
|
9378
|
+
* Load a dispatch's models ONE STEP AT A TIME (D657), and answer which steps
|
|
9379
|
+
* did not load. Never throws: each step is its own load, joined, backed off
|
|
9380
|
+
* and charged on its own model, so a step whose model cannot load on this
|
|
9381
|
+
* pool — no build for its format, a compile that failed — costs that step,
|
|
9382
|
+
* never the root or a sibling. The caller decides what the dispatch runs
|
|
9383
|
+
* without it ({@link stepsSurvivingLoad}).
|
|
9384
|
+
*/
|
|
9385
|
+
async loadDispatchModels(steps, factory, format, context, policy) {
|
|
9386
|
+
const failures = [];
|
|
9387
|
+
const walk = async (nodes, root) => {
|
|
9388
|
+
for (const step of nodes) {
|
|
9389
|
+
if (!step.enabled || step.slot === "audio-classifier") continue;
|
|
9390
|
+
if (factory === null || factory.needsPoolUpdate(step)) try {
|
|
9391
|
+
await this.ensureModelsForSteps([{
|
|
9392
|
+
...step,
|
|
9393
|
+
children: []
|
|
9394
|
+
}], factory, format, {
|
|
9395
|
+
...context,
|
|
9396
|
+
outcome: policy === "all-or-nothing" ? "call-failed" : root ? "root-lost" : "step-dropped"
|
|
9397
|
+
});
|
|
9398
|
+
} catch (err) {
|
|
9399
|
+
failures.push({
|
|
9400
|
+
step,
|
|
9401
|
+
error: err
|
|
9402
|
+
});
|
|
9403
|
+
continue;
|
|
9404
|
+
}
|
|
9405
|
+
await walk(step.children ?? [], false);
|
|
9406
|
+
}
|
|
9407
|
+
};
|
|
9408
|
+
await walk(steps, true);
|
|
9409
|
+
return failures;
|
|
9410
|
+
}
|
|
8342
9411
|
/** Ensure all models needed by steps are downloaded and loaded in the engine pool.
|
|
8343
9412
|
* `factory`/`format` default to the node's engine; a per-device dispatch passes
|
|
8344
|
-
* that device's factory + format (Phase 2 multi-device).
|
|
8345
|
-
|
|
8346
|
-
|
|
8347
|
-
const
|
|
9413
|
+
* that device's factory + format (Phase 2 multi-device). `context` names the
|
|
9414
|
+
* camera and the accelerator on the line a rejected load writes. */
|
|
9415
|
+
async ensureModelsForSteps(steps, factory = this.engineFactory, format = this.currentEngine?.format ?? "onnx", context = {}) {
|
|
9416
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9417
|
+
if (needed.length === 0) return;
|
|
9418
|
+
const pool = factory ?? this.nullPoolKey;
|
|
9419
|
+
const deviceKey = context.deviceKey ?? deviceKeyOf(this.currentEngine);
|
|
9420
|
+
const models = needed.map(stepModelKey);
|
|
9421
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9422
|
+
if (refusal !== null) {
|
|
9423
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, refusal.model, refusal.state, refusal.message);
|
|
9424
|
+
throw refusal;
|
|
9425
|
+
}
|
|
9426
|
+
const setKey = [...models].sort().join(",");
|
|
9427
|
+
let perPool = this.modelLoadInFlight.get(pool);
|
|
9428
|
+
if (perPool === void 0) {
|
|
9429
|
+
perPool = /* @__PURE__ */ new Map();
|
|
9430
|
+
this.modelLoadInFlight.set(pool, perPool);
|
|
9431
|
+
}
|
|
9432
|
+
const flight = perPool.get(setKey) ?? this.issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey);
|
|
9433
|
+
try {
|
|
9434
|
+
await flight.work;
|
|
9435
|
+
} catch (err) {
|
|
9436
|
+
if (err instanceof ModelLoadRefusedError) {
|
|
9437
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, err.model, err.state, err.message);
|
|
9438
|
+
throw err;
|
|
9439
|
+
}
|
|
9440
|
+
const failed = failedModelOf(err) ?? models[0] ?? "";
|
|
9441
|
+
this.reportDispatchLoadLoss(context, deviceKey, models, failed, `failed:${flight.generation ?? "unknown"}`, errMsg(err));
|
|
9442
|
+
throw err;
|
|
9443
|
+
}
|
|
9444
|
+
}
|
|
9445
|
+
/**
|
|
9446
|
+
* Why `models` may not be loaded on `pool` right now — a back-off after a
|
|
9447
|
+
* failure, or a refusal by name after repeated compile timeouts — or `null`.
|
|
9448
|
+
*/
|
|
9449
|
+
loadRefusal(pool, deviceKey, models) {
|
|
9450
|
+
const verdict = this.loadGovernor.check(pool, deviceKey, models);
|
|
9451
|
+
if (verdict.kind === "go") return null;
|
|
9452
|
+
return new ModelLoadRefusedError(verdict.kind === "refused" ? `model ${verdict.model} is refused on ${deviceKey}: its compile timed out ${verdict.timeouts} times (${verdict.error}) — re-arm with pipelineExecutor.rearmInferenceDevice` : `model ${verdict.model} failed to load on ${deviceKey} ${verdict.failures} time(s); not retried for ${verdict.retryInMs}ms (${verdict.error})`, verdict.model, stateOf(verdict));
|
|
9453
|
+
}
|
|
9454
|
+
/**
|
|
9455
|
+
* A compile that outlived its soft bound came back on `factory`'s pool
|
|
9456
|
+
* (D653 rounds 2-3). A SUCCESS wrote that model's cache and the worker loads
|
|
9457
|
+
* again, so that model's back-off is lifted at once rather than waited out;
|
|
9458
|
+
* no other model's is. A FAILURE is one more failure of that model, and its
|
|
9459
|
+
* back-off grows.
|
|
9460
|
+
*/
|
|
9461
|
+
noteLateCompile(factory, deviceKey, event) {
|
|
9462
|
+
const models = this.loadGovernor.settleLateCompile(factory, deviceKey, event.model, event.ok, event.error ?? "late compile failed");
|
|
9463
|
+
if (event.ok) {
|
|
9464
|
+
this.log.info("model loadable again after late compile", { meta: {
|
|
9465
|
+
deviceKey,
|
|
9466
|
+
model: event.model,
|
|
9467
|
+
backoffLifted: models
|
|
9468
|
+
} });
|
|
9469
|
+
return;
|
|
9470
|
+
}
|
|
9471
|
+
this.log.warn("late compile failed — the model stays backed off", { meta: {
|
|
9472
|
+
deviceKey,
|
|
9473
|
+
model: event.model,
|
|
9474
|
+
backoffExtended: models,
|
|
9475
|
+
error: event.error ?? null
|
|
9476
|
+
} });
|
|
9477
|
+
}
|
|
9478
|
+
/** The enabled steps the factory's pool does not reflect yet. */
|
|
9479
|
+
stepsNeedingLoad(steps, factory) {
|
|
8348
9480
|
const needed = [];
|
|
8349
|
-
for (const step of
|
|
9481
|
+
for (const step of flattenSteps(steps)) {
|
|
8350
9482
|
if (!step.enabled) continue;
|
|
8351
9483
|
if (factory !== null && !factory.needsPoolUpdate(step)) continue;
|
|
8352
9484
|
needed.push(step);
|
|
8353
9485
|
}
|
|
8354
|
-
|
|
8355
|
-
|
|
8356
|
-
|
|
8357
|
-
|
|
8358
|
-
|
|
8359
|
-
|
|
8360
|
-
|
|
8361
|
-
|
|
8362
|
-
|
|
8363
|
-
|
|
8364
|
-
|
|
8365
|
-
|
|
8366
|
-
|
|
8367
|
-
|
|
9486
|
+
return needed;
|
|
9487
|
+
}
|
|
9488
|
+
/**
|
|
9489
|
+
* Issue ONE load for a model set on a pool, queued behind the pool's
|
|
9490
|
+
* previous load, and record its outcome with the governor. The set is
|
|
9491
|
+
* re-derived once the queue reaches it: the load ahead may have brought
|
|
9492
|
+
* some of it in.
|
|
9493
|
+
*/
|
|
9494
|
+
issueModelLoad(pool, perPool, setKey, steps, factory, format, deviceKey) {
|
|
9495
|
+
const previous = this.modelLoadTail.get(pool) ?? Promise.resolve();
|
|
9496
|
+
const flight = {
|
|
9497
|
+
work: Promise.resolve(),
|
|
9498
|
+
generation: null
|
|
9499
|
+
};
|
|
9500
|
+
flight.work = (async () => {
|
|
9501
|
+
await previous;
|
|
9502
|
+
const needed = this.stepsNeedingLoad(steps, factory);
|
|
9503
|
+
if (needed.length === 0) return;
|
|
9504
|
+
const models = needed.map(stepModelKey);
|
|
9505
|
+
const refusal = this.loadRefusal(pool, deviceKey, models);
|
|
9506
|
+
if (refusal !== null) throw refusal;
|
|
9507
|
+
try {
|
|
9508
|
+
await this.downloadNeededModels(needed, format);
|
|
9509
|
+
this.log.info("Loading additional models for benchmark", { meta: {
|
|
9510
|
+
count: needed.length,
|
|
9511
|
+
models,
|
|
9512
|
+
deviceKey
|
|
9513
|
+
} });
|
|
9514
|
+
await factory.loadAdditional(needed);
|
|
9515
|
+
} catch (err) {
|
|
9516
|
+
flight.generation = this.recordModelLoadFailure(pool, deviceKey, models, err);
|
|
9517
|
+
throw err;
|
|
8368
9518
|
}
|
|
8369
|
-
this.
|
|
8370
|
-
count: needed.length,
|
|
8371
|
-
models: needed.map((s) => `${s.addonId}/${s.modelId}`)
|
|
8372
|
-
} });
|
|
8373
|
-
await factory.loadAdditional(needed);
|
|
9519
|
+
this.loadGovernor.recordSuccess(pool, deviceKey, models);
|
|
8374
9520
|
})();
|
|
8375
|
-
|
|
8376
|
-
|
|
8377
|
-
|
|
8378
|
-
|
|
8379
|
-
if (
|
|
9521
|
+
perPool.set(setKey, flight);
|
|
9522
|
+
const tail = flight.work.catch(() => void 0);
|
|
9523
|
+
this.modelLoadTail.set(pool, tail);
|
|
9524
|
+
tail.then(() => {
|
|
9525
|
+
if (perPool.get(setKey) === flight) perPool.delete(setKey);
|
|
9526
|
+
if (this.modelLoadTail.get(pool) === tail) this.modelLoadTail.delete(pool);
|
|
9527
|
+
});
|
|
9528
|
+
return flight;
|
|
9529
|
+
}
|
|
9530
|
+
/** Charge a failed load to the model that failed; say so once if it is now refused. */
|
|
9531
|
+
recordModelLoadFailure(pool, deviceKey, models, err) {
|
|
9532
|
+
const named = failedModelOf(err);
|
|
9533
|
+
const failedModels = named !== null ? [named] : models;
|
|
9534
|
+
const compileTimeout = err instanceof StepVariantLoadError && err.reason === "compile-timeout";
|
|
9535
|
+
const poolModel = err instanceof StepVariantLoadError ? err.poolModel : null;
|
|
9536
|
+
let generation = 0;
|
|
9537
|
+
for (const model of failedModels) {
|
|
9538
|
+
const outcome = this.loadGovernor.recordFailure(pool, deviceKey, model, errMsg(err), compileTimeout, Date.now(), poolModel, err instanceof StepVariantLoadError ? err.reason : null);
|
|
9539
|
+
generation = outcome.generation;
|
|
9540
|
+
if (outcome.refusedNow) this.log.error("model REFUSED on this device — its compile timed out repeatedly", { meta: {
|
|
9541
|
+
deviceKey,
|
|
9542
|
+
model,
|
|
9543
|
+
compileTimeouts: outcome.compileTimeouts,
|
|
9544
|
+
error: errMsg(err),
|
|
9545
|
+
device: "keeps serving every other model",
|
|
9546
|
+
rearm: "pipelineExecutor.rearmInferenceDevice"
|
|
9547
|
+
} });
|
|
9548
|
+
}
|
|
9549
|
+
return generation;
|
|
9550
|
+
}
|
|
9551
|
+
/**
|
|
9552
|
+
* One WARN per camera per state change of the model that cost it its frame.
|
|
9553
|
+
* The ERROR for the failed load itself is written once, where the load
|
|
9554
|
+
* failed (`Step variant load failed`); this line answers "which cameras
|
|
9555
|
+
* did it cost?", tagged so it can be counted per camera.
|
|
9556
|
+
*/
|
|
9557
|
+
reportDispatchLoadLoss(context, deviceKey, models, model, state, error) {
|
|
9558
|
+
if (!this.loadGovernor.shouldReport(context.deviceId, deviceKey, model, state)) return;
|
|
9559
|
+
this.log.warn("model load for dispatch failed", {
|
|
9560
|
+
...context.deviceId !== void 0 ? { tags: { deviceId: context.deviceId } } : {},
|
|
9561
|
+
meta: {
|
|
9562
|
+
deviceKey,
|
|
9563
|
+
models,
|
|
9564
|
+
model,
|
|
9565
|
+
state,
|
|
9566
|
+
error,
|
|
9567
|
+
...context.outcome !== void 0 ? { outcome: context.outcome } : {}
|
|
9568
|
+
}
|
|
9569
|
+
});
|
|
9570
|
+
}
|
|
9571
|
+
/** Download any model of `needed` missing in `format` (the device's). */
|
|
9572
|
+
async downloadNeededModels(needed, format) {
|
|
9573
|
+
for (const step of needed) try {
|
|
9574
|
+
await this.ensureStepModelArtifact(step, format);
|
|
9575
|
+
} catch (err) {
|
|
9576
|
+
throw new StepModelArtifactError(step.addonId, step.modelId, err);
|
|
9577
|
+
}
|
|
9578
|
+
}
|
|
9579
|
+
/** Make one step's model artifact present on disk for `format`, or say why it cannot be. */
|
|
9580
|
+
async ensureStepModelArtifact(step, format) {
|
|
9581
|
+
const modelEntry = await this.resolveModelEntry(step.addonId, step.modelId);
|
|
9582
|
+
if (modelEntry && !isModelDownloaded(this.modelsDir, modelEntry, format)) {
|
|
9583
|
+
if (modelEntry.formats[format] === void 0) throw new Error(`Model "${modelEntry.id}" has no ${format} format build (available: ${Object.keys(modelEntry.formats).join(", ") || "none"}) — not retrying a permanent format mismatch`);
|
|
9584
|
+
if (modelEntry.formats[format]?.url.startsWith("camstack-local://") === true) throw new Error(`Custom model "${modelEntry.id}" (${format}) is not present in this node's models directory — distribute it to this node from Model Studio before selecting it here`);
|
|
9585
|
+
this.log.info("Downloading model for step", { meta: {
|
|
9586
|
+
modelId: step.modelId,
|
|
9587
|
+
format,
|
|
9588
|
+
step: step.addonId
|
|
9589
|
+
} });
|
|
9590
|
+
await this.downloadWithRetry(modelEntry, format, 3);
|
|
8380
9591
|
}
|
|
8381
9592
|
}
|
|
8382
9593
|
/** Download a model with retry + exponential backoff */
|
|
@@ -8474,6 +9685,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8474
9685
|
}
|
|
8475
9686
|
async runPipelineBatchImpl(input) {
|
|
8476
9687
|
if (input.frames.length === 0) return { results: [] };
|
|
9688
|
+
await this.getCustomModels();
|
|
8477
9689
|
const dispatchResolution = this.resolveStepsForDispatch({
|
|
8478
9690
|
steps: input.steps,
|
|
8479
9691
|
deviceKey: input.deviceKey,
|
|
@@ -8504,7 +9716,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8504
9716
|
await this.ensureEngineFactory();
|
|
8505
9717
|
const factory = dispatchDeviceKey ? await this.resolveDeviceFactory(dispatchDeviceKey) : this.engineFactory;
|
|
8506
9718
|
if (!factory) throw new Error("runPipelineBatch: factory not initialised");
|
|
8507
|
-
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat);
|
|
9719
|
+
if (enabledSteps.filter((s) => factory.needsPoolUpdate(s)).length > 0) await this.ensureModelsForSteps(benchmarkSteps, factory, resolveFormat, { deviceKey: dispatchDeviceKey ?? deviceKeyOf(this.currentEngine) });
|
|
8508
9720
|
const canFastPath = singleRoot && allRaw && uniformDims && factory.supportsBatch();
|
|
8509
9721
|
this.log.info("runPipelineBatch path decision", { meta: {
|
|
8510
9722
|
phase: "batch",
|
|
@@ -8716,9 +9928,10 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8716
9928
|
return this.engineFactory.cacheFrameInPool(buf, input.width, input.height, input.format);
|
|
8717
9929
|
}
|
|
8718
9930
|
async inferCached(input) {
|
|
9931
|
+
if (typeof input.modelId !== "string" || input.modelId === "") throw new Error(`inferCached: no modelId for step "${input.stepId}" — refusing rather than running the step's active variant`);
|
|
8719
9932
|
await this.ensureEngineFactory();
|
|
8720
9933
|
if (!this.engineFactory) throw new Error("inferCached: factory not initialized");
|
|
8721
|
-
return this.engineFactory.inferCached(input.stepId, input.frameId);
|
|
9934
|
+
return this.engineFactory.inferCached(input.stepId, input.frameId, input.modelId);
|
|
8722
9935
|
}
|
|
8723
9936
|
async uncacheFrame(input) {
|
|
8724
9937
|
if (!this.engineFactory) return;
|
|
@@ -8759,7 +9972,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8759
9972
|
if (this.engineFactory.isReady()) return;
|
|
8760
9973
|
const dead = this.engineFactory;
|
|
8761
9974
|
this.engineFactory = null;
|
|
8762
|
-
this.
|
|
9975
|
+
this.noteFactoryDeath(defaultDeviceKey, dead);
|
|
8763
9976
|
await dead.dispose().catch(() => void 0);
|
|
8764
9977
|
this.refuseIfDeviceUnusable(defaultDeviceKey);
|
|
8765
9978
|
}
|
|
@@ -8777,7 +9990,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8777
9990
|
logger: this.log.child("engine"),
|
|
8778
9991
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8779
9992
|
provisioning: this.executorOptions.provisioning,
|
|
8780
|
-
resolveCustomModel: this.customModelResolver
|
|
9993
|
+
resolveCustomModel: this.customModelResolver,
|
|
9994
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, defaultDeviceKey, event)
|
|
8781
9995
|
});
|
|
8782
9996
|
try {
|
|
8783
9997
|
await factory.initialize([]);
|
|
@@ -8897,15 +10111,35 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8897
10111
|
* dispose — which rejects whatever was still in flight on it with a reason
|
|
8898
10112
|
* rather than letting those requests sit until their own deadlines.
|
|
8899
10113
|
*/
|
|
8900
|
-
async condemnDeviceFactory(deviceKey, factory
|
|
10114
|
+
async condemnDeviceFactory(deviceKey, factory) {
|
|
8901
10115
|
if (this.factoriesByDevice.get(deviceKey) === factory) {
|
|
8902
10116
|
this.factoriesByDevice.delete(deviceKey);
|
|
8903
10117
|
this.deviceReaper.cancel(deviceKey);
|
|
8904
|
-
this.
|
|
10118
|
+
this.noteFactoryDeath(deviceKey, factory);
|
|
8905
10119
|
}
|
|
8906
10120
|
await factory.dispose().catch(() => void 0);
|
|
8907
10121
|
}
|
|
8908
10122
|
/**
|
|
10123
|
+
* Charge one dead pool to its device — unless it died of a compile that was
|
|
10124
|
+
* still running at its hard bound (D653, fix round 1). That death is not a
|
|
10125
|
+
* crash of the DEVICE: the model that hung was already counted when its
|
|
10126
|
+
* load answered `compile-timeout`, and the same model timing out again on
|
|
10127
|
+
* the respawn is refused by name (`ModelLoadGovernor`), which is what bounds
|
|
10128
|
+
* the respawns. A crash stays a crash.
|
|
10129
|
+
*/
|
|
10130
|
+
noteFactoryDeath(deviceKey, factory) {
|
|
10131
|
+
const cause = factory.getDeathCause();
|
|
10132
|
+
if (cause !== null && !cause.chargesDeviceBudget) {
|
|
10133
|
+
this.log.warn("inference pool recycled after a compile hung — not charged to the device budget", { meta: {
|
|
10134
|
+
deviceKey,
|
|
10135
|
+
reason: cause.reason,
|
|
10136
|
+
cause: cause.message
|
|
10137
|
+
} });
|
|
10138
|
+
return;
|
|
10139
|
+
}
|
|
10140
|
+
this.noteDeviceDeath(deviceKey, cause?.message ?? "pool worker is not ready");
|
|
10141
|
+
}
|
|
10142
|
+
/**
|
|
8909
10143
|
* Every inference device this node currently refuses, and why.
|
|
8910
10144
|
*
|
|
8911
10145
|
* The CHANNEL the 2026-08-26 analysis found missing. Pool health lived
|
|
@@ -8940,7 +10174,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8940
10174
|
state: "backoff",
|
|
8941
10175
|
since: Date.now(),
|
|
8942
10176
|
deaths: 0,
|
|
8943
|
-
lastError:
|
|
10177
|
+
lastError: deathReasonOf(factory)
|
|
8944
10178
|
});
|
|
8945
10179
|
}
|
|
8946
10180
|
return { unhealthy: [...byKey.values()].toSorted((a, b) => a.deviceKey.localeCompare(b.deviceKey)) };
|
|
@@ -8957,10 +10191,13 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8957
10191
|
* no-op, reported as such.
|
|
8958
10192
|
*/
|
|
8959
10193
|
async rearmInferenceDevice(input) {
|
|
8960
|
-
const
|
|
10194
|
+
const rearmedDevice = this.deviceLiveness.rearm(input.deviceKey);
|
|
10195
|
+
const rearmedModels = this.loadGovernor.rearm(input.deviceKey);
|
|
10196
|
+
const rearmed = rearmedDevice || rearmedModels > 0;
|
|
8961
10197
|
this.log.info("inference device re-armed by operator", { meta: {
|
|
8962
10198
|
deviceKey: input.deviceKey,
|
|
8963
|
-
rearmed
|
|
10199
|
+
rearmed,
|
|
10200
|
+
rearmedModels
|
|
8964
10201
|
} });
|
|
8965
10202
|
return { rearmed };
|
|
8966
10203
|
}
|
|
@@ -8977,7 +10214,7 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8977
10214
|
this.deviceReaper.touch(deviceKey);
|
|
8978
10215
|
return existing;
|
|
8979
10216
|
}
|
|
8980
|
-
await this.condemnDeviceFactory(deviceKey, existing
|
|
10217
|
+
await this.condemnDeviceFactory(deviceKey, existing);
|
|
8981
10218
|
this.refuseIfDeviceUnusable(deviceKey);
|
|
8982
10219
|
}
|
|
8983
10220
|
const inflight = this.deviceFactoryInflight.get(deviceKey);
|
|
@@ -8990,7 +10227,8 @@ var DetectionPipelineProvider = class DetectionPipelineProvider {
|
|
|
8990
10227
|
logger: this.log.child(`engine:${deviceKey}`),
|
|
8991
10228
|
pythonPath: this.executorOptions.pythonPath ?? "",
|
|
8992
10229
|
provisioning: this.executorOptions.provisioning,
|
|
8993
|
-
resolveCustomModel: this.customModelResolver
|
|
10230
|
+
resolveCustomModel: this.customModelResolver,
|
|
10231
|
+
onCompileFinishedLate: (event) => this.noteLateCompile(factory, deviceKey, event)
|
|
8994
10232
|
});
|
|
8995
10233
|
try {
|
|
8996
10234
|
await factory.initialize([]);
|
|
@@ -9497,13 +10735,22 @@ function isDetailPlaneStep(addonId) {
|
|
|
9497
10735
|
}
|
|
9498
10736
|
}
|
|
9499
10737
|
/**
|
|
9500
|
-
*
|
|
9501
|
-
*
|
|
9502
|
-
*
|
|
9503
|
-
*
|
|
9504
|
-
|
|
9505
|
-
|
|
9506
|
-
|
|
10738
|
+
* What a dispatch runs after its per-step loads (D657).
|
|
10739
|
+
*
|
|
10740
|
+
* An ALL-OR-NOTHING call (a benchmark, a replay — D56 — a capacity ramp) fails
|
|
10741
|
+
* on any failure, naming every step it lost ({@link DispatchStepsNotLoadedError}):
|
|
10742
|
+
* it measures or reproduces exactly what it names. A camera's LIVE dispatch
|
|
10743
|
+
* runs the tree without the steps whose model did not load, and fails — with
|
|
10744
|
+
* the first root's own named error — only when no root survives: a dispatch
|
|
10745
|
+
* with no root has nothing to produce.
|
|
10746
|
+
*/
|
|
10747
|
+
function stepsSurvivingLoad(steps, failures, policy) {
|
|
10748
|
+
if (failures.length === 0) return steps;
|
|
10749
|
+
if (policy === "all-or-nothing") throw new DispatchStepsNotLoadedError(failures);
|
|
10750
|
+
const kept = withoutFailedSteps(steps, new Set(failures.map((f) => f.step)));
|
|
10751
|
+
if (kept.some((s) => s.enabled)) return kept;
|
|
10752
|
+
const roots = new Set(steps);
|
|
10753
|
+
throw (failures.find((f) => roots.has(f.step)) ?? failures[0]).error;
|
|
9507
10754
|
}
|
|
9508
10755
|
/**
|
|
9509
10756
|
* The union of model formats a step's catalog ships ANY build for — the
|