@camstack/addon-matter-broker 0.2.58 → 0.2.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/addon.js +428 -25
- package/dist/addon.mjs +428 -25
- package/package.json +1 -1
package/dist/addon.js
CHANGED
|
@@ -12836,7 +12836,26 @@ var DiscoveryCandidateSchema = object({
|
|
|
12836
12836
|
* identity ahead of adoption. Rendering metadata (unit, precision)
|
|
12837
12837
|
* flows live through the cap STATUS SLICE after adoption.
|
|
12838
12838
|
*/
|
|
12839
|
-
sourceInfo: SourceInfoSchema.optional()
|
|
12839
|
+
sourceInfo: SourceInfoSchema.optional(),
|
|
12840
|
+
/**
|
|
12841
|
+
* Set when this candidate is a device the provider ALREADY owns.
|
|
12842
|
+
*
|
|
12843
|
+
* A scan cannot generally produce the identity a device was onboarded under
|
|
12844
|
+
* (Reolink keys on `mac-<mac>`, learned at adopt time), so a stableId
|
|
12845
|
+
* comparison never matches and an owned device looks addable. Re-adopting one
|
|
12846
|
+
* overwrites its config with scan-derived values — that is how a Home Hub's
|
|
12847
|
+
* Baichuan port was overwritten with its ONVIF port, taking the hub and its
|
|
12848
|
+
* three child cameras offline for four hours.
|
|
12849
|
+
*
|
|
12850
|
+
* A provider that can recognise its own devices says so here. Absent means
|
|
12851
|
+
* "not recognised", which is not the same as "known to be new" — a provider
|
|
12852
|
+
* that cannot tell simply never sets it.
|
|
12853
|
+
*/
|
|
12854
|
+
alreadyOnboarded: boolean().optional(),
|
|
12855
|
+
/** Numeric id of the device this candidate was matched to. Set with `alreadyOnboarded`. */
|
|
12856
|
+
onboardedDeviceId: number().optional(),
|
|
12857
|
+
/** Operator-facing name of the matched device, so the UI can say WHICH one it is. */
|
|
12858
|
+
onboardedName: string$2().optional()
|
|
12840
12859
|
});
|
|
12841
12860
|
/**
|
|
12842
12861
|
* Flat device summary returned by `createDevice` / `adoptDiscoveredDevice`.
|
|
@@ -12892,6 +12911,35 @@ var deviceProviderCapability = {
|
|
|
12892
12911
|
name: string$2(),
|
|
12893
12912
|
type: string$2()
|
|
12894
12913
|
}))),
|
|
12914
|
+
/**
|
|
12915
|
+
* Tear down and reconstruct ONE device in place from its persisted rows —
|
|
12916
|
+
* touching no other device this provider owns.
|
|
12917
|
+
*
|
|
12918
|
+
* The primitive `deviceManager.migrateDevice` uses to flush the two
|
|
12919
|
+
* migrated numbers: after `swapIds` the runner's live instance still
|
|
12920
|
+
* carries the PRE-swap numeric id (baked into the object, its native-cap
|
|
12921
|
+
* registrations and its log tags), and a live object cannot be renumbered.
|
|
12922
|
+
* Before this method the only flush was restarting the whole owning addon
|
|
12923
|
+
* — which took every camera the provider owns down with it (28 devices
|
|
12924
|
+
* for one migrated camera, measured 2026-09-04, and the morning of the
|
|
12925
|
+
* same day ~27 devices' native caps did not come back on their own).
|
|
12926
|
+
*
|
|
12927
|
+
* Keyed by `stableId`, deliberately: the numeric id is exactly the thing
|
|
12928
|
+
* that changes. The reply carries the id the device answers on NOW.
|
|
12929
|
+
* Implemented once in `BaseDeviceProvider` — decommission the live
|
|
12930
|
+
* instance (if any), then re-create from the persisted row: the same
|
|
12931
|
+
* teardown/rehydrate pair every graceful shutdown + boot already uses.
|
|
12932
|
+
* An RPC, never an event: a dropped event would leave the runner writing
|
|
12933
|
+
* against the wrong camera (D8).
|
|
12934
|
+
*
|
|
12935
|
+
* Construction can dial hardware, and the migrated source is
|
|
12936
|
+
* characteristically dead — the timeout covers a full activate window
|
|
12937
|
+
* rather than the 60 s default.
|
|
12938
|
+
*/
|
|
12939
|
+
reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
|
|
12940
|
+
kind: "mutation",
|
|
12941
|
+
timeoutMs: 3 * 6e4
|
|
12942
|
+
}),
|
|
12895
12943
|
supportsDiscovery: method(object({}), boolean()),
|
|
12896
12944
|
/**
|
|
12897
12945
|
* Run a network scan. `params` carries optional provider-specific scan
|
|
@@ -13219,7 +13267,8 @@ method(object({
|
|
|
13219
13267
|
targetId: number()
|
|
13220
13268
|
}), MigrateDeviceResultSchema, {
|
|
13221
13269
|
kind: "mutation",
|
|
13222
|
-
auth: "admin"
|
|
13270
|
+
auth: "admin",
|
|
13271
|
+
timeoutMs: 12 * 6e4
|
|
13223
13272
|
}), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
|
|
13224
13273
|
deviceId: number(),
|
|
13225
13274
|
name: string$2()
|
|
@@ -32592,6 +32641,147 @@ var BaseDevice = class {
|
|
|
32592
32641
|
}
|
|
32593
32642
|
};
|
|
32594
32643
|
/**
|
|
32644
|
+
* Delays before retry rounds 1..N — the round count IS the bound.
|
|
32645
|
+
* 10 s catches "the hub was busy for a moment"; the full schedule
|
|
32646
|
+
* (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
|
|
32647
|
+
* per attempt) covers a device-manager lock held for minutes — the
|
|
32648
|
+
* 2026-09-04 outage's migration hold was ~3.5 min.
|
|
32649
|
+
*/
|
|
32650
|
+
var DEVICE_RESTORE_RETRY_DELAYS_MS = [
|
|
32651
|
+
1e4,
|
|
32652
|
+
3e4,
|
|
32653
|
+
9e4
|
|
32654
|
+
];
|
|
32655
|
+
/** Abortable sleep — resolves early (never rejects) on abort. */
|
|
32656
|
+
function sleep$1(ms, signal) {
|
|
32657
|
+
return new Promise((resolve) => {
|
|
32658
|
+
if (signal.aborted) {
|
|
32659
|
+
resolve();
|
|
32660
|
+
return;
|
|
32661
|
+
}
|
|
32662
|
+
const onAbort = () => {
|
|
32663
|
+
clearTimeout(timer);
|
|
32664
|
+
resolve();
|
|
32665
|
+
};
|
|
32666
|
+
const timer = setTimeout(() => {
|
|
32667
|
+
signal.removeEventListener("abort", onAbort);
|
|
32668
|
+
resolve();
|
|
32669
|
+
}, ms);
|
|
32670
|
+
timer.unref?.();
|
|
32671
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
32672
|
+
});
|
|
32673
|
+
}
|
|
32674
|
+
/** Drain `items` through at most `width` concurrent lanes. `fn` must
|
|
32675
|
+
* not reject (callers wrap their own try/catch). */
|
|
32676
|
+
async function runWithConcurrency(items, width, fn) {
|
|
32677
|
+
const queue = [...items];
|
|
32678
|
+
const laneCount = Math.max(1, Math.min(width, queue.length));
|
|
32679
|
+
const lane = async () => {
|
|
32680
|
+
for (;;) {
|
|
32681
|
+
const item = queue.shift();
|
|
32682
|
+
if (item === void 0) return;
|
|
32683
|
+
await fn(item);
|
|
32684
|
+
}
|
|
32685
|
+
};
|
|
32686
|
+
await Promise.all(Array.from({ length: laneCount }, lane));
|
|
32687
|
+
}
|
|
32688
|
+
var DeviceRestoreRetryScheduler = class {
|
|
32689
|
+
#logger;
|
|
32690
|
+
#attempt;
|
|
32691
|
+
#onPermanentFailure;
|
|
32692
|
+
#delaysMs;
|
|
32693
|
+
#concurrency;
|
|
32694
|
+
#now;
|
|
32695
|
+
#abort = new AbortController();
|
|
32696
|
+
constructor(options) {
|
|
32697
|
+
this.#logger = options.logger;
|
|
32698
|
+
this.#attempt = options.attempt;
|
|
32699
|
+
this.#onPermanentFailure = options.onPermanentFailure;
|
|
32700
|
+
this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
|
|
32701
|
+
this.#concurrency = options.concurrency ?? 4;
|
|
32702
|
+
this.#now = options.now ?? Date.now;
|
|
32703
|
+
}
|
|
32704
|
+
/** Stop retrying (shutdown). Pending entries are NOT marked
|
|
32705
|
+
* permanently failed — the next boot restores them from disk. */
|
|
32706
|
+
cancel() {
|
|
32707
|
+
this.#abort.abort();
|
|
32708
|
+
}
|
|
32709
|
+
/**
|
|
32710
|
+
* Run the bounded retry rounds. Resolves when every entry has either
|
|
32711
|
+
* restored, been marked permanently failed, or the scheduler was
|
|
32712
|
+
* cancelled. Never rejects.
|
|
32713
|
+
*/
|
|
32714
|
+
async run(initialFailures) {
|
|
32715
|
+
let pending = initialFailures.map((failure) => ({
|
|
32716
|
+
saved: failure.saved,
|
|
32717
|
+
lastError: failure.error,
|
|
32718
|
+
attempts: 1
|
|
32719
|
+
}));
|
|
32720
|
+
for (let round = 0; round < this.#delaysMs.length; round += 1) {
|
|
32721
|
+
if (pending.length === 0 || this.#abort.signal.aborted) break;
|
|
32722
|
+
await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
|
|
32723
|
+
if (this.#abort.signal.aborted) break;
|
|
32724
|
+
pending = await this.#runRound(pending, round);
|
|
32725
|
+
}
|
|
32726
|
+
if (this.#abort.signal.aborted) return [];
|
|
32727
|
+
const terminal = pending.map((entry) => ({
|
|
32728
|
+
deviceId: entry.saved.id,
|
|
32729
|
+
stableId: entry.saved.stableId,
|
|
32730
|
+
type: String(entry.saved.type),
|
|
32731
|
+
attempts: entry.attempts,
|
|
32732
|
+
lastError: entry.lastError,
|
|
32733
|
+
failedAt: this.#now()
|
|
32734
|
+
}));
|
|
32735
|
+
for (const failure of terminal) this.#onPermanentFailure(failure);
|
|
32736
|
+
return terminal;
|
|
32737
|
+
}
|
|
32738
|
+
/** One retry round: parents first (phase 0), then hub-adopted
|
|
32739
|
+
* children (phase 1) — a child's attempt depends on its parent
|
|
32740
|
+
* having landed, exactly like the initial two-pass restore. */
|
|
32741
|
+
async #runRound(pending, round) {
|
|
32742
|
+
const next = [];
|
|
32743
|
+
const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
|
|
32744
|
+
const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
|
|
32745
|
+
for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
|
|
32746
|
+
if (this.#abort.signal.aborted) {
|
|
32747
|
+
next.push(entry);
|
|
32748
|
+
return;
|
|
32749
|
+
}
|
|
32750
|
+
const attemptNo = entry.attempts + 1;
|
|
32751
|
+
try {
|
|
32752
|
+
await this.#attempt(entry.saved);
|
|
32753
|
+
this.#logger.info("Device restored on retry", {
|
|
32754
|
+
tags: {
|
|
32755
|
+
deviceId: entry.saved.id,
|
|
32756
|
+
stableId: entry.saved.stableId
|
|
32757
|
+
},
|
|
32758
|
+
meta: { attempt: attemptNo }
|
|
32759
|
+
});
|
|
32760
|
+
} catch (err) {
|
|
32761
|
+
const lastError = err instanceof Error ? err.message : String(err);
|
|
32762
|
+
const remainingRetries = this.#delaysMs.length - (round + 1);
|
|
32763
|
+
this.#logger.warn("Device restore retry failed", {
|
|
32764
|
+
tags: {
|
|
32765
|
+
deviceId: entry.saved.id,
|
|
32766
|
+
stableId: entry.saved.stableId
|
|
32767
|
+
},
|
|
32768
|
+
meta: {
|
|
32769
|
+
attempt: attemptNo,
|
|
32770
|
+
remainingRetries,
|
|
32771
|
+
error: lastError
|
|
32772
|
+
}
|
|
32773
|
+
});
|
|
32774
|
+
next.push({
|
|
32775
|
+
saved: entry.saved,
|
|
32776
|
+
lastError,
|
|
32777
|
+
attempts: attemptNo
|
|
32778
|
+
});
|
|
32779
|
+
}
|
|
32780
|
+
});
|
|
32781
|
+
return next;
|
|
32782
|
+
}
|
|
32783
|
+
};
|
|
32784
|
+
/**
|
|
32595
32785
|
* Convert an IDevice to the flat DeviceSummary shape expected by the
|
|
32596
32786
|
* device-provider cap router. Shared across all providers.
|
|
32597
32787
|
*/
|
|
@@ -32640,6 +32830,7 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32640
32830
|
}];
|
|
32641
32831
|
}
|
|
32642
32832
|
async onShutdown() {
|
|
32833
|
+
this.cancelRestoreRetries();
|
|
32643
32834
|
const devices = await this.ctx.kernel.devices?.getAll() ?? [];
|
|
32644
32835
|
for (const device of devices) try {
|
|
32645
32836
|
await this.ctx.kernel.devices?.decommission(device.id);
|
|
@@ -32657,9 +32848,16 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32657
32848
|
async start() {}
|
|
32658
32849
|
async stop() {}
|
|
32659
32850
|
async getStatus() {
|
|
32851
|
+
const all = await this.ctx.kernel.devices?.getAll() ?? [];
|
|
32852
|
+
const summary = this.restoreFailureSummary();
|
|
32853
|
+
if (summary === null) return {
|
|
32854
|
+
connected: true,
|
|
32855
|
+
deviceCount: all.length
|
|
32856
|
+
};
|
|
32660
32857
|
return {
|
|
32661
32858
|
connected: true,
|
|
32662
|
-
deviceCount:
|
|
32859
|
+
deviceCount: all.length,
|
|
32860
|
+
error: summary
|
|
32663
32861
|
};
|
|
32664
32862
|
}
|
|
32665
32863
|
async getDevices() {
|
|
@@ -32749,8 +32947,137 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32749
32947
|
};
|
|
32750
32948
|
}
|
|
32751
32949
|
async restoreDevices(savedDevices) {
|
|
32752
|
-
await this.onRestoreDevices(savedDevices);
|
|
32753
|
-
if (savedDevices.length
|
|
32950
|
+
const report = await this.onRestoreDevices(savedDevices);
|
|
32951
|
+
if (savedDevices.length === 0) return;
|
|
32952
|
+
if (report && report.failedCount > 0) {
|
|
32953
|
+
this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
|
|
32954
|
+
return;
|
|
32955
|
+
}
|
|
32956
|
+
const restoredCount = report ? report.restoredCount : savedDevices.length;
|
|
32957
|
+
this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
|
|
32958
|
+
}
|
|
32959
|
+
/** Retry schedule. Overridable (tests use millisecond delays). */
|
|
32960
|
+
restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
|
|
32961
|
+
/** Retry lane width. See `device-restore-retry.ts` for why retries
|
|
32962
|
+
* never re-stampede full-width while the initial pass does (D167). */
|
|
32963
|
+
restoreRetryConcurrency = 4;
|
|
32964
|
+
_restoreRetryScheduler = null;
|
|
32965
|
+
_restoreRetryCompletion = null;
|
|
32966
|
+
_permanentRestoreFailures = /* @__PURE__ */ new Map();
|
|
32967
|
+
/** Settles when the background retry rounds finish (or `null` when
|
|
32968
|
+
* nothing failed). Exposed for tests and subclass diagnostics —
|
|
32969
|
+
* boot NEVER awaits this: the runner's post-init handshake goes out
|
|
32970
|
+
* with the devices that restored, and a late success is announced
|
|
32971
|
+
* through the `native-cap-change` → `updateCaps` path. */
|
|
32972
|
+
get restoreRetryCompletion() {
|
|
32973
|
+
return this._restoreRetryCompletion;
|
|
32974
|
+
}
|
|
32975
|
+
/** Devices that exhausted the retry bound this process lifetime. */
|
|
32976
|
+
get permanentRestoreFailures() {
|
|
32977
|
+
return [...this._permanentRestoreFailures.values()];
|
|
32978
|
+
}
|
|
32979
|
+
/** One-line operator-facing summary for `getStatus().error`, or
|
|
32980
|
+
* `null` when every device restored. */
|
|
32981
|
+
restoreFailureSummary() {
|
|
32982
|
+
if (this._permanentRestoreFailures.size === 0) return null;
|
|
32983
|
+
const ids = [...this._permanentRestoreFailures.keys()].join(", ");
|
|
32984
|
+
return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
|
|
32985
|
+
}
|
|
32986
|
+
cancelRestoreRetries() {
|
|
32987
|
+
this._restoreRetryScheduler?.cancel();
|
|
32988
|
+
this._restoreRetryScheduler = null;
|
|
32989
|
+
}
|
|
32990
|
+
recordPermanentRestoreFailure(failure) {
|
|
32991
|
+
this._permanentRestoreFailures.set(failure.deviceId, failure);
|
|
32992
|
+
this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
|
|
32993
|
+
tags: {
|
|
32994
|
+
deviceId: failure.deviceId,
|
|
32995
|
+
stableId: failure.stableId
|
|
32996
|
+
},
|
|
32997
|
+
meta: {
|
|
32998
|
+
type: failure.type,
|
|
32999
|
+
attempts: failure.attempts,
|
|
33000
|
+
error: failure.lastError
|
|
33001
|
+
}
|
|
33002
|
+
});
|
|
33003
|
+
}
|
|
33004
|
+
scheduleRestoreRetries(failures, attempt) {
|
|
33005
|
+
const scheduler = new DeviceRestoreRetryScheduler({
|
|
33006
|
+
logger: this.ctx.logger,
|
|
33007
|
+
delaysMs: this.restoreRetryDelaysMs,
|
|
33008
|
+
concurrency: this.restoreRetryConcurrency,
|
|
33009
|
+
attempt,
|
|
33010
|
+
onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
|
|
33011
|
+
});
|
|
33012
|
+
this._restoreRetryScheduler = scheduler;
|
|
33013
|
+
this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
|
|
33014
|
+
this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
33015
|
+
});
|
|
33016
|
+
}
|
|
33017
|
+
/**
|
|
33018
|
+
* Tear down and reconstruct ONE device from its persisted rows — the
|
|
33019
|
+
* `deviceProvider.reloadDevice` cap method. Persistence is never touched,
|
|
33020
|
+
* and no other device this provider owns is disturbed.
|
|
33021
|
+
*
|
|
33022
|
+
* Keyed by `stableId` because the caller's whole reason to be here is that
|
|
33023
|
+
* the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
|
|
33024
|
+
* fresh instance resolves its id through `allocateDeviceId`, which returns
|
|
33025
|
+
* whatever number the row carries NOW. The teardown is `decommission` —
|
|
33026
|
+
* exactly what a graceful shutdown runs per device (fires `removeDevice()`,
|
|
33027
|
+
* unregisters native caps, drops the registry entry) — and the rebuild is
|
|
33028
|
+
* the boot restore's own `create()` path, including its pass 2: first-class
|
|
33029
|
+
* children (hub-adopted cameras under an NVR) are decommissioned with the
|
|
33030
|
+
* parent by the cascade and must be re-created explicitly, because only
|
|
33031
|
+
* accessory children come back through `getAccessoryChildren()`.
|
|
33032
|
+
*
|
|
33033
|
+
* Reloading an accessory child directly is refused (no device class) —
|
|
33034
|
+
* reload its parent instead.
|
|
33035
|
+
*/
|
|
33036
|
+
async reloadDevice(input) {
|
|
33037
|
+
const { stableId } = input;
|
|
33038
|
+
const devices = this.ctx.kernel.devices;
|
|
33039
|
+
if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
|
|
33040
|
+
const live = (await devices.getAll()).find((d) => d.stableId === stableId);
|
|
33041
|
+
if (live) await devices.decommission(live.id);
|
|
33042
|
+
const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
|
|
33043
|
+
addonId: this.addonId,
|
|
33044
|
+
stableId
|
|
33045
|
+
});
|
|
33046
|
+
const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
|
|
33047
|
+
if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
|
|
33048
|
+
const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
|
|
33049
|
+
const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
|
|
33050
|
+
if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
|
|
33051
|
+
await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
|
|
33052
|
+
const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
|
|
33053
|
+
for (const row of rows) {
|
|
33054
|
+
if (row.parentDeviceId !== id) continue;
|
|
33055
|
+
const childType = Object.values(DeviceType).find((t) => t === row.type);
|
|
33056
|
+
const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
|
|
33057
|
+
if (!ChildClass) continue;
|
|
33058
|
+
try {
|
|
33059
|
+
await devices.create(row.stableId, ChildClass, {}, id);
|
|
33060
|
+
} catch (err) {
|
|
33061
|
+
this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
|
|
33062
|
+
tags: {
|
|
33063
|
+
deviceId: row.id,
|
|
33064
|
+
stableId: row.stableId
|
|
33065
|
+
},
|
|
33066
|
+
meta: {
|
|
33067
|
+
parentDeviceId: id,
|
|
33068
|
+
error: err instanceof Error ? err.message : String(err)
|
|
33069
|
+
}
|
|
33070
|
+
});
|
|
33071
|
+
}
|
|
33072
|
+
}
|
|
33073
|
+
this.ctx.logger.info("device reloaded in place from persisted rows", {
|
|
33074
|
+
tags: { deviceId: id },
|
|
33075
|
+
meta: {
|
|
33076
|
+
stableId,
|
|
33077
|
+
type: meta.type
|
|
33078
|
+
}
|
|
33079
|
+
});
|
|
33080
|
+
return { deviceId: id };
|
|
32754
33081
|
}
|
|
32755
33082
|
/**
|
|
32756
33083
|
* Restore devices from persisted state. Two-pass:
|
|
@@ -32776,55 +33103,125 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32776
33103
|
* accessory-spawn flow handles via the parent's
|
|
32777
33104
|
* `getAccessoryChildren()`. Override only when the default doesn't
|
|
32778
33105
|
* fit.
|
|
33106
|
+
*
|
|
33107
|
+
* A row that fails either pass is NOT terminal (D347): it is handed
|
|
33108
|
+
* to a bounded background retry (`DeviceRestoreRetryScheduler`).
|
|
33109
|
+
* Only after the bound is exhausted is the device marked permanently
|
|
33110
|
+
* failed — logged at ERROR with `tags.deviceId` and surfaced via
|
|
33111
|
+
* `getStatus().error`.
|
|
33112
|
+
*/
|
|
33113
|
+
/**
|
|
33114
|
+
* Repair a row's PERSISTED config blob immediately before it is restored.
|
|
33115
|
+
* Default: no-op — most providers have nothing to heal.
|
|
33116
|
+
*
|
|
33117
|
+
* This exists because a restored device self-hydrates from the DB: `create()`
|
|
33118
|
+
* passes `{}` and `BaseDevice` parses the stored blob against the device
|
|
33119
|
+
* schema. A blob that lost a REQUIRED field therefore fails restore forever,
|
|
33120
|
+
* and no later pass revisits it — a hub-adopted Reolink camera whose blob had
|
|
33121
|
+
* been emptied failed all four bounded attempts against fields
|
|
33122
|
+
* (`host`, `password`) it inherits from its parent and never dials itself.
|
|
33123
|
+
*
|
|
33124
|
+
* Implementations get every saved row, so a child can read its parent's blob.
|
|
33125
|
+
* A heal that throws is treated like any other restore failure: retried under
|
|
33126
|
+
* the bound, then reported — never swallowed.
|
|
32779
33127
|
*/
|
|
33128
|
+
async healSavedConfig(_saved, _allSaved) {}
|
|
32780
33129
|
async onRestoreDevices(savedDevices) {
|
|
32781
33130
|
const restored = /* @__PURE__ */ new Set();
|
|
33131
|
+
const failures = [];
|
|
33132
|
+
const attemptRestore = async (saved) => {
|
|
33133
|
+
if (restored.has(saved.id)) return;
|
|
33134
|
+
const Class = this.deviceClasses[saved.type];
|
|
33135
|
+
if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
|
|
33136
|
+
if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
|
|
33137
|
+
await this.healSavedConfig(saved, savedDevices);
|
|
33138
|
+
await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
|
|
33139
|
+
restored.add(saved.id);
|
|
33140
|
+
};
|
|
32782
33141
|
const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
|
|
32783
33142
|
const restoreOne = async (saved) => {
|
|
32784
|
-
|
|
32785
|
-
if (!Class) {
|
|
33143
|
+
if (!this.deviceClasses[saved.type]) {
|
|
32786
33144
|
this.ctx.logger.warn("No device class registered for restored type — skipping", {
|
|
32787
|
-
tags: {
|
|
33145
|
+
tags: {
|
|
33146
|
+
deviceId: saved.id,
|
|
33147
|
+
stableId: saved.stableId
|
|
33148
|
+
},
|
|
32788
33149
|
meta: { type: saved.type }
|
|
32789
33150
|
});
|
|
32790
33151
|
return;
|
|
32791
33152
|
}
|
|
32792
33153
|
try {
|
|
32793
|
-
await
|
|
32794
|
-
restored.add(saved.id);
|
|
33154
|
+
await attemptRestore(saved);
|
|
32795
33155
|
} catch (err) {
|
|
32796
|
-
|
|
32797
|
-
|
|
33156
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
33157
|
+
this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
|
|
33158
|
+
tags: {
|
|
33159
|
+
deviceId: saved.id,
|
|
33160
|
+
stableId: saved.stableId
|
|
33161
|
+
},
|
|
32798
33162
|
meta: {
|
|
32799
33163
|
type: saved.type,
|
|
32800
|
-
|
|
33164
|
+
attempt: 1,
|
|
33165
|
+
error
|
|
32801
33166
|
}
|
|
32802
33167
|
});
|
|
33168
|
+
failures.push({
|
|
33169
|
+
saved,
|
|
33170
|
+
error
|
|
33171
|
+
});
|
|
32803
33172
|
}
|
|
32804
33173
|
};
|
|
32805
33174
|
await Promise.all(topLevel.map((saved) => restoreOne(saved)));
|
|
33175
|
+
const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
|
|
32806
33176
|
const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
|
|
32807
33177
|
for (const saved of childRows) {
|
|
32808
|
-
|
|
32809
|
-
if (!Class) continue;
|
|
33178
|
+
if (!this.deviceClasses[saved.type]) continue;
|
|
32810
33179
|
if (saved.parentDeviceId === null) continue;
|
|
32811
|
-
if (
|
|
32812
|
-
|
|
32813
|
-
|
|
32814
|
-
|
|
32815
|
-
|
|
32816
|
-
|
|
33180
|
+
if (restored.has(saved.parentDeviceId)) {
|
|
33181
|
+
try {
|
|
33182
|
+
await attemptRestore(saved);
|
|
33183
|
+
} catch (err) {
|
|
33184
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
33185
|
+
this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
|
|
33186
|
+
tags: {
|
|
33187
|
+
deviceId: saved.id,
|
|
33188
|
+
stableId: saved.stableId,
|
|
33189
|
+
parentDeviceId: saved.parentDeviceId
|
|
33190
|
+
},
|
|
33191
|
+
meta: {
|
|
33192
|
+
type: saved.type,
|
|
33193
|
+
attempt: 1,
|
|
33194
|
+
error
|
|
33195
|
+
}
|
|
33196
|
+
});
|
|
33197
|
+
failures.push({
|
|
33198
|
+
saved,
|
|
33199
|
+
error
|
|
33200
|
+
});
|
|
33201
|
+
}
|
|
33202
|
+
continue;
|
|
33203
|
+
}
|
|
33204
|
+
if (failedTopLevelIds.has(saved.parentDeviceId)) {
|
|
33205
|
+
this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
|
|
32817
33206
|
tags: {
|
|
33207
|
+
deviceId: saved.id,
|
|
32818
33208
|
stableId: saved.stableId,
|
|
32819
33209
|
parentDeviceId: saved.parentDeviceId
|
|
32820
33210
|
},
|
|
32821
|
-
meta: {
|
|
32822
|
-
type: saved.type,
|
|
32823
|
-
error: err instanceof Error ? err.message : String(err)
|
|
32824
|
-
}
|
|
33211
|
+
meta: { type: saved.type }
|
|
32825
33212
|
});
|
|
33213
|
+
failures.push({
|
|
33214
|
+
saved,
|
|
33215
|
+
error: `parent device ${saved.parentDeviceId} not restored`
|
|
33216
|
+
});
|
|
33217
|
+
continue;
|
|
32826
33218
|
}
|
|
32827
33219
|
}
|
|
33220
|
+
if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
|
|
33221
|
+
return {
|
|
33222
|
+
restoredCount: restored.size,
|
|
33223
|
+
failedCount: failures.length
|
|
33224
|
+
};
|
|
32828
33225
|
}
|
|
32829
33226
|
/** Convert an IDevice to the flat DeviceSummary for the cap router. */
|
|
32830
33227
|
toSummary(device) {
|
|
@@ -34587,6 +34984,12 @@ Object.freeze({
|
|
|
34587
34984
|
addonId: null,
|
|
34588
34985
|
access: "view"
|
|
34589
34986
|
},
|
|
34987
|
+
"deviceProvider.reloadDevice": {
|
|
34988
|
+
capName: "device-provider",
|
|
34989
|
+
capScope: "system",
|
|
34990
|
+
addonId: null,
|
|
34991
|
+
access: "create"
|
|
34992
|
+
},
|
|
34590
34993
|
"deviceProvider.start": {
|
|
34591
34994
|
capName: "device-provider",
|
|
34592
34995
|
capScope: "system",
|
package/dist/addon.mjs
CHANGED
|
@@ -12834,7 +12834,26 @@ var DiscoveryCandidateSchema = object({
|
|
|
12834
12834
|
* identity ahead of adoption. Rendering metadata (unit, precision)
|
|
12835
12835
|
* flows live through the cap STATUS SLICE after adoption.
|
|
12836
12836
|
*/
|
|
12837
|
-
sourceInfo: SourceInfoSchema.optional()
|
|
12837
|
+
sourceInfo: SourceInfoSchema.optional(),
|
|
12838
|
+
/**
|
|
12839
|
+
* Set when this candidate is a device the provider ALREADY owns.
|
|
12840
|
+
*
|
|
12841
|
+
* A scan cannot generally produce the identity a device was onboarded under
|
|
12842
|
+
* (Reolink keys on `mac-<mac>`, learned at adopt time), so a stableId
|
|
12843
|
+
* comparison never matches and an owned device looks addable. Re-adopting one
|
|
12844
|
+
* overwrites its config with scan-derived values — that is how a Home Hub's
|
|
12845
|
+
* Baichuan port was overwritten with its ONVIF port, taking the hub and its
|
|
12846
|
+
* three child cameras offline for four hours.
|
|
12847
|
+
*
|
|
12848
|
+
* A provider that can recognise its own devices says so here. Absent means
|
|
12849
|
+
* "not recognised", which is not the same as "known to be new" — a provider
|
|
12850
|
+
* that cannot tell simply never sets it.
|
|
12851
|
+
*/
|
|
12852
|
+
alreadyOnboarded: boolean().optional(),
|
|
12853
|
+
/** Numeric id of the device this candidate was matched to. Set with `alreadyOnboarded`. */
|
|
12854
|
+
onboardedDeviceId: number().optional(),
|
|
12855
|
+
/** Operator-facing name of the matched device, so the UI can say WHICH one it is. */
|
|
12856
|
+
onboardedName: string$2().optional()
|
|
12838
12857
|
});
|
|
12839
12858
|
/**
|
|
12840
12859
|
* Flat device summary returned by `createDevice` / `adoptDiscoveredDevice`.
|
|
@@ -12890,6 +12909,35 @@ var deviceProviderCapability = {
|
|
|
12890
12909
|
name: string$2(),
|
|
12891
12910
|
type: string$2()
|
|
12892
12911
|
}))),
|
|
12912
|
+
/**
|
|
12913
|
+
* Tear down and reconstruct ONE device in place from its persisted rows —
|
|
12914
|
+
* touching no other device this provider owns.
|
|
12915
|
+
*
|
|
12916
|
+
* The primitive `deviceManager.migrateDevice` uses to flush the two
|
|
12917
|
+
* migrated numbers: after `swapIds` the runner's live instance still
|
|
12918
|
+
* carries the PRE-swap numeric id (baked into the object, its native-cap
|
|
12919
|
+
* registrations and its log tags), and a live object cannot be renumbered.
|
|
12920
|
+
* Before this method the only flush was restarting the whole owning addon
|
|
12921
|
+
* — which took every camera the provider owns down with it (28 devices
|
|
12922
|
+
* for one migrated camera, measured 2026-09-04, and the morning of the
|
|
12923
|
+
* same day ~27 devices' native caps did not come back on their own).
|
|
12924
|
+
*
|
|
12925
|
+
* Keyed by `stableId`, deliberately: the numeric id is exactly the thing
|
|
12926
|
+
* that changes. The reply carries the id the device answers on NOW.
|
|
12927
|
+
* Implemented once in `BaseDeviceProvider` — decommission the live
|
|
12928
|
+
* instance (if any), then re-create from the persisted row: the same
|
|
12929
|
+
* teardown/rehydrate pair every graceful shutdown + boot already uses.
|
|
12930
|
+
* An RPC, never an event: a dropped event would leave the runner writing
|
|
12931
|
+
* against the wrong camera (D8).
|
|
12932
|
+
*
|
|
12933
|
+
* Construction can dial hardware, and the migrated source is
|
|
12934
|
+
* characteristically dead — the timeout covers a full activate window
|
|
12935
|
+
* rather than the 60 s default.
|
|
12936
|
+
*/
|
|
12937
|
+
reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
|
|
12938
|
+
kind: "mutation",
|
|
12939
|
+
timeoutMs: 3 * 6e4
|
|
12940
|
+
}),
|
|
12893
12941
|
supportsDiscovery: method(object({}), boolean()),
|
|
12894
12942
|
/**
|
|
12895
12943
|
* Run a network scan. `params` carries optional provider-specific scan
|
|
@@ -13217,7 +13265,8 @@ method(object({
|
|
|
13217
13265
|
targetId: number()
|
|
13218
13266
|
}), MigrateDeviceResultSchema, {
|
|
13219
13267
|
kind: "mutation",
|
|
13220
|
-
auth: "admin"
|
|
13268
|
+
auth: "admin",
|
|
13269
|
+
timeoutMs: 12 * 6e4
|
|
13221
13270
|
}), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
|
|
13222
13271
|
deviceId: number(),
|
|
13223
13272
|
name: string$2()
|
|
@@ -32590,6 +32639,147 @@ var BaseDevice = class {
|
|
|
32590
32639
|
}
|
|
32591
32640
|
};
|
|
32592
32641
|
/**
|
|
32642
|
+
* Delays before retry rounds 1..N — the round count IS the bound.
|
|
32643
|
+
* 10 s catches "the hub was busy for a moment"; the full schedule
|
|
32644
|
+
* (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
|
|
32645
|
+
* per attempt) covers a device-manager lock held for minutes — the
|
|
32646
|
+
* 2026-09-04 outage's migration hold was ~3.5 min.
|
|
32647
|
+
*/
|
|
32648
|
+
var DEVICE_RESTORE_RETRY_DELAYS_MS = [
|
|
32649
|
+
1e4,
|
|
32650
|
+
3e4,
|
|
32651
|
+
9e4
|
|
32652
|
+
];
|
|
32653
|
+
/** Abortable sleep — resolves early (never rejects) on abort. */
|
|
32654
|
+
function sleep$1(ms, signal) {
|
|
32655
|
+
return new Promise((resolve) => {
|
|
32656
|
+
if (signal.aborted) {
|
|
32657
|
+
resolve();
|
|
32658
|
+
return;
|
|
32659
|
+
}
|
|
32660
|
+
const onAbort = () => {
|
|
32661
|
+
clearTimeout(timer);
|
|
32662
|
+
resolve();
|
|
32663
|
+
};
|
|
32664
|
+
const timer = setTimeout(() => {
|
|
32665
|
+
signal.removeEventListener("abort", onAbort);
|
|
32666
|
+
resolve();
|
|
32667
|
+
}, ms);
|
|
32668
|
+
timer.unref?.();
|
|
32669
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
32670
|
+
});
|
|
32671
|
+
}
|
|
32672
|
+
/** Drain `items` through at most `width` concurrent lanes. `fn` must
|
|
32673
|
+
* not reject (callers wrap their own try/catch). */
|
|
32674
|
+
async function runWithConcurrency(items, width, fn) {
|
|
32675
|
+
const queue = [...items];
|
|
32676
|
+
const laneCount = Math.max(1, Math.min(width, queue.length));
|
|
32677
|
+
const lane = async () => {
|
|
32678
|
+
for (;;) {
|
|
32679
|
+
const item = queue.shift();
|
|
32680
|
+
if (item === void 0) return;
|
|
32681
|
+
await fn(item);
|
|
32682
|
+
}
|
|
32683
|
+
};
|
|
32684
|
+
await Promise.all(Array.from({ length: laneCount }, lane));
|
|
32685
|
+
}
|
|
32686
|
+
var DeviceRestoreRetryScheduler = class {
|
|
32687
|
+
#logger;
|
|
32688
|
+
#attempt;
|
|
32689
|
+
#onPermanentFailure;
|
|
32690
|
+
#delaysMs;
|
|
32691
|
+
#concurrency;
|
|
32692
|
+
#now;
|
|
32693
|
+
#abort = new AbortController();
|
|
32694
|
+
constructor(options) {
|
|
32695
|
+
this.#logger = options.logger;
|
|
32696
|
+
this.#attempt = options.attempt;
|
|
32697
|
+
this.#onPermanentFailure = options.onPermanentFailure;
|
|
32698
|
+
this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
|
|
32699
|
+
this.#concurrency = options.concurrency ?? 4;
|
|
32700
|
+
this.#now = options.now ?? Date.now;
|
|
32701
|
+
}
|
|
32702
|
+
/** Stop retrying (shutdown). Pending entries are NOT marked
|
|
32703
|
+
* permanently failed — the next boot restores them from disk. */
|
|
32704
|
+
cancel() {
|
|
32705
|
+
this.#abort.abort();
|
|
32706
|
+
}
|
|
32707
|
+
/**
|
|
32708
|
+
* Run the bounded retry rounds. Resolves when every entry has either
|
|
32709
|
+
* restored, been marked permanently failed, or the scheduler was
|
|
32710
|
+
* cancelled. Never rejects.
|
|
32711
|
+
*/
|
|
32712
|
+
async run(initialFailures) {
|
|
32713
|
+
let pending = initialFailures.map((failure) => ({
|
|
32714
|
+
saved: failure.saved,
|
|
32715
|
+
lastError: failure.error,
|
|
32716
|
+
attempts: 1
|
|
32717
|
+
}));
|
|
32718
|
+
for (let round = 0; round < this.#delaysMs.length; round += 1) {
|
|
32719
|
+
if (pending.length === 0 || this.#abort.signal.aborted) break;
|
|
32720
|
+
await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
|
|
32721
|
+
if (this.#abort.signal.aborted) break;
|
|
32722
|
+
pending = await this.#runRound(pending, round);
|
|
32723
|
+
}
|
|
32724
|
+
if (this.#abort.signal.aborted) return [];
|
|
32725
|
+
const terminal = pending.map((entry) => ({
|
|
32726
|
+
deviceId: entry.saved.id,
|
|
32727
|
+
stableId: entry.saved.stableId,
|
|
32728
|
+
type: String(entry.saved.type),
|
|
32729
|
+
attempts: entry.attempts,
|
|
32730
|
+
lastError: entry.lastError,
|
|
32731
|
+
failedAt: this.#now()
|
|
32732
|
+
}));
|
|
32733
|
+
for (const failure of terminal) this.#onPermanentFailure(failure);
|
|
32734
|
+
return terminal;
|
|
32735
|
+
}
|
|
32736
|
+
/** One retry round: parents first (phase 0), then hub-adopted
|
|
32737
|
+
* children (phase 1) — a child's attempt depends on its parent
|
|
32738
|
+
* having landed, exactly like the initial two-pass restore. */
|
|
32739
|
+
async #runRound(pending, round) {
|
|
32740
|
+
const next = [];
|
|
32741
|
+
const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
|
|
32742
|
+
const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
|
|
32743
|
+
for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
|
|
32744
|
+
if (this.#abort.signal.aborted) {
|
|
32745
|
+
next.push(entry);
|
|
32746
|
+
return;
|
|
32747
|
+
}
|
|
32748
|
+
const attemptNo = entry.attempts + 1;
|
|
32749
|
+
try {
|
|
32750
|
+
await this.#attempt(entry.saved);
|
|
32751
|
+
this.#logger.info("Device restored on retry", {
|
|
32752
|
+
tags: {
|
|
32753
|
+
deviceId: entry.saved.id,
|
|
32754
|
+
stableId: entry.saved.stableId
|
|
32755
|
+
},
|
|
32756
|
+
meta: { attempt: attemptNo }
|
|
32757
|
+
});
|
|
32758
|
+
} catch (err) {
|
|
32759
|
+
const lastError = err instanceof Error ? err.message : String(err);
|
|
32760
|
+
const remainingRetries = this.#delaysMs.length - (round + 1);
|
|
32761
|
+
this.#logger.warn("Device restore retry failed", {
|
|
32762
|
+
tags: {
|
|
32763
|
+
deviceId: entry.saved.id,
|
|
32764
|
+
stableId: entry.saved.stableId
|
|
32765
|
+
},
|
|
32766
|
+
meta: {
|
|
32767
|
+
attempt: attemptNo,
|
|
32768
|
+
remainingRetries,
|
|
32769
|
+
error: lastError
|
|
32770
|
+
}
|
|
32771
|
+
});
|
|
32772
|
+
next.push({
|
|
32773
|
+
saved: entry.saved,
|
|
32774
|
+
lastError,
|
|
32775
|
+
attempts: attemptNo
|
|
32776
|
+
});
|
|
32777
|
+
}
|
|
32778
|
+
});
|
|
32779
|
+
return next;
|
|
32780
|
+
}
|
|
32781
|
+
};
|
|
32782
|
+
/**
|
|
32593
32783
|
* Convert an IDevice to the flat DeviceSummary shape expected by the
|
|
32594
32784
|
* device-provider cap router. Shared across all providers.
|
|
32595
32785
|
*/
|
|
@@ -32638,6 +32828,7 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32638
32828
|
}];
|
|
32639
32829
|
}
|
|
32640
32830
|
async onShutdown() {
|
|
32831
|
+
this.cancelRestoreRetries();
|
|
32641
32832
|
const devices = await this.ctx.kernel.devices?.getAll() ?? [];
|
|
32642
32833
|
for (const device of devices) try {
|
|
32643
32834
|
await this.ctx.kernel.devices?.decommission(device.id);
|
|
@@ -32655,9 +32846,16 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32655
32846
|
async start() {}
|
|
32656
32847
|
async stop() {}
|
|
32657
32848
|
async getStatus() {
|
|
32849
|
+
const all = await this.ctx.kernel.devices?.getAll() ?? [];
|
|
32850
|
+
const summary = this.restoreFailureSummary();
|
|
32851
|
+
if (summary === null) return {
|
|
32852
|
+
connected: true,
|
|
32853
|
+
deviceCount: all.length
|
|
32854
|
+
};
|
|
32658
32855
|
return {
|
|
32659
32856
|
connected: true,
|
|
32660
|
-
deviceCount:
|
|
32857
|
+
deviceCount: all.length,
|
|
32858
|
+
error: summary
|
|
32661
32859
|
};
|
|
32662
32860
|
}
|
|
32663
32861
|
async getDevices() {
|
|
@@ -32747,8 +32945,137 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32747
32945
|
};
|
|
32748
32946
|
}
|
|
32749
32947
|
async restoreDevices(savedDevices) {
|
|
32750
|
-
await this.onRestoreDevices(savedDevices);
|
|
32751
|
-
if (savedDevices.length
|
|
32948
|
+
const report = await this.onRestoreDevices(savedDevices);
|
|
32949
|
+
if (savedDevices.length === 0) return;
|
|
32950
|
+
if (report && report.failedCount > 0) {
|
|
32951
|
+
this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
|
|
32952
|
+
return;
|
|
32953
|
+
}
|
|
32954
|
+
const restoredCount = report ? report.restoredCount : savedDevices.length;
|
|
32955
|
+
this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
|
|
32956
|
+
}
|
|
32957
|
+
/** Retry schedule. Overridable (tests use millisecond delays). */
|
|
32958
|
+
restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
|
|
32959
|
+
/** Retry lane width. See `device-restore-retry.ts` for why retries
|
|
32960
|
+
* never re-stampede full-width while the initial pass does (D167). */
|
|
32961
|
+
restoreRetryConcurrency = 4;
|
|
32962
|
+
_restoreRetryScheduler = null;
|
|
32963
|
+
_restoreRetryCompletion = null;
|
|
32964
|
+
_permanentRestoreFailures = /* @__PURE__ */ new Map();
|
|
32965
|
+
/** Settles when the background retry rounds finish (or `null` when
|
|
32966
|
+
* nothing failed). Exposed for tests and subclass diagnostics —
|
|
32967
|
+
* boot NEVER awaits this: the runner's post-init handshake goes out
|
|
32968
|
+
* with the devices that restored, and a late success is announced
|
|
32969
|
+
* through the `native-cap-change` → `updateCaps` path. */
|
|
32970
|
+
get restoreRetryCompletion() {
|
|
32971
|
+
return this._restoreRetryCompletion;
|
|
32972
|
+
}
|
|
32973
|
+
/** Devices that exhausted the retry bound this process lifetime. */
|
|
32974
|
+
get permanentRestoreFailures() {
|
|
32975
|
+
return [...this._permanentRestoreFailures.values()];
|
|
32976
|
+
}
|
|
32977
|
+
/** One-line operator-facing summary for `getStatus().error`, or
|
|
32978
|
+
* `null` when every device restored. */
|
|
32979
|
+
restoreFailureSummary() {
|
|
32980
|
+
if (this._permanentRestoreFailures.size === 0) return null;
|
|
32981
|
+
const ids = [...this._permanentRestoreFailures.keys()].join(", ");
|
|
32982
|
+
return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
|
|
32983
|
+
}
|
|
32984
|
+
cancelRestoreRetries() {
|
|
32985
|
+
this._restoreRetryScheduler?.cancel();
|
|
32986
|
+
this._restoreRetryScheduler = null;
|
|
32987
|
+
}
|
|
32988
|
+
recordPermanentRestoreFailure(failure) {
|
|
32989
|
+
this._permanentRestoreFailures.set(failure.deviceId, failure);
|
|
32990
|
+
this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
|
|
32991
|
+
tags: {
|
|
32992
|
+
deviceId: failure.deviceId,
|
|
32993
|
+
stableId: failure.stableId
|
|
32994
|
+
},
|
|
32995
|
+
meta: {
|
|
32996
|
+
type: failure.type,
|
|
32997
|
+
attempts: failure.attempts,
|
|
32998
|
+
error: failure.lastError
|
|
32999
|
+
}
|
|
33000
|
+
});
|
|
33001
|
+
}
|
|
33002
|
+
scheduleRestoreRetries(failures, attempt) {
|
|
33003
|
+
const scheduler = new DeviceRestoreRetryScheduler({
|
|
33004
|
+
logger: this.ctx.logger,
|
|
33005
|
+
delaysMs: this.restoreRetryDelaysMs,
|
|
33006
|
+
concurrency: this.restoreRetryConcurrency,
|
|
33007
|
+
attempt,
|
|
33008
|
+
onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
|
|
33009
|
+
});
|
|
33010
|
+
this._restoreRetryScheduler = scheduler;
|
|
33011
|
+
this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
|
|
33012
|
+
this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
|
|
33013
|
+
});
|
|
33014
|
+
}
|
|
33015
|
+
/**
|
|
33016
|
+
* Tear down and reconstruct ONE device from its persisted rows — the
|
|
33017
|
+
* `deviceProvider.reloadDevice` cap method. Persistence is never touched,
|
|
33018
|
+
* and no other device this provider owns is disturbed.
|
|
33019
|
+
*
|
|
33020
|
+
* Keyed by `stableId` because the caller's whole reason to be here is that
|
|
33021
|
+
* the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
|
|
33022
|
+
* fresh instance resolves its id through `allocateDeviceId`, which returns
|
|
33023
|
+
* whatever number the row carries NOW. The teardown is `decommission` —
|
|
33024
|
+
* exactly what a graceful shutdown runs per device (fires `removeDevice()`,
|
|
33025
|
+
* unregisters native caps, drops the registry entry) — and the rebuild is
|
|
33026
|
+
* the boot restore's own `create()` path, including its pass 2: first-class
|
|
33027
|
+
* children (hub-adopted cameras under an NVR) are decommissioned with the
|
|
33028
|
+
* parent by the cascade and must be re-created explicitly, because only
|
|
33029
|
+
* accessory children come back through `getAccessoryChildren()`.
|
|
33030
|
+
*
|
|
33031
|
+
* Reloading an accessory child directly is refused (no device class) —
|
|
33032
|
+
* reload its parent instead.
|
|
33033
|
+
*/
|
|
33034
|
+
async reloadDevice(input) {
|
|
33035
|
+
const { stableId } = input;
|
|
33036
|
+
const devices = this.ctx.kernel.devices;
|
|
33037
|
+
if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
|
|
33038
|
+
const live = (await devices.getAll()).find((d) => d.stableId === stableId);
|
|
33039
|
+
if (live) await devices.decommission(live.id);
|
|
33040
|
+
const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
|
|
33041
|
+
addonId: this.addonId,
|
|
33042
|
+
stableId
|
|
33043
|
+
});
|
|
33044
|
+
const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
|
|
33045
|
+
if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
|
|
33046
|
+
const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
|
|
33047
|
+
const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
|
|
33048
|
+
if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
|
|
33049
|
+
await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
|
|
33050
|
+
const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
|
|
33051
|
+
for (const row of rows) {
|
|
33052
|
+
if (row.parentDeviceId !== id) continue;
|
|
33053
|
+
const childType = Object.values(DeviceType).find((t) => t === row.type);
|
|
33054
|
+
const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
|
|
33055
|
+
if (!ChildClass) continue;
|
|
33056
|
+
try {
|
|
33057
|
+
await devices.create(row.stableId, ChildClass, {}, id);
|
|
33058
|
+
} catch (err) {
|
|
33059
|
+
this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
|
|
33060
|
+
tags: {
|
|
33061
|
+
deviceId: row.id,
|
|
33062
|
+
stableId: row.stableId
|
|
33063
|
+
},
|
|
33064
|
+
meta: {
|
|
33065
|
+
parentDeviceId: id,
|
|
33066
|
+
error: err instanceof Error ? err.message : String(err)
|
|
33067
|
+
}
|
|
33068
|
+
});
|
|
33069
|
+
}
|
|
33070
|
+
}
|
|
33071
|
+
this.ctx.logger.info("device reloaded in place from persisted rows", {
|
|
33072
|
+
tags: { deviceId: id },
|
|
33073
|
+
meta: {
|
|
33074
|
+
stableId,
|
|
33075
|
+
type: meta.type
|
|
33076
|
+
}
|
|
33077
|
+
});
|
|
33078
|
+
return { deviceId: id };
|
|
32752
33079
|
}
|
|
32753
33080
|
/**
|
|
32754
33081
|
* Restore devices from persisted state. Two-pass:
|
|
@@ -32774,55 +33101,125 @@ var BaseDeviceProvider = class extends BaseAddon {
|
|
|
32774
33101
|
* accessory-spawn flow handles via the parent's
|
|
32775
33102
|
* `getAccessoryChildren()`. Override only when the default doesn't
|
|
32776
33103
|
* fit.
|
|
33104
|
+
*
|
|
33105
|
+
* A row that fails either pass is NOT terminal (D347): it is handed
|
|
33106
|
+
* to a bounded background retry (`DeviceRestoreRetryScheduler`).
|
|
33107
|
+
* Only after the bound is exhausted is the device marked permanently
|
|
33108
|
+
* failed — logged at ERROR with `tags.deviceId` and surfaced via
|
|
33109
|
+
* `getStatus().error`.
|
|
33110
|
+
*/
|
|
33111
|
+
/**
|
|
33112
|
+
* Repair a row's PERSISTED config blob immediately before it is restored.
|
|
33113
|
+
* Default: no-op — most providers have nothing to heal.
|
|
33114
|
+
*
|
|
33115
|
+
* This exists because a restored device self-hydrates from the DB: `create()`
|
|
33116
|
+
* passes `{}` and `BaseDevice` parses the stored blob against the device
|
|
33117
|
+
* schema. A blob that lost a REQUIRED field therefore fails restore forever,
|
|
33118
|
+
* and no later pass revisits it — a hub-adopted Reolink camera whose blob had
|
|
33119
|
+
* been emptied failed all four bounded attempts against fields
|
|
33120
|
+
* (`host`, `password`) it inherits from its parent and never dials itself.
|
|
33121
|
+
*
|
|
33122
|
+
* Implementations get every saved row, so a child can read its parent's blob.
|
|
33123
|
+
* A heal that throws is treated like any other restore failure: retried under
|
|
33124
|
+
* the bound, then reported — never swallowed.
|
|
32777
33125
|
*/
|
|
33126
|
+
async healSavedConfig(_saved, _allSaved) {}
|
|
32778
33127
|
async onRestoreDevices(savedDevices) {
|
|
32779
33128
|
const restored = /* @__PURE__ */ new Set();
|
|
33129
|
+
const failures = [];
|
|
33130
|
+
const attemptRestore = async (saved) => {
|
|
33131
|
+
if (restored.has(saved.id)) return;
|
|
33132
|
+
const Class = this.deviceClasses[saved.type];
|
|
33133
|
+
if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
|
|
33134
|
+
if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
|
|
33135
|
+
await this.healSavedConfig(saved, savedDevices);
|
|
33136
|
+
await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
|
|
33137
|
+
restored.add(saved.id);
|
|
33138
|
+
};
|
|
32780
33139
|
const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
|
|
32781
33140
|
const restoreOne = async (saved) => {
|
|
32782
|
-
|
|
32783
|
-
if (!Class) {
|
|
33141
|
+
if (!this.deviceClasses[saved.type]) {
|
|
32784
33142
|
this.ctx.logger.warn("No device class registered for restored type — skipping", {
|
|
32785
|
-
tags: {
|
|
33143
|
+
tags: {
|
|
33144
|
+
deviceId: saved.id,
|
|
33145
|
+
stableId: saved.stableId
|
|
33146
|
+
},
|
|
32786
33147
|
meta: { type: saved.type }
|
|
32787
33148
|
});
|
|
32788
33149
|
return;
|
|
32789
33150
|
}
|
|
32790
33151
|
try {
|
|
32791
|
-
await
|
|
32792
|
-
restored.add(saved.id);
|
|
33152
|
+
await attemptRestore(saved);
|
|
32793
33153
|
} catch (err) {
|
|
32794
|
-
|
|
32795
|
-
|
|
33154
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
33155
|
+
this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
|
|
33156
|
+
tags: {
|
|
33157
|
+
deviceId: saved.id,
|
|
33158
|
+
stableId: saved.stableId
|
|
33159
|
+
},
|
|
32796
33160
|
meta: {
|
|
32797
33161
|
type: saved.type,
|
|
32798
|
-
|
|
33162
|
+
attempt: 1,
|
|
33163
|
+
error
|
|
32799
33164
|
}
|
|
32800
33165
|
});
|
|
33166
|
+
failures.push({
|
|
33167
|
+
saved,
|
|
33168
|
+
error
|
|
33169
|
+
});
|
|
32801
33170
|
}
|
|
32802
33171
|
};
|
|
32803
33172
|
await Promise.all(topLevel.map((saved) => restoreOne(saved)));
|
|
33173
|
+
const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
|
|
32804
33174
|
const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
|
|
32805
33175
|
for (const saved of childRows) {
|
|
32806
|
-
|
|
32807
|
-
if (!Class) continue;
|
|
33176
|
+
if (!this.deviceClasses[saved.type]) continue;
|
|
32808
33177
|
if (saved.parentDeviceId === null) continue;
|
|
32809
|
-
if (
|
|
32810
|
-
|
|
32811
|
-
|
|
32812
|
-
|
|
32813
|
-
|
|
32814
|
-
|
|
33178
|
+
if (restored.has(saved.parentDeviceId)) {
|
|
33179
|
+
try {
|
|
33180
|
+
await attemptRestore(saved);
|
|
33181
|
+
} catch (err) {
|
|
33182
|
+
const error = err instanceof Error ? err.message : String(err);
|
|
33183
|
+
this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
|
|
33184
|
+
tags: {
|
|
33185
|
+
deviceId: saved.id,
|
|
33186
|
+
stableId: saved.stableId,
|
|
33187
|
+
parentDeviceId: saved.parentDeviceId
|
|
33188
|
+
},
|
|
33189
|
+
meta: {
|
|
33190
|
+
type: saved.type,
|
|
33191
|
+
attempt: 1,
|
|
33192
|
+
error
|
|
33193
|
+
}
|
|
33194
|
+
});
|
|
33195
|
+
failures.push({
|
|
33196
|
+
saved,
|
|
33197
|
+
error
|
|
33198
|
+
});
|
|
33199
|
+
}
|
|
33200
|
+
continue;
|
|
33201
|
+
}
|
|
33202
|
+
if (failedTopLevelIds.has(saved.parentDeviceId)) {
|
|
33203
|
+
this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
|
|
32815
33204
|
tags: {
|
|
33205
|
+
deviceId: saved.id,
|
|
32816
33206
|
stableId: saved.stableId,
|
|
32817
33207
|
parentDeviceId: saved.parentDeviceId
|
|
32818
33208
|
},
|
|
32819
|
-
meta: {
|
|
32820
|
-
type: saved.type,
|
|
32821
|
-
error: err instanceof Error ? err.message : String(err)
|
|
32822
|
-
}
|
|
33209
|
+
meta: { type: saved.type }
|
|
32823
33210
|
});
|
|
33211
|
+
failures.push({
|
|
33212
|
+
saved,
|
|
33213
|
+
error: `parent device ${saved.parentDeviceId} not restored`
|
|
33214
|
+
});
|
|
33215
|
+
continue;
|
|
32824
33216
|
}
|
|
32825
33217
|
}
|
|
33218
|
+
if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
|
|
33219
|
+
return {
|
|
33220
|
+
restoredCount: restored.size,
|
|
33221
|
+
failedCount: failures.length
|
|
33222
|
+
};
|
|
32826
33223
|
}
|
|
32827
33224
|
/** Convert an IDevice to the flat DeviceSummary for the cap router. */
|
|
32828
33225
|
toSummary(device) {
|
|
@@ -34585,6 +34982,12 @@ Object.freeze({
|
|
|
34585
34982
|
addonId: null,
|
|
34586
34983
|
access: "view"
|
|
34587
34984
|
},
|
|
34985
|
+
"deviceProvider.reloadDevice": {
|
|
34986
|
+
capName: "device-provider",
|
|
34987
|
+
capScope: "system",
|
|
34988
|
+
addonId: null,
|
|
34989
|
+
access: "create"
|
|
34990
|
+
},
|
|
34588
34991
|
"deviceProvider.start": {
|
|
34589
34992
|
capName: "device-provider",
|
|
34590
34993
|
capScope: "system",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@camstack/addon-matter-broker",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.60",
|
|
4
4
|
"description": "Matter broker addon for CamStack — owns a Matter fabric (commissioning + the long-lived controller) via the matter.js controller and brokers commissioned Matter nodes into CamStack",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"camstack",
|