@camstack/addon-matter-broker 0.2.58 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +391 -24
  2. package/dist/addon.mjs +391 -24
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -12892,6 +12892,35 @@ var deviceProviderCapability = {
12892
12892
  name: string$2(),
12893
12893
  type: string$2()
12894
12894
  }))),
12895
+ /**
12896
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12897
+ * touching no other device this provider owns.
12898
+ *
12899
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12900
+ * migrated numbers: after `swapIds` the runner's live instance still
12901
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12902
+ * registrations and its log tags), and a live object cannot be renumbered.
12903
+ * Before this method the only flush was restarting the whole owning addon
12904
+ * — which took every camera the provider owns down with it (28 devices
12905
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12906
+ * same day ~27 devices' native caps did not come back on their own).
12907
+ *
12908
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12909
+ * that changes. The reply carries the id the device answers on NOW.
12910
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12911
+ * instance (if any), then re-create from the persisted row: the same
12912
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12913
+ * An RPC, never an event: a dropped event would leave the runner writing
12914
+ * against the wrong camera (D8).
12915
+ *
12916
+ * Construction can dial hardware, and the migrated source is
12917
+ * characteristically dead — the timeout covers a full activate window
12918
+ * rather than the 60 s default.
12919
+ */
12920
+ reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
12921
+ kind: "mutation",
12922
+ timeoutMs: 3 * 6e4
12923
+ }),
12895
12924
  supportsDiscovery: method(object({}), boolean()),
12896
12925
  /**
12897
12926
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13219,7 +13248,8 @@ method(object({
13219
13248
  targetId: number()
13220
13249
  }), MigrateDeviceResultSchema, {
13221
13250
  kind: "mutation",
13222
- auth: "admin"
13251
+ auth: "admin",
13252
+ timeoutMs: 12 * 6e4
13223
13253
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13224
13254
  deviceId: number(),
13225
13255
  name: string$2()
@@ -32592,6 +32622,147 @@ var BaseDevice = class {
32592
32622
  }
32593
32623
  };
32594
32624
  /**
32625
+ * Delays before retry rounds 1..N — the round count IS the bound.
32626
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32627
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32628
+ * per attempt) covers a device-manager lock held for minutes — the
32629
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32630
+ */
32631
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32632
+ 1e4,
32633
+ 3e4,
32634
+ 9e4
32635
+ ];
32636
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32637
+ function sleep$1(ms, signal) {
32638
+ return new Promise((resolve) => {
32639
+ if (signal.aborted) {
32640
+ resolve();
32641
+ return;
32642
+ }
32643
+ const onAbort = () => {
32644
+ clearTimeout(timer);
32645
+ resolve();
32646
+ };
32647
+ const timer = setTimeout(() => {
32648
+ signal.removeEventListener("abort", onAbort);
32649
+ resolve();
32650
+ }, ms);
32651
+ timer.unref?.();
32652
+ signal.addEventListener("abort", onAbort, { once: true });
32653
+ });
32654
+ }
32655
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32656
+ * not reject (callers wrap their own try/catch). */
32657
+ async function runWithConcurrency(items, width, fn) {
32658
+ const queue = [...items];
32659
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32660
+ const lane = async () => {
32661
+ for (;;) {
32662
+ const item = queue.shift();
32663
+ if (item === void 0) return;
32664
+ await fn(item);
32665
+ }
32666
+ };
32667
+ await Promise.all(Array.from({ length: laneCount }, lane));
32668
+ }
32669
+ var DeviceRestoreRetryScheduler = class {
32670
+ #logger;
32671
+ #attempt;
32672
+ #onPermanentFailure;
32673
+ #delaysMs;
32674
+ #concurrency;
32675
+ #now;
32676
+ #abort = new AbortController();
32677
+ constructor(options) {
32678
+ this.#logger = options.logger;
32679
+ this.#attempt = options.attempt;
32680
+ this.#onPermanentFailure = options.onPermanentFailure;
32681
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32682
+ this.#concurrency = options.concurrency ?? 4;
32683
+ this.#now = options.now ?? Date.now;
32684
+ }
32685
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32686
+ * permanently failed — the next boot restores them from disk. */
32687
+ cancel() {
32688
+ this.#abort.abort();
32689
+ }
32690
+ /**
32691
+ * Run the bounded retry rounds. Resolves when every entry has either
32692
+ * restored, been marked permanently failed, or the scheduler was
32693
+ * cancelled. Never rejects.
32694
+ */
32695
+ async run(initialFailures) {
32696
+ let pending = initialFailures.map((failure) => ({
32697
+ saved: failure.saved,
32698
+ lastError: failure.error,
32699
+ attempts: 1
32700
+ }));
32701
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32702
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32703
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32704
+ if (this.#abort.signal.aborted) break;
32705
+ pending = await this.#runRound(pending, round);
32706
+ }
32707
+ if (this.#abort.signal.aborted) return [];
32708
+ const terminal = pending.map((entry) => ({
32709
+ deviceId: entry.saved.id,
32710
+ stableId: entry.saved.stableId,
32711
+ type: String(entry.saved.type),
32712
+ attempts: entry.attempts,
32713
+ lastError: entry.lastError,
32714
+ failedAt: this.#now()
32715
+ }));
32716
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32717
+ return terminal;
32718
+ }
32719
+ /** One retry round: parents first (phase 0), then hub-adopted
32720
+ * children (phase 1) — a child's attempt depends on its parent
32721
+ * having landed, exactly like the initial two-pass restore. */
32722
+ async #runRound(pending, round) {
32723
+ const next = [];
32724
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32725
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32726
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32727
+ if (this.#abort.signal.aborted) {
32728
+ next.push(entry);
32729
+ return;
32730
+ }
32731
+ const attemptNo = entry.attempts + 1;
32732
+ try {
32733
+ await this.#attempt(entry.saved);
32734
+ this.#logger.info("Device restored on retry", {
32735
+ tags: {
32736
+ deviceId: entry.saved.id,
32737
+ stableId: entry.saved.stableId
32738
+ },
32739
+ meta: { attempt: attemptNo }
32740
+ });
32741
+ } catch (err) {
32742
+ const lastError = err instanceof Error ? err.message : String(err);
32743
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32744
+ this.#logger.warn("Device restore retry failed", {
32745
+ tags: {
32746
+ deviceId: entry.saved.id,
32747
+ stableId: entry.saved.stableId
32748
+ },
32749
+ meta: {
32750
+ attempt: attemptNo,
32751
+ remainingRetries,
32752
+ error: lastError
32753
+ }
32754
+ });
32755
+ next.push({
32756
+ saved: entry.saved,
32757
+ lastError,
32758
+ attempts: attemptNo
32759
+ });
32760
+ }
32761
+ });
32762
+ return next;
32763
+ }
32764
+ };
32765
+ /**
32595
32766
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32596
32767
  * device-provider cap router. Shared across all providers.
32597
32768
  */
@@ -32640,6 +32811,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32640
32811
  }];
32641
32812
  }
32642
32813
  async onShutdown() {
32814
+ this.cancelRestoreRetries();
32643
32815
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32644
32816
  for (const device of devices) try {
32645
32817
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32657,9 +32829,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32657
32829
  async start() {}
32658
32830
  async stop() {}
32659
32831
  async getStatus() {
32832
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32833
+ const summary = this.restoreFailureSummary();
32834
+ if (summary === null) return {
32835
+ connected: true,
32836
+ deviceCount: all.length
32837
+ };
32660
32838
  return {
32661
32839
  connected: true,
32662
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32840
+ deviceCount: all.length,
32841
+ error: summary
32663
32842
  };
32664
32843
  }
32665
32844
  async getDevices() {
@@ -32749,8 +32928,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32749
32928
  };
32750
32929
  }
32751
32930
  async restoreDevices(savedDevices) {
32752
- await this.onRestoreDevices(savedDevices);
32753
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32931
+ const report = await this.onRestoreDevices(savedDevices);
32932
+ if (savedDevices.length === 0) return;
32933
+ if (report && report.failedCount > 0) {
32934
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32935
+ return;
32936
+ }
32937
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32938
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32939
+ }
32940
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32941
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32942
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32943
+ * never re-stampede full-width while the initial pass does (D167). */
32944
+ restoreRetryConcurrency = 4;
32945
+ _restoreRetryScheduler = null;
32946
+ _restoreRetryCompletion = null;
32947
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32948
+ /** Settles when the background retry rounds finish (or `null` when
32949
+ * nothing failed). Exposed for tests and subclass diagnostics —
32950
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32951
+ * with the devices that restored, and a late success is announced
32952
+ * through the `native-cap-change` → `updateCaps` path. */
32953
+ get restoreRetryCompletion() {
32954
+ return this._restoreRetryCompletion;
32955
+ }
32956
+ /** Devices that exhausted the retry bound this process lifetime. */
32957
+ get permanentRestoreFailures() {
32958
+ return [...this._permanentRestoreFailures.values()];
32959
+ }
32960
+ /** One-line operator-facing summary for `getStatus().error`, or
32961
+ * `null` when every device restored. */
32962
+ restoreFailureSummary() {
32963
+ if (this._permanentRestoreFailures.size === 0) return null;
32964
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32965
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32966
+ }
32967
+ cancelRestoreRetries() {
32968
+ this._restoreRetryScheduler?.cancel();
32969
+ this._restoreRetryScheduler = null;
32970
+ }
32971
+ recordPermanentRestoreFailure(failure) {
32972
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32973
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32974
+ tags: {
32975
+ deviceId: failure.deviceId,
32976
+ stableId: failure.stableId
32977
+ },
32978
+ meta: {
32979
+ type: failure.type,
32980
+ attempts: failure.attempts,
32981
+ error: failure.lastError
32982
+ }
32983
+ });
32984
+ }
32985
+ scheduleRestoreRetries(failures, attempt) {
32986
+ const scheduler = new DeviceRestoreRetryScheduler({
32987
+ logger: this.ctx.logger,
32988
+ delaysMs: this.restoreRetryDelaysMs,
32989
+ concurrency: this.restoreRetryConcurrency,
32990
+ attempt,
32991
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32992
+ });
32993
+ this._restoreRetryScheduler = scheduler;
32994
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32995
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32996
+ });
32997
+ }
32998
+ /**
32999
+ * Tear down and reconstruct ONE device from its persisted rows — the
33000
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33001
+ * and no other device this provider owns is disturbed.
33002
+ *
33003
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33004
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33005
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33006
+ * whatever number the row carries NOW. The teardown is `decommission` —
33007
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33008
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33009
+ * the boot restore's own `create()` path, including its pass 2: first-class
33010
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33011
+ * parent by the cascade and must be re-created explicitly, because only
33012
+ * accessory children come back through `getAccessoryChildren()`.
33013
+ *
33014
+ * Reloading an accessory child directly is refused (no device class) —
33015
+ * reload its parent instead.
33016
+ */
33017
+ async reloadDevice(input) {
33018
+ const { stableId } = input;
33019
+ const devices = this.ctx.kernel.devices;
33020
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33021
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33022
+ if (live) await devices.decommission(live.id);
33023
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33024
+ addonId: this.addonId,
33025
+ stableId
33026
+ });
33027
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33028
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33029
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33030
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33031
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33032
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33033
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33034
+ for (const row of rows) {
33035
+ if (row.parentDeviceId !== id) continue;
33036
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33037
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33038
+ if (!ChildClass) continue;
33039
+ try {
33040
+ await devices.create(row.stableId, ChildClass, {}, id);
33041
+ } catch (err) {
33042
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33043
+ tags: {
33044
+ deviceId: row.id,
33045
+ stableId: row.stableId
33046
+ },
33047
+ meta: {
33048
+ parentDeviceId: id,
33049
+ error: err instanceof Error ? err.message : String(err)
33050
+ }
33051
+ });
33052
+ }
33053
+ }
33054
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33055
+ tags: { deviceId: id },
33056
+ meta: {
33057
+ stableId,
33058
+ type: meta.type
33059
+ }
33060
+ });
33061
+ return { deviceId: id };
32754
33062
  }
32755
33063
  /**
32756
33064
  * Restore devices from persisted state. Two-pass:
@@ -32776,55 +33084,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32776
33084
  * accessory-spawn flow handles via the parent's
32777
33085
  * `getAccessoryChildren()`. Override only when the default doesn't
32778
33086
  * fit.
33087
+ *
33088
+ * A row that fails either pass is NOT terminal (D347): it is handed
33089
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33090
+ * Only after the bound is exhausted is the device marked permanently
33091
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33092
+ * `getStatus().error`.
32779
33093
  */
32780
33094
  async onRestoreDevices(savedDevices) {
32781
33095
  const restored = /* @__PURE__ */ new Set();
33096
+ const failures = [];
33097
+ const attemptRestore = async (saved) => {
33098
+ if (restored.has(saved.id)) return;
33099
+ const Class = this.deviceClasses[saved.type];
33100
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33101
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33102
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33103
+ restored.add(saved.id);
33104
+ };
32782
33105
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32783
33106
  const restoreOne = async (saved) => {
32784
- const Class = this.deviceClasses[saved.type];
32785
- if (!Class) {
33107
+ if (!this.deviceClasses[saved.type]) {
32786
33108
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32787
- tags: { stableId: saved.stableId },
33109
+ tags: {
33110
+ deviceId: saved.id,
33111
+ stableId: saved.stableId
33112
+ },
32788
33113
  meta: { type: saved.type }
32789
33114
  });
32790
33115
  return;
32791
33116
  }
32792
33117
  try {
32793
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32794
- restored.add(saved.id);
33118
+ await attemptRestore(saved);
32795
33119
  } catch (err) {
32796
- this.ctx.logger.warn("Failed to restore device", {
32797
- tags: { stableId: saved.stableId },
33120
+ const error = err instanceof Error ? err.message : String(err);
33121
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33122
+ tags: {
33123
+ deviceId: saved.id,
33124
+ stableId: saved.stableId
33125
+ },
32798
33126
  meta: {
32799
33127
  type: saved.type,
32800
- error: err instanceof Error ? err.message : String(err)
33128
+ attempt: 1,
33129
+ error
32801
33130
  }
32802
33131
  });
33132
+ failures.push({
33133
+ saved,
33134
+ error
33135
+ });
32803
33136
  }
32804
33137
  };
32805
33138
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33139
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32806
33140
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32807
33141
  for (const saved of childRows) {
32808
- const Class = this.deviceClasses[saved.type];
32809
- if (!Class) continue;
33142
+ if (!this.deviceClasses[saved.type]) continue;
32810
33143
  if (saved.parentDeviceId === null) continue;
32811
- if (!restored.has(saved.parentDeviceId)) continue;
32812
- try {
32813
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32814
- restored.add(saved.id);
32815
- } catch (err) {
32816
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33144
+ if (restored.has(saved.parentDeviceId)) {
33145
+ try {
33146
+ await attemptRestore(saved);
33147
+ } catch (err) {
33148
+ const error = err instanceof Error ? err.message : String(err);
33149
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33150
+ tags: {
33151
+ deviceId: saved.id,
33152
+ stableId: saved.stableId,
33153
+ parentDeviceId: saved.parentDeviceId
33154
+ },
33155
+ meta: {
33156
+ type: saved.type,
33157
+ attempt: 1,
33158
+ error
33159
+ }
33160
+ });
33161
+ failures.push({
33162
+ saved,
33163
+ error
33164
+ });
33165
+ }
33166
+ continue;
33167
+ }
33168
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33169
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32817
33170
  tags: {
33171
+ deviceId: saved.id,
32818
33172
  stableId: saved.stableId,
32819
33173
  parentDeviceId: saved.parentDeviceId
32820
33174
  },
32821
- meta: {
32822
- type: saved.type,
32823
- error: err instanceof Error ? err.message : String(err)
32824
- }
33175
+ meta: { type: saved.type }
33176
+ });
33177
+ failures.push({
33178
+ saved,
33179
+ error: `parent device ${saved.parentDeviceId} not restored`
32825
33180
  });
33181
+ continue;
32826
33182
  }
32827
33183
  }
33184
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33185
+ return {
33186
+ restoredCount: restored.size,
33187
+ failedCount: failures.length
33188
+ };
32828
33189
  }
32829
33190
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32830
33191
  toSummary(device) {
@@ -34587,6 +34948,12 @@ Object.freeze({
34587
34948
  addonId: null,
34588
34949
  access: "view"
34589
34950
  },
34951
+ "deviceProvider.reloadDevice": {
34952
+ capName: "device-provider",
34953
+ capScope: "system",
34954
+ addonId: null,
34955
+ access: "create"
34956
+ },
34590
34957
  "deviceProvider.start": {
34591
34958
  capName: "device-provider",
34592
34959
  capScope: "system",
package/dist/addon.mjs CHANGED
@@ -12890,6 +12890,35 @@ var deviceProviderCapability = {
12890
12890
  name: string$2(),
12891
12891
  type: string$2()
12892
12892
  }))),
12893
+ /**
12894
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12895
+ * touching no other device this provider owns.
12896
+ *
12897
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12898
+ * migrated numbers: after `swapIds` the runner's live instance still
12899
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12900
+ * registrations and its log tags), and a live object cannot be renumbered.
12901
+ * Before this method the only flush was restarting the whole owning addon
12902
+ * — which took every camera the provider owns down with it (28 devices
12903
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12904
+ * same day ~27 devices' native caps did not come back on their own).
12905
+ *
12906
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12907
+ * that changes. The reply carries the id the device answers on NOW.
12908
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12909
+ * instance (if any), then re-create from the persisted row: the same
12910
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12911
+ * An RPC, never an event: a dropped event would leave the runner writing
12912
+ * against the wrong camera (D8).
12913
+ *
12914
+ * Construction can dial hardware, and the migrated source is
12915
+ * characteristically dead — the timeout covers a full activate window
12916
+ * rather than the 60 s default.
12917
+ */
12918
+ reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
12919
+ kind: "mutation",
12920
+ timeoutMs: 3 * 6e4
12921
+ }),
12893
12922
  supportsDiscovery: method(object({}), boolean()),
12894
12923
  /**
12895
12924
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13217,7 +13246,8 @@ method(object({
13217
13246
  targetId: number()
13218
13247
  }), MigrateDeviceResultSchema, {
13219
13248
  kind: "mutation",
13220
- auth: "admin"
13249
+ auth: "admin",
13250
+ timeoutMs: 12 * 6e4
13221
13251
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13222
13252
  deviceId: number(),
13223
13253
  name: string$2()
@@ -32590,6 +32620,147 @@ var BaseDevice = class {
32590
32620
  }
32591
32621
  };
32592
32622
  /**
32623
+ * Delays before retry rounds 1..N — the round count IS the bound.
32624
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32625
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32626
+ * per attempt) covers a device-manager lock held for minutes — the
32627
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32628
+ */
32629
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32630
+ 1e4,
32631
+ 3e4,
32632
+ 9e4
32633
+ ];
32634
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32635
+ function sleep$1(ms, signal) {
32636
+ return new Promise((resolve) => {
32637
+ if (signal.aborted) {
32638
+ resolve();
32639
+ return;
32640
+ }
32641
+ const onAbort = () => {
32642
+ clearTimeout(timer);
32643
+ resolve();
32644
+ };
32645
+ const timer = setTimeout(() => {
32646
+ signal.removeEventListener("abort", onAbort);
32647
+ resolve();
32648
+ }, ms);
32649
+ timer.unref?.();
32650
+ signal.addEventListener("abort", onAbort, { once: true });
32651
+ });
32652
+ }
32653
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32654
+ * not reject (callers wrap their own try/catch). */
32655
+ async function runWithConcurrency(items, width, fn) {
32656
+ const queue = [...items];
32657
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32658
+ const lane = async () => {
32659
+ for (;;) {
32660
+ const item = queue.shift();
32661
+ if (item === void 0) return;
32662
+ await fn(item);
32663
+ }
32664
+ };
32665
+ await Promise.all(Array.from({ length: laneCount }, lane));
32666
+ }
32667
+ var DeviceRestoreRetryScheduler = class {
32668
+ #logger;
32669
+ #attempt;
32670
+ #onPermanentFailure;
32671
+ #delaysMs;
32672
+ #concurrency;
32673
+ #now;
32674
+ #abort = new AbortController();
32675
+ constructor(options) {
32676
+ this.#logger = options.logger;
32677
+ this.#attempt = options.attempt;
32678
+ this.#onPermanentFailure = options.onPermanentFailure;
32679
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32680
+ this.#concurrency = options.concurrency ?? 4;
32681
+ this.#now = options.now ?? Date.now;
32682
+ }
32683
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32684
+ * permanently failed — the next boot restores them from disk. */
32685
+ cancel() {
32686
+ this.#abort.abort();
32687
+ }
32688
+ /**
32689
+ * Run the bounded retry rounds. Resolves when every entry has either
32690
+ * restored, been marked permanently failed, or the scheduler was
32691
+ * cancelled. Never rejects.
32692
+ */
32693
+ async run(initialFailures) {
32694
+ let pending = initialFailures.map((failure) => ({
32695
+ saved: failure.saved,
32696
+ lastError: failure.error,
32697
+ attempts: 1
32698
+ }));
32699
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32700
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32701
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32702
+ if (this.#abort.signal.aborted) break;
32703
+ pending = await this.#runRound(pending, round);
32704
+ }
32705
+ if (this.#abort.signal.aborted) return [];
32706
+ const terminal = pending.map((entry) => ({
32707
+ deviceId: entry.saved.id,
32708
+ stableId: entry.saved.stableId,
32709
+ type: String(entry.saved.type),
32710
+ attempts: entry.attempts,
32711
+ lastError: entry.lastError,
32712
+ failedAt: this.#now()
32713
+ }));
32714
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32715
+ return terminal;
32716
+ }
32717
+ /** One retry round: parents first (phase 0), then hub-adopted
32718
+ * children (phase 1) — a child's attempt depends on its parent
32719
+ * having landed, exactly like the initial two-pass restore. */
32720
+ async #runRound(pending, round) {
32721
+ const next = [];
32722
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32723
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32724
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32725
+ if (this.#abort.signal.aborted) {
32726
+ next.push(entry);
32727
+ return;
32728
+ }
32729
+ const attemptNo = entry.attempts + 1;
32730
+ try {
32731
+ await this.#attempt(entry.saved);
32732
+ this.#logger.info("Device restored on retry", {
32733
+ tags: {
32734
+ deviceId: entry.saved.id,
32735
+ stableId: entry.saved.stableId
32736
+ },
32737
+ meta: { attempt: attemptNo }
32738
+ });
32739
+ } catch (err) {
32740
+ const lastError = err instanceof Error ? err.message : String(err);
32741
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32742
+ this.#logger.warn("Device restore retry failed", {
32743
+ tags: {
32744
+ deviceId: entry.saved.id,
32745
+ stableId: entry.saved.stableId
32746
+ },
32747
+ meta: {
32748
+ attempt: attemptNo,
32749
+ remainingRetries,
32750
+ error: lastError
32751
+ }
32752
+ });
32753
+ next.push({
32754
+ saved: entry.saved,
32755
+ lastError,
32756
+ attempts: attemptNo
32757
+ });
32758
+ }
32759
+ });
32760
+ return next;
32761
+ }
32762
+ };
32763
+ /**
32593
32764
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32594
32765
  * device-provider cap router. Shared across all providers.
32595
32766
  */
@@ -32638,6 +32809,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32638
32809
  }];
32639
32810
  }
32640
32811
  async onShutdown() {
32812
+ this.cancelRestoreRetries();
32641
32813
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32642
32814
  for (const device of devices) try {
32643
32815
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32655,9 +32827,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32655
32827
  async start() {}
32656
32828
  async stop() {}
32657
32829
  async getStatus() {
32830
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32831
+ const summary = this.restoreFailureSummary();
32832
+ if (summary === null) return {
32833
+ connected: true,
32834
+ deviceCount: all.length
32835
+ };
32658
32836
  return {
32659
32837
  connected: true,
32660
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32838
+ deviceCount: all.length,
32839
+ error: summary
32661
32840
  };
32662
32841
  }
32663
32842
  async getDevices() {
@@ -32747,8 +32926,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32747
32926
  };
32748
32927
  }
32749
32928
  async restoreDevices(savedDevices) {
32750
- await this.onRestoreDevices(savedDevices);
32751
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32929
+ const report = await this.onRestoreDevices(savedDevices);
32930
+ if (savedDevices.length === 0) return;
32931
+ if (report && report.failedCount > 0) {
32932
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32933
+ return;
32934
+ }
32935
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32936
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32937
+ }
32938
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32939
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32940
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32941
+ * never re-stampede full-width while the initial pass does (D167). */
32942
+ restoreRetryConcurrency = 4;
32943
+ _restoreRetryScheduler = null;
32944
+ _restoreRetryCompletion = null;
32945
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32946
+ /** Settles when the background retry rounds finish (or `null` when
32947
+ * nothing failed). Exposed for tests and subclass diagnostics —
32948
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32949
+ * with the devices that restored, and a late success is announced
32950
+ * through the `native-cap-change` → `updateCaps` path. */
32951
+ get restoreRetryCompletion() {
32952
+ return this._restoreRetryCompletion;
32953
+ }
32954
+ /** Devices that exhausted the retry bound this process lifetime. */
32955
+ get permanentRestoreFailures() {
32956
+ return [...this._permanentRestoreFailures.values()];
32957
+ }
32958
+ /** One-line operator-facing summary for `getStatus().error`, or
32959
+ * `null` when every device restored. */
32960
+ restoreFailureSummary() {
32961
+ if (this._permanentRestoreFailures.size === 0) return null;
32962
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32963
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32964
+ }
32965
+ cancelRestoreRetries() {
32966
+ this._restoreRetryScheduler?.cancel();
32967
+ this._restoreRetryScheduler = null;
32968
+ }
32969
+ recordPermanentRestoreFailure(failure) {
32970
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32971
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32972
+ tags: {
32973
+ deviceId: failure.deviceId,
32974
+ stableId: failure.stableId
32975
+ },
32976
+ meta: {
32977
+ type: failure.type,
32978
+ attempts: failure.attempts,
32979
+ error: failure.lastError
32980
+ }
32981
+ });
32982
+ }
32983
+ scheduleRestoreRetries(failures, attempt) {
32984
+ const scheduler = new DeviceRestoreRetryScheduler({
32985
+ logger: this.ctx.logger,
32986
+ delaysMs: this.restoreRetryDelaysMs,
32987
+ concurrency: this.restoreRetryConcurrency,
32988
+ attempt,
32989
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32990
+ });
32991
+ this._restoreRetryScheduler = scheduler;
32992
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32993
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32994
+ });
32995
+ }
32996
+ /**
32997
+ * Tear down and reconstruct ONE device from its persisted rows — the
32998
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32999
+ * and no other device this provider owns is disturbed.
33000
+ *
33001
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33002
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33003
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33004
+ * whatever number the row carries NOW. The teardown is `decommission` —
33005
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33006
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33007
+ * the boot restore's own `create()` path, including its pass 2: first-class
33008
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33009
+ * parent by the cascade and must be re-created explicitly, because only
33010
+ * accessory children come back through `getAccessoryChildren()`.
33011
+ *
33012
+ * Reloading an accessory child directly is refused (no device class) —
33013
+ * reload its parent instead.
33014
+ */
33015
+ async reloadDevice(input) {
33016
+ const { stableId } = input;
33017
+ const devices = this.ctx.kernel.devices;
33018
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33019
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33020
+ if (live) await devices.decommission(live.id);
33021
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33022
+ addonId: this.addonId,
33023
+ stableId
33024
+ });
33025
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33026
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33027
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33028
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33029
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33030
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33031
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33032
+ for (const row of rows) {
33033
+ if (row.parentDeviceId !== id) continue;
33034
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33035
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33036
+ if (!ChildClass) continue;
33037
+ try {
33038
+ await devices.create(row.stableId, ChildClass, {}, id);
33039
+ } catch (err) {
33040
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33041
+ tags: {
33042
+ deviceId: row.id,
33043
+ stableId: row.stableId
33044
+ },
33045
+ meta: {
33046
+ parentDeviceId: id,
33047
+ error: err instanceof Error ? err.message : String(err)
33048
+ }
33049
+ });
33050
+ }
33051
+ }
33052
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33053
+ tags: { deviceId: id },
33054
+ meta: {
33055
+ stableId,
33056
+ type: meta.type
33057
+ }
33058
+ });
33059
+ return { deviceId: id };
32752
33060
  }
32753
33061
  /**
32754
33062
  * Restore devices from persisted state. Two-pass:
@@ -32774,55 +33082,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32774
33082
  * accessory-spawn flow handles via the parent's
32775
33083
  * `getAccessoryChildren()`. Override only when the default doesn't
32776
33084
  * fit.
33085
+ *
33086
+ * A row that fails either pass is NOT terminal (D347): it is handed
33087
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33088
+ * Only after the bound is exhausted is the device marked permanently
33089
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33090
+ * `getStatus().error`.
32777
33091
  */
32778
33092
  async onRestoreDevices(savedDevices) {
32779
33093
  const restored = /* @__PURE__ */ new Set();
33094
+ const failures = [];
33095
+ const attemptRestore = async (saved) => {
33096
+ if (restored.has(saved.id)) return;
33097
+ const Class = this.deviceClasses[saved.type];
33098
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33099
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33100
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33101
+ restored.add(saved.id);
33102
+ };
32780
33103
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32781
33104
  const restoreOne = async (saved) => {
32782
- const Class = this.deviceClasses[saved.type];
32783
- if (!Class) {
33105
+ if (!this.deviceClasses[saved.type]) {
32784
33106
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32785
- tags: { stableId: saved.stableId },
33107
+ tags: {
33108
+ deviceId: saved.id,
33109
+ stableId: saved.stableId
33110
+ },
32786
33111
  meta: { type: saved.type }
32787
33112
  });
32788
33113
  return;
32789
33114
  }
32790
33115
  try {
32791
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32792
- restored.add(saved.id);
33116
+ await attemptRestore(saved);
32793
33117
  } catch (err) {
32794
- this.ctx.logger.warn("Failed to restore device", {
32795
- tags: { stableId: saved.stableId },
33118
+ const error = err instanceof Error ? err.message : String(err);
33119
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33120
+ tags: {
33121
+ deviceId: saved.id,
33122
+ stableId: saved.stableId
33123
+ },
32796
33124
  meta: {
32797
33125
  type: saved.type,
32798
- error: err instanceof Error ? err.message : String(err)
33126
+ attempt: 1,
33127
+ error
32799
33128
  }
32800
33129
  });
33130
+ failures.push({
33131
+ saved,
33132
+ error
33133
+ });
32801
33134
  }
32802
33135
  };
32803
33136
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33137
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32804
33138
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32805
33139
  for (const saved of childRows) {
32806
- const Class = this.deviceClasses[saved.type];
32807
- if (!Class) continue;
33140
+ if (!this.deviceClasses[saved.type]) continue;
32808
33141
  if (saved.parentDeviceId === null) continue;
32809
- if (!restored.has(saved.parentDeviceId)) continue;
32810
- try {
32811
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32812
- restored.add(saved.id);
32813
- } catch (err) {
32814
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33142
+ if (restored.has(saved.parentDeviceId)) {
33143
+ try {
33144
+ await attemptRestore(saved);
33145
+ } catch (err) {
33146
+ const error = err instanceof Error ? err.message : String(err);
33147
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33148
+ tags: {
33149
+ deviceId: saved.id,
33150
+ stableId: saved.stableId,
33151
+ parentDeviceId: saved.parentDeviceId
33152
+ },
33153
+ meta: {
33154
+ type: saved.type,
33155
+ attempt: 1,
33156
+ error
33157
+ }
33158
+ });
33159
+ failures.push({
33160
+ saved,
33161
+ error
33162
+ });
33163
+ }
33164
+ continue;
33165
+ }
33166
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33167
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32815
33168
  tags: {
33169
+ deviceId: saved.id,
32816
33170
  stableId: saved.stableId,
32817
33171
  parentDeviceId: saved.parentDeviceId
32818
33172
  },
32819
- meta: {
32820
- type: saved.type,
32821
- error: err instanceof Error ? err.message : String(err)
32822
- }
33173
+ meta: { type: saved.type }
33174
+ });
33175
+ failures.push({
33176
+ saved,
33177
+ error: `parent device ${saved.parentDeviceId} not restored`
32823
33178
  });
33179
+ continue;
32824
33180
  }
32825
33181
  }
33182
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33183
+ return {
33184
+ restoredCount: restored.size,
33185
+ failedCount: failures.length
33186
+ };
32826
33187
  }
32827
33188
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32828
33189
  toSummary(device) {
@@ -34585,6 +34946,12 @@ Object.freeze({
34585
34946
  addonId: null,
34586
34947
  access: "view"
34587
34948
  },
34949
+ "deviceProvider.reloadDevice": {
34950
+ capName: "device-provider",
34951
+ capScope: "system",
34952
+ addonId: null,
34953
+ access: "create"
34954
+ },
34588
34955
  "deviceProvider.start": {
34589
34956
  capName: "device-provider",
34590
34957
  capScope: "system",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-matter-broker",
3
- "version": "0.2.58",
3
+ "version": "0.2.59",
4
4
  "description": "Matter broker addon for CamStack — owns a Matter fabric (commissioning + the long-lived controller) via the matter.js controller and brokers commissioned Matter nodes into CamStack",
5
5
  "keywords": [
6
6
  "camstack",