@camstack/addon-matter-broker 0.2.57 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10541,6 +10541,89 @@ var LocationStatSchema = object({
10541
10541
  fileCount: number(),
10542
10542
  present: boolean()
10543
10543
  });
10544
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10545
+ var BackupRunStateSchema = _enum([
10546
+ "queued",
10547
+ "running",
10548
+ "succeeded",
10549
+ "failed",
10550
+ "cancelled"
10551
+ ]);
10552
+ /**
10553
+ * Where a running backup currently is. `queued` before it starts,
10554
+ * `building` while the tar.gz is being staged, `uploading` during the
10555
+ * per-destination fan-out, `done` once terminal.
10556
+ */
10557
+ var BackupRunPhaseSchema = _enum([
10558
+ "queued",
10559
+ "building",
10560
+ "uploading",
10561
+ "done"
10562
+ ]);
10563
+ /**
10564
+ * Observable state of one backup run — readable WHILE it runs via
10565
+ * `backup.listRuns`. This is what makes the execution queue and
10566
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10567
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10568
+ * diagnosable with `du` because nothing reported that runs existed or
10569
+ * how large the staged archive had grown.
10570
+ */
10571
+ var BackupRunSchema = object({
10572
+ /** Stable run id — the handle `backup.cancel` takes. */
10573
+ id: string$2(),
10574
+ state: BackupRunStateSchema,
10575
+ phase: BackupRunPhaseSchema,
10576
+ /**
10577
+ * Resolved destination location ids. Empty while queued (targets are
10578
+ * resolved when the run starts, against the then-current policies).
10579
+ */
10580
+ destinationIds: array(string$2()).readonly(),
10581
+ label: string$2().optional(),
10582
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10583
+ requestedAt: number(),
10584
+ /** ms-epoch when the run left the queue and started building. */
10585
+ startedAt: number().optional(),
10586
+ /** ms-epoch when the run reached a terminal state. */
10587
+ finishedAt: number().optional(),
10588
+ /** Compressed bytes of the staging archive written so far. */
10589
+ stagedBytes: number(),
10590
+ /** Final staged archive size, once the build phase completes. */
10591
+ archiveSizeBytes: number().optional(),
10592
+ /** Bytes pushed to the destination currently uploading. */
10593
+ uploadedBytes: number(),
10594
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10595
+ completedDestinationIds: array(string$2()).readonly(),
10596
+ /** Destinations that failed during the fan-out. */
10597
+ failedDestinationIds: array(string$2()).readonly(),
10598
+ /** Failure message when `state === 'failed'`. */
10599
+ error: string$2().optional(),
10600
+ /**
10601
+ * 1-based place in the execution queue — 1 = runs next. Present only
10602
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10603
+ * queue's OWN pending order, never derived from timestamps, so the
10604
+ * UI cannot show an order the executor will not honour.
10605
+ */
10606
+ queuePosition: number().int().min(1).optional()
10607
+ });
10608
+ /**
10609
+ * Result of `backup.trigger`. The call still resolves when the run
10610
+ * terminates (compat with schedule-driven runs and the admin UI), but
10611
+ * it now names the run and says whether it had to WAIT: a trigger that
10612
+ * arrives while another run is in flight is enqueued (or joined onto
10613
+ * an identical already-queued run), never started concurrently.
10614
+ */
10615
+ var BackupTriggerResultSchema = object({
10616
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10617
+ runId: string$2(),
10618
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10619
+ queued: boolean(),
10620
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10621
+ joined: boolean(),
10622
+ /** True when the run was cancelled before completing every destination. */
10623
+ cancelled: boolean(),
10624
+ /** One entry per destination the archive landed at (partial on cancel). */
10625
+ entries: array(BackupEntrySchema).readonly()
10626
+ });
10544
10627
  /**
10545
10628
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10546
10629
  * SET of destination locations. Supersedes the per-location cron on
@@ -10588,7 +10671,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10588
10671
  * retention (manual runs).
10589
10672
  */
10590
10673
  retentionCount: number().int().min(1).max(1e3).optional()
10591
- }).optional(), array(BackupEntrySchema).readonly(), {
10674
+ }).optional(), BackupTriggerResultSchema, {
10675
+ kind: "mutation",
10676
+ auth: "admin"
10677
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string$2() }), object({ cancelled: boolean() }), {
10592
10678
  kind: "mutation",
10593
10679
  auth: "admin"
10594
10680
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11354,6 +11440,14 @@ method(object({
11354
11440
  }), object({ success: literal(true) }), {
11355
11441
  kind: "mutation",
11356
11442
  auth: "admin"
11443
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11444
+ derivedStreamsDeleted: array(string$2()).readonly(),
11445
+ assignmentsPurged: boolean(),
11446
+ probeSnapshotsDropped: number().int().nonnegative(),
11447
+ rtspTokenRowsDeleted: number().int().nonnegative()
11448
+ }), {
11449
+ kind: "mutation",
11450
+ auth: "admin"
11357
11451
  }), method(object({
11358
11452
  deviceId: number(),
11359
11453
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12798,6 +12892,35 @@ var deviceProviderCapability = {
12798
12892
  name: string$2(),
12799
12893
  type: string$2()
12800
12894
  }))),
12895
+ /**
12896
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12897
+ * touching no other device this provider owns.
12898
+ *
12899
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12900
+ * migrated numbers: after `swapIds` the runner's live instance still
12901
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12902
+ * registrations and its log tags), and a live object cannot be renumbered.
12903
+ * Before this method the only flush was restarting the whole owning addon
12904
+ * — which took every camera the provider owns down with it (28 devices
12905
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12906
+ * same day ~27 devices' native caps did not come back on their own).
12907
+ *
12908
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12909
+ * that changes. The reply carries the id the device answers on NOW.
12910
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12911
+ * instance (if any), then re-create from the persisted row: the same
12912
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12913
+ * An RPC, never an event: a dropped event would leave the runner writing
12914
+ * against the wrong camera (D8).
12915
+ *
12916
+ * Construction can dial hardware, and the migrated source is
12917
+ * characteristically dead — the timeout covers a full activate window
12918
+ * rather than the 60 s default.
12919
+ */
12920
+ reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
12921
+ kind: "mutation",
12922
+ timeoutMs: 3 * 6e4
12923
+ }),
12801
12924
  supportsDiscovery: method(object({}), boolean()),
12802
12925
  /**
12803
12926
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13125,7 +13248,8 @@ method(object({
13125
13248
  targetId: number()
13126
13249
  }), MigrateDeviceResultSchema, {
13127
13250
  kind: "mutation",
13128
- auth: "admin"
13251
+ auth: "admin",
13252
+ timeoutMs: 12 * 6e4
13129
13253
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13130
13254
  deviceId: number(),
13131
13255
  name: string$2()
@@ -32498,6 +32622,147 @@ var BaseDevice = class {
32498
32622
  }
32499
32623
  };
32500
32624
  /**
32625
+ * Delays before retry rounds 1..N — the round count IS the bound.
32626
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32627
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32628
+ * per attempt) covers a device-manager lock held for minutes — the
32629
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32630
+ */
32631
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32632
+ 1e4,
32633
+ 3e4,
32634
+ 9e4
32635
+ ];
32636
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32637
+ function sleep$1(ms, signal) {
32638
+ return new Promise((resolve) => {
32639
+ if (signal.aborted) {
32640
+ resolve();
32641
+ return;
32642
+ }
32643
+ const onAbort = () => {
32644
+ clearTimeout(timer);
32645
+ resolve();
32646
+ };
32647
+ const timer = setTimeout(() => {
32648
+ signal.removeEventListener("abort", onAbort);
32649
+ resolve();
32650
+ }, ms);
32651
+ timer.unref?.();
32652
+ signal.addEventListener("abort", onAbort, { once: true });
32653
+ });
32654
+ }
32655
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32656
+ * not reject (callers wrap their own try/catch). */
32657
+ async function runWithConcurrency(items, width, fn) {
32658
+ const queue = [...items];
32659
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32660
+ const lane = async () => {
32661
+ for (;;) {
32662
+ const item = queue.shift();
32663
+ if (item === void 0) return;
32664
+ await fn(item);
32665
+ }
32666
+ };
32667
+ await Promise.all(Array.from({ length: laneCount }, lane));
32668
+ }
32669
+ var DeviceRestoreRetryScheduler = class {
32670
+ #logger;
32671
+ #attempt;
32672
+ #onPermanentFailure;
32673
+ #delaysMs;
32674
+ #concurrency;
32675
+ #now;
32676
+ #abort = new AbortController();
32677
+ constructor(options) {
32678
+ this.#logger = options.logger;
32679
+ this.#attempt = options.attempt;
32680
+ this.#onPermanentFailure = options.onPermanentFailure;
32681
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32682
+ this.#concurrency = options.concurrency ?? 4;
32683
+ this.#now = options.now ?? Date.now;
32684
+ }
32685
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32686
+ * permanently failed — the next boot restores them from disk. */
32687
+ cancel() {
32688
+ this.#abort.abort();
32689
+ }
32690
+ /**
32691
+ * Run the bounded retry rounds. Resolves when every entry has either
32692
+ * restored, been marked permanently failed, or the scheduler was
32693
+ * cancelled. Never rejects.
32694
+ */
32695
+ async run(initialFailures) {
32696
+ let pending = initialFailures.map((failure) => ({
32697
+ saved: failure.saved,
32698
+ lastError: failure.error,
32699
+ attempts: 1
32700
+ }));
32701
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32702
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32703
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32704
+ if (this.#abort.signal.aborted) break;
32705
+ pending = await this.#runRound(pending, round);
32706
+ }
32707
+ if (this.#abort.signal.aborted) return [];
32708
+ const terminal = pending.map((entry) => ({
32709
+ deviceId: entry.saved.id,
32710
+ stableId: entry.saved.stableId,
32711
+ type: String(entry.saved.type),
32712
+ attempts: entry.attempts,
32713
+ lastError: entry.lastError,
32714
+ failedAt: this.#now()
32715
+ }));
32716
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32717
+ return terminal;
32718
+ }
32719
+ /** One retry round: parents first (phase 0), then hub-adopted
32720
+ * children (phase 1) — a child's attempt depends on its parent
32721
+ * having landed, exactly like the initial two-pass restore. */
32722
+ async #runRound(pending, round) {
32723
+ const next = [];
32724
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32725
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32726
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32727
+ if (this.#abort.signal.aborted) {
32728
+ next.push(entry);
32729
+ return;
32730
+ }
32731
+ const attemptNo = entry.attempts + 1;
32732
+ try {
32733
+ await this.#attempt(entry.saved);
32734
+ this.#logger.info("Device restored on retry", {
32735
+ tags: {
32736
+ deviceId: entry.saved.id,
32737
+ stableId: entry.saved.stableId
32738
+ },
32739
+ meta: { attempt: attemptNo }
32740
+ });
32741
+ } catch (err) {
32742
+ const lastError = err instanceof Error ? err.message : String(err);
32743
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32744
+ this.#logger.warn("Device restore retry failed", {
32745
+ tags: {
32746
+ deviceId: entry.saved.id,
32747
+ stableId: entry.saved.stableId
32748
+ },
32749
+ meta: {
32750
+ attempt: attemptNo,
32751
+ remainingRetries,
32752
+ error: lastError
32753
+ }
32754
+ });
32755
+ next.push({
32756
+ saved: entry.saved,
32757
+ lastError,
32758
+ attempts: attemptNo
32759
+ });
32760
+ }
32761
+ });
32762
+ return next;
32763
+ }
32764
+ };
32765
+ /**
32501
32766
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32502
32767
  * device-provider cap router. Shared across all providers.
32503
32768
  */
@@ -32546,6 +32811,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32546
32811
  }];
32547
32812
  }
32548
32813
  async onShutdown() {
32814
+ this.cancelRestoreRetries();
32549
32815
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32550
32816
  for (const device of devices) try {
32551
32817
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32563,9 +32829,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32563
32829
  async start() {}
32564
32830
  async stop() {}
32565
32831
  async getStatus() {
32832
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32833
+ const summary = this.restoreFailureSummary();
32834
+ if (summary === null) return {
32835
+ connected: true,
32836
+ deviceCount: all.length
32837
+ };
32566
32838
  return {
32567
32839
  connected: true,
32568
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32840
+ deviceCount: all.length,
32841
+ error: summary
32569
32842
  };
32570
32843
  }
32571
32844
  async getDevices() {
@@ -32655,8 +32928,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32655
32928
  };
32656
32929
  }
32657
32930
  async restoreDevices(savedDevices) {
32658
- await this.onRestoreDevices(savedDevices);
32659
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32931
+ const report = await this.onRestoreDevices(savedDevices);
32932
+ if (savedDevices.length === 0) return;
32933
+ if (report && report.failedCount > 0) {
32934
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32935
+ return;
32936
+ }
32937
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32938
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32939
+ }
32940
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32941
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32942
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32943
+ * never re-stampede full-width while the initial pass does (D167). */
32944
+ restoreRetryConcurrency = 4;
32945
+ _restoreRetryScheduler = null;
32946
+ _restoreRetryCompletion = null;
32947
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32948
+ /** Settles when the background retry rounds finish (or `null` when
32949
+ * nothing failed). Exposed for tests and subclass diagnostics —
32950
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32951
+ * with the devices that restored, and a late success is announced
32952
+ * through the `native-cap-change` → `updateCaps` path. */
32953
+ get restoreRetryCompletion() {
32954
+ return this._restoreRetryCompletion;
32955
+ }
32956
+ /** Devices that exhausted the retry bound this process lifetime. */
32957
+ get permanentRestoreFailures() {
32958
+ return [...this._permanentRestoreFailures.values()];
32959
+ }
32960
+ /** One-line operator-facing summary for `getStatus().error`, or
32961
+ * `null` when every device restored. */
32962
+ restoreFailureSummary() {
32963
+ if (this._permanentRestoreFailures.size === 0) return null;
32964
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32965
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32966
+ }
32967
+ cancelRestoreRetries() {
32968
+ this._restoreRetryScheduler?.cancel();
32969
+ this._restoreRetryScheduler = null;
32970
+ }
32971
+ recordPermanentRestoreFailure(failure) {
32972
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32973
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32974
+ tags: {
32975
+ deviceId: failure.deviceId,
32976
+ stableId: failure.stableId
32977
+ },
32978
+ meta: {
32979
+ type: failure.type,
32980
+ attempts: failure.attempts,
32981
+ error: failure.lastError
32982
+ }
32983
+ });
32984
+ }
32985
+ scheduleRestoreRetries(failures, attempt) {
32986
+ const scheduler = new DeviceRestoreRetryScheduler({
32987
+ logger: this.ctx.logger,
32988
+ delaysMs: this.restoreRetryDelaysMs,
32989
+ concurrency: this.restoreRetryConcurrency,
32990
+ attempt,
32991
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32992
+ });
32993
+ this._restoreRetryScheduler = scheduler;
32994
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32995
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32996
+ });
32997
+ }
32998
+ /**
32999
+ * Tear down and reconstruct ONE device from its persisted rows — the
33000
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33001
+ * and no other device this provider owns is disturbed.
33002
+ *
33003
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33004
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33005
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33006
+ * whatever number the row carries NOW. The teardown is `decommission` —
33007
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33008
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33009
+ * the boot restore's own `create()` path, including its pass 2: first-class
33010
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33011
+ * parent by the cascade and must be re-created explicitly, because only
33012
+ * accessory children come back through `getAccessoryChildren()`.
33013
+ *
33014
+ * Reloading an accessory child directly is refused (no device class) —
33015
+ * reload its parent instead.
33016
+ */
33017
+ async reloadDevice(input) {
33018
+ const { stableId } = input;
33019
+ const devices = this.ctx.kernel.devices;
33020
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33021
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33022
+ if (live) await devices.decommission(live.id);
33023
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33024
+ addonId: this.addonId,
33025
+ stableId
33026
+ });
33027
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33028
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33029
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33030
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33031
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33032
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33033
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33034
+ for (const row of rows) {
33035
+ if (row.parentDeviceId !== id) continue;
33036
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33037
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33038
+ if (!ChildClass) continue;
33039
+ try {
33040
+ await devices.create(row.stableId, ChildClass, {}, id);
33041
+ } catch (err) {
33042
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33043
+ tags: {
33044
+ deviceId: row.id,
33045
+ stableId: row.stableId
33046
+ },
33047
+ meta: {
33048
+ parentDeviceId: id,
33049
+ error: err instanceof Error ? err.message : String(err)
33050
+ }
33051
+ });
33052
+ }
33053
+ }
33054
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33055
+ tags: { deviceId: id },
33056
+ meta: {
33057
+ stableId,
33058
+ type: meta.type
33059
+ }
33060
+ });
33061
+ return { deviceId: id };
32660
33062
  }
32661
33063
  /**
32662
33064
  * Restore devices from persisted state. Two-pass:
@@ -32682,55 +33084,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32682
33084
  * accessory-spawn flow handles via the parent's
32683
33085
  * `getAccessoryChildren()`. Override only when the default doesn't
32684
33086
  * fit.
33087
+ *
33088
+ * A row that fails either pass is NOT terminal (D347): it is handed
33089
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33090
+ * Only after the bound is exhausted is the device marked permanently
33091
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33092
+ * `getStatus().error`.
32685
33093
  */
32686
33094
  async onRestoreDevices(savedDevices) {
32687
33095
  const restored = /* @__PURE__ */ new Set();
33096
+ const failures = [];
33097
+ const attemptRestore = async (saved) => {
33098
+ if (restored.has(saved.id)) return;
33099
+ const Class = this.deviceClasses[saved.type];
33100
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33101
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33102
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33103
+ restored.add(saved.id);
33104
+ };
32688
33105
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32689
33106
  const restoreOne = async (saved) => {
32690
- const Class = this.deviceClasses[saved.type];
32691
- if (!Class) {
33107
+ if (!this.deviceClasses[saved.type]) {
32692
33108
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32693
- tags: { stableId: saved.stableId },
33109
+ tags: {
33110
+ deviceId: saved.id,
33111
+ stableId: saved.stableId
33112
+ },
32694
33113
  meta: { type: saved.type }
32695
33114
  });
32696
33115
  return;
32697
33116
  }
32698
33117
  try {
32699
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32700
- restored.add(saved.id);
33118
+ await attemptRestore(saved);
32701
33119
  } catch (err) {
32702
- this.ctx.logger.warn("Failed to restore device", {
32703
- tags: { stableId: saved.stableId },
33120
+ const error = err instanceof Error ? err.message : String(err);
33121
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33122
+ tags: {
33123
+ deviceId: saved.id,
33124
+ stableId: saved.stableId
33125
+ },
32704
33126
  meta: {
32705
33127
  type: saved.type,
32706
- error: err instanceof Error ? err.message : String(err)
33128
+ attempt: 1,
33129
+ error
32707
33130
  }
32708
33131
  });
33132
+ failures.push({
33133
+ saved,
33134
+ error
33135
+ });
32709
33136
  }
32710
33137
  };
32711
33138
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33139
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32712
33140
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32713
33141
  for (const saved of childRows) {
32714
- const Class = this.deviceClasses[saved.type];
32715
- if (!Class) continue;
33142
+ if (!this.deviceClasses[saved.type]) continue;
32716
33143
  if (saved.parentDeviceId === null) continue;
32717
- if (!restored.has(saved.parentDeviceId)) continue;
32718
- try {
32719
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32720
- restored.add(saved.id);
32721
- } catch (err) {
32722
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33144
+ if (restored.has(saved.parentDeviceId)) {
33145
+ try {
33146
+ await attemptRestore(saved);
33147
+ } catch (err) {
33148
+ const error = err instanceof Error ? err.message : String(err);
33149
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33150
+ tags: {
33151
+ deviceId: saved.id,
33152
+ stableId: saved.stableId,
33153
+ parentDeviceId: saved.parentDeviceId
33154
+ },
33155
+ meta: {
33156
+ type: saved.type,
33157
+ attempt: 1,
33158
+ error
33159
+ }
33160
+ });
33161
+ failures.push({
33162
+ saved,
33163
+ error
33164
+ });
33165
+ }
33166
+ continue;
33167
+ }
33168
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33169
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32723
33170
  tags: {
33171
+ deviceId: saved.id,
32724
33172
  stableId: saved.stableId,
32725
33173
  parentDeviceId: saved.parentDeviceId
32726
33174
  },
32727
- meta: {
32728
- type: saved.type,
32729
- error: err instanceof Error ? err.message : String(err)
32730
- }
33175
+ meta: { type: saved.type }
33176
+ });
33177
+ failures.push({
33178
+ saved,
33179
+ error: `parent device ${saved.parentDeviceId} not restored`
32731
33180
  });
33181
+ continue;
32732
33182
  }
32733
33183
  }
33184
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33185
+ return {
33186
+ restoredCount: restored.size,
33187
+ failedCount: failures.length
33188
+ };
32734
33189
  }
32735
33190
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32736
33191
  toSummary(device) {
@@ -33239,6 +33694,12 @@ Object.freeze({
33239
33694
  addonId: null,
33240
33695
  access: "create"
33241
33696
  },
33697
+ "backup.cancel": {
33698
+ capName: "backup",
33699
+ capScope: "system",
33700
+ addonId: null,
33701
+ access: "create"
33702
+ },
33242
33703
  "backup.delete": {
33243
33704
  capName: "backup",
33244
33705
  capScope: "system",
@@ -33281,6 +33742,12 @@ Object.freeze({
33281
33742
  addonId: null,
33282
33743
  access: "view"
33283
33744
  },
33745
+ "backup.listRuns": {
33746
+ capName: "backup",
33747
+ capScope: "system",
33748
+ addonId: null,
33749
+ access: "view"
33750
+ },
33284
33751
  "backup.listSchedules": {
33285
33752
  capName: "backup",
33286
33753
  capScope: "system",
@@ -34481,6 +34948,12 @@ Object.freeze({
34481
34948
  addonId: null,
34482
34949
  access: "view"
34483
34950
  },
34951
+ "deviceProvider.reloadDevice": {
34952
+ capName: "device-provider",
34953
+ capScope: "system",
34954
+ addonId: null,
34955
+ access: "create"
34956
+ },
34484
34957
  "deviceProvider.start": {
34485
34958
  capName: "device-provider",
34486
34959
  capScope: "system",
@@ -37847,6 +38320,12 @@ Object.freeze({
37847
38320
  addonId: null,
37848
38321
  access: "create"
37849
38322
  },
38323
+ "streamBroker.forgetDeviceHardware": {
38324
+ capName: "stream-broker",
38325
+ capScope: "system",
38326
+ addonId: null,
38327
+ access: "delete"
38328
+ },
37850
38329
  "streamBroker.getAllRtspEntries": {
37851
38330
  capName: "stream-broker",
37852
38331
  capScope: "system",
@@ -40305,6 +40784,11 @@ Object.freeze({
40305
40784
  form: "single",
40306
40785
  optional: false
40307
40786
  }],
40787
+ "streamBroker.forgetDeviceHardware": [{
40788
+ name: "deviceId",
40789
+ form: "single",
40790
+ optional: false
40791
+ }],
40308
40792
  "streamBroker.getDeviceAudioMute": [{
40309
40793
  name: "deviceId",
40310
40794
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10539,6 +10539,89 @@ var LocationStatSchema = object({
10539
10539
  fileCount: number(),
10540
10540
  present: boolean()
10541
10541
  });
10542
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10543
+ var BackupRunStateSchema = _enum([
10544
+ "queued",
10545
+ "running",
10546
+ "succeeded",
10547
+ "failed",
10548
+ "cancelled"
10549
+ ]);
10550
+ /**
10551
+ * Where a running backup currently is. `queued` before it starts,
10552
+ * `building` while the tar.gz is being staged, `uploading` during the
10553
+ * per-destination fan-out, `done` once terminal.
10554
+ */
10555
+ var BackupRunPhaseSchema = _enum([
10556
+ "queued",
10557
+ "building",
10558
+ "uploading",
10559
+ "done"
10560
+ ]);
10561
+ /**
10562
+ * Observable state of one backup run — readable WHILE it runs via
10563
+ * `backup.listRuns`. This is what makes the execution queue and
10564
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10565
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10566
+ * diagnosable with `du` because nothing reported that runs existed or
10567
+ * how large the staged archive had grown.
10568
+ */
10569
+ var BackupRunSchema = object({
10570
+ /** Stable run id — the handle `backup.cancel` takes. */
10571
+ id: string$2(),
10572
+ state: BackupRunStateSchema,
10573
+ phase: BackupRunPhaseSchema,
10574
+ /**
10575
+ * Resolved destination location ids. Empty while queued (targets are
10576
+ * resolved when the run starts, against the then-current policies).
10577
+ */
10578
+ destinationIds: array(string$2()).readonly(),
10579
+ label: string$2().optional(),
10580
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10581
+ requestedAt: number(),
10582
+ /** ms-epoch when the run left the queue and started building. */
10583
+ startedAt: number().optional(),
10584
+ /** ms-epoch when the run reached a terminal state. */
10585
+ finishedAt: number().optional(),
10586
+ /** Compressed bytes of the staging archive written so far. */
10587
+ stagedBytes: number(),
10588
+ /** Final staged archive size, once the build phase completes. */
10589
+ archiveSizeBytes: number().optional(),
10590
+ /** Bytes pushed to the destination currently uploading. */
10591
+ uploadedBytes: number(),
10592
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10593
+ completedDestinationIds: array(string$2()).readonly(),
10594
+ /** Destinations that failed during the fan-out. */
10595
+ failedDestinationIds: array(string$2()).readonly(),
10596
+ /** Failure message when `state === 'failed'`. */
10597
+ error: string$2().optional(),
10598
+ /**
10599
+ * 1-based place in the execution queue — 1 = runs next. Present only
10600
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10601
+ * queue's OWN pending order, never derived from timestamps, so the
10602
+ * UI cannot show an order the executor will not honour.
10603
+ */
10604
+ queuePosition: number().int().min(1).optional()
10605
+ });
10606
+ /**
10607
+ * Result of `backup.trigger`. The call still resolves when the run
10608
+ * terminates (compat with schedule-driven runs and the admin UI), but
10609
+ * it now names the run and says whether it had to WAIT: a trigger that
10610
+ * arrives while another run is in flight is enqueued (or joined onto
10611
+ * an identical already-queued run), never started concurrently.
10612
+ */
10613
+ var BackupTriggerResultSchema = object({
10614
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10615
+ runId: string$2(),
10616
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10617
+ queued: boolean(),
10618
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10619
+ joined: boolean(),
10620
+ /** True when the run was cancelled before completing every destination. */
10621
+ cancelled: boolean(),
10622
+ /** One entry per destination the archive landed at (partial on cancel). */
10623
+ entries: array(BackupEntrySchema).readonly()
10624
+ });
10542
10625
  /**
10543
10626
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10544
10627
  * SET of destination locations. Supersedes the per-location cron on
@@ -10586,7 +10669,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10586
10669
  * retention (manual runs).
10587
10670
  */
10588
10671
  retentionCount: number().int().min(1).max(1e3).optional()
10589
- }).optional(), array(BackupEntrySchema).readonly(), {
10672
+ }).optional(), BackupTriggerResultSchema, {
10673
+ kind: "mutation",
10674
+ auth: "admin"
10675
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string$2() }), object({ cancelled: boolean() }), {
10590
10676
  kind: "mutation",
10591
10677
  auth: "admin"
10592
10678
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11352,6 +11438,14 @@ method(object({
11352
11438
  }), object({ success: literal(true) }), {
11353
11439
  kind: "mutation",
11354
11440
  auth: "admin"
11441
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11442
+ derivedStreamsDeleted: array(string$2()).readonly(),
11443
+ assignmentsPurged: boolean(),
11444
+ probeSnapshotsDropped: number().int().nonnegative(),
11445
+ rtspTokenRowsDeleted: number().int().nonnegative()
11446
+ }), {
11447
+ kind: "mutation",
11448
+ auth: "admin"
11355
11449
  }), method(object({
11356
11450
  deviceId: number(),
11357
11451
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12796,6 +12890,35 @@ var deviceProviderCapability = {
12796
12890
  name: string$2(),
12797
12891
  type: string$2()
12798
12892
  }))),
12893
+ /**
12894
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12895
+ * touching no other device this provider owns.
12896
+ *
12897
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12898
+ * migrated numbers: after `swapIds` the runner's live instance still
12899
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12900
+ * registrations and its log tags), and a live object cannot be renumbered.
12901
+ * Before this method the only flush was restarting the whole owning addon
12902
+ * — which took every camera the provider owns down with it (28 devices
12903
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12904
+ * same day ~27 devices' native caps did not come back on their own).
12905
+ *
12906
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12907
+ * that changes. The reply carries the id the device answers on NOW.
12908
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12909
+ * instance (if any), then re-create from the persisted row: the same
12910
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12911
+ * An RPC, never an event: a dropped event would leave the runner writing
12912
+ * against the wrong camera (D8).
12913
+ *
12914
+ * Construction can dial hardware, and the migrated source is
12915
+ * characteristically dead — the timeout covers a full activate window
12916
+ * rather than the 60 s default.
12917
+ */
12918
+ reloadDevice: method(object({ stableId: string$2() }), object({ deviceId: number() }), {
12919
+ kind: "mutation",
12920
+ timeoutMs: 3 * 6e4
12921
+ }),
12799
12922
  supportsDiscovery: method(object({}), boolean()),
12800
12923
  /**
12801
12924
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13123,7 +13246,8 @@ method(object({
13123
13246
  targetId: number()
13124
13247
  }), MigrateDeviceResultSchema, {
13125
13248
  kind: "mutation",
13126
- auth: "admin"
13249
+ auth: "admin",
13250
+ timeoutMs: 12 * 6e4
13127
13251
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), record(string$2(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13128
13252
  deviceId: number(),
13129
13253
  name: string$2()
@@ -32496,6 +32620,147 @@ var BaseDevice = class {
32496
32620
  }
32497
32621
  };
32498
32622
  /**
32623
+ * Delays before retry rounds 1..N — the round count IS the bound.
32624
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32625
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32626
+ * per attempt) covers a device-manager lock held for minutes — the
32627
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32628
+ */
32629
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32630
+ 1e4,
32631
+ 3e4,
32632
+ 9e4
32633
+ ];
32634
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32635
+ function sleep$1(ms, signal) {
32636
+ return new Promise((resolve) => {
32637
+ if (signal.aborted) {
32638
+ resolve();
32639
+ return;
32640
+ }
32641
+ const onAbort = () => {
32642
+ clearTimeout(timer);
32643
+ resolve();
32644
+ };
32645
+ const timer = setTimeout(() => {
32646
+ signal.removeEventListener("abort", onAbort);
32647
+ resolve();
32648
+ }, ms);
32649
+ timer.unref?.();
32650
+ signal.addEventListener("abort", onAbort, { once: true });
32651
+ });
32652
+ }
32653
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32654
+ * not reject (callers wrap their own try/catch). */
32655
+ async function runWithConcurrency(items, width, fn) {
32656
+ const queue = [...items];
32657
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32658
+ const lane = async () => {
32659
+ for (;;) {
32660
+ const item = queue.shift();
32661
+ if (item === void 0) return;
32662
+ await fn(item);
32663
+ }
32664
+ };
32665
+ await Promise.all(Array.from({ length: laneCount }, lane));
32666
+ }
32667
+ var DeviceRestoreRetryScheduler = class {
32668
+ #logger;
32669
+ #attempt;
32670
+ #onPermanentFailure;
32671
+ #delaysMs;
32672
+ #concurrency;
32673
+ #now;
32674
+ #abort = new AbortController();
32675
+ constructor(options) {
32676
+ this.#logger = options.logger;
32677
+ this.#attempt = options.attempt;
32678
+ this.#onPermanentFailure = options.onPermanentFailure;
32679
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32680
+ this.#concurrency = options.concurrency ?? 4;
32681
+ this.#now = options.now ?? Date.now;
32682
+ }
32683
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32684
+ * permanently failed — the next boot restores them from disk. */
32685
+ cancel() {
32686
+ this.#abort.abort();
32687
+ }
32688
+ /**
32689
+ * Run the bounded retry rounds. Resolves when every entry has either
32690
+ * restored, been marked permanently failed, or the scheduler was
32691
+ * cancelled. Never rejects.
32692
+ */
32693
+ async run(initialFailures) {
32694
+ let pending = initialFailures.map((failure) => ({
32695
+ saved: failure.saved,
32696
+ lastError: failure.error,
32697
+ attempts: 1
32698
+ }));
32699
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32700
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32701
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32702
+ if (this.#abort.signal.aborted) break;
32703
+ pending = await this.#runRound(pending, round);
32704
+ }
32705
+ if (this.#abort.signal.aborted) return [];
32706
+ const terminal = pending.map((entry) => ({
32707
+ deviceId: entry.saved.id,
32708
+ stableId: entry.saved.stableId,
32709
+ type: String(entry.saved.type),
32710
+ attempts: entry.attempts,
32711
+ lastError: entry.lastError,
32712
+ failedAt: this.#now()
32713
+ }));
32714
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32715
+ return terminal;
32716
+ }
32717
+ /** One retry round: parents first (phase 0), then hub-adopted
32718
+ * children (phase 1) — a child's attempt depends on its parent
32719
+ * having landed, exactly like the initial two-pass restore. */
32720
+ async #runRound(pending, round) {
32721
+ const next = [];
32722
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32723
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32724
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32725
+ if (this.#abort.signal.aborted) {
32726
+ next.push(entry);
32727
+ return;
32728
+ }
32729
+ const attemptNo = entry.attempts + 1;
32730
+ try {
32731
+ await this.#attempt(entry.saved);
32732
+ this.#logger.info("Device restored on retry", {
32733
+ tags: {
32734
+ deviceId: entry.saved.id,
32735
+ stableId: entry.saved.stableId
32736
+ },
32737
+ meta: { attempt: attemptNo }
32738
+ });
32739
+ } catch (err) {
32740
+ const lastError = err instanceof Error ? err.message : String(err);
32741
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32742
+ this.#logger.warn("Device restore retry failed", {
32743
+ tags: {
32744
+ deviceId: entry.saved.id,
32745
+ stableId: entry.saved.stableId
32746
+ },
32747
+ meta: {
32748
+ attempt: attemptNo,
32749
+ remainingRetries,
32750
+ error: lastError
32751
+ }
32752
+ });
32753
+ next.push({
32754
+ saved: entry.saved,
32755
+ lastError,
32756
+ attempts: attemptNo
32757
+ });
32758
+ }
32759
+ });
32760
+ return next;
32761
+ }
32762
+ };
32763
+ /**
32499
32764
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32500
32765
  * device-provider cap router. Shared across all providers.
32501
32766
  */
@@ -32544,6 +32809,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32544
32809
  }];
32545
32810
  }
32546
32811
  async onShutdown() {
32812
+ this.cancelRestoreRetries();
32547
32813
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32548
32814
  for (const device of devices) try {
32549
32815
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32561,9 +32827,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32561
32827
  async start() {}
32562
32828
  async stop() {}
32563
32829
  async getStatus() {
32830
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32831
+ const summary = this.restoreFailureSummary();
32832
+ if (summary === null) return {
32833
+ connected: true,
32834
+ deviceCount: all.length
32835
+ };
32564
32836
  return {
32565
32837
  connected: true,
32566
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32838
+ deviceCount: all.length,
32839
+ error: summary
32567
32840
  };
32568
32841
  }
32569
32842
  async getDevices() {
@@ -32653,8 +32926,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32653
32926
  };
32654
32927
  }
32655
32928
  async restoreDevices(savedDevices) {
32656
- await this.onRestoreDevices(savedDevices);
32657
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32929
+ const report = await this.onRestoreDevices(savedDevices);
32930
+ if (savedDevices.length === 0) return;
32931
+ if (report && report.failedCount > 0) {
32932
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32933
+ return;
32934
+ }
32935
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32936
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32937
+ }
32938
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32939
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32940
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32941
+ * never re-stampede full-width while the initial pass does (D167). */
32942
+ restoreRetryConcurrency = 4;
32943
+ _restoreRetryScheduler = null;
32944
+ _restoreRetryCompletion = null;
32945
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32946
+ /** Settles when the background retry rounds finish (or `null` when
32947
+ * nothing failed). Exposed for tests and subclass diagnostics —
32948
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32949
+ * with the devices that restored, and a late success is announced
32950
+ * through the `native-cap-change` → `updateCaps` path. */
32951
+ get restoreRetryCompletion() {
32952
+ return this._restoreRetryCompletion;
32953
+ }
32954
+ /** Devices that exhausted the retry bound this process lifetime. */
32955
+ get permanentRestoreFailures() {
32956
+ return [...this._permanentRestoreFailures.values()];
32957
+ }
32958
+ /** One-line operator-facing summary for `getStatus().error`, or
32959
+ * `null` when every device restored. */
32960
+ restoreFailureSummary() {
32961
+ if (this._permanentRestoreFailures.size === 0) return null;
32962
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32963
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32964
+ }
32965
+ cancelRestoreRetries() {
32966
+ this._restoreRetryScheduler?.cancel();
32967
+ this._restoreRetryScheduler = null;
32968
+ }
32969
+ recordPermanentRestoreFailure(failure) {
32970
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32971
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32972
+ tags: {
32973
+ deviceId: failure.deviceId,
32974
+ stableId: failure.stableId
32975
+ },
32976
+ meta: {
32977
+ type: failure.type,
32978
+ attempts: failure.attempts,
32979
+ error: failure.lastError
32980
+ }
32981
+ });
32982
+ }
32983
+ scheduleRestoreRetries(failures, attempt) {
32984
+ const scheduler = new DeviceRestoreRetryScheduler({
32985
+ logger: this.ctx.logger,
32986
+ delaysMs: this.restoreRetryDelaysMs,
32987
+ concurrency: this.restoreRetryConcurrency,
32988
+ attempt,
32989
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32990
+ });
32991
+ this._restoreRetryScheduler = scheduler;
32992
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32993
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32994
+ });
32995
+ }
32996
+ /**
32997
+ * Tear down and reconstruct ONE device from its persisted rows — the
32998
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32999
+ * and no other device this provider owns is disturbed.
33000
+ *
33001
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33002
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33003
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33004
+ * whatever number the row carries NOW. The teardown is `decommission` —
33005
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33006
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33007
+ * the boot restore's own `create()` path, including its pass 2: first-class
33008
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33009
+ * parent by the cascade and must be re-created explicitly, because only
33010
+ * accessory children come back through `getAccessoryChildren()`.
33011
+ *
33012
+ * Reloading an accessory child directly is refused (no device class) —
33013
+ * reload its parent instead.
33014
+ */
33015
+ async reloadDevice(input) {
33016
+ const { stableId } = input;
33017
+ const devices = this.ctx.kernel.devices;
33018
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33019
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33020
+ if (live) await devices.decommission(live.id);
33021
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33022
+ addonId: this.addonId,
33023
+ stableId
33024
+ });
33025
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33026
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33027
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33028
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33029
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33030
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33031
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33032
+ for (const row of rows) {
33033
+ if (row.parentDeviceId !== id) continue;
33034
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33035
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33036
+ if (!ChildClass) continue;
33037
+ try {
33038
+ await devices.create(row.stableId, ChildClass, {}, id);
33039
+ } catch (err) {
33040
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33041
+ tags: {
33042
+ deviceId: row.id,
33043
+ stableId: row.stableId
33044
+ },
33045
+ meta: {
33046
+ parentDeviceId: id,
33047
+ error: err instanceof Error ? err.message : String(err)
33048
+ }
33049
+ });
33050
+ }
33051
+ }
33052
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33053
+ tags: { deviceId: id },
33054
+ meta: {
33055
+ stableId,
33056
+ type: meta.type
33057
+ }
33058
+ });
33059
+ return { deviceId: id };
32658
33060
  }
32659
33061
  /**
32660
33062
  * Restore devices from persisted state. Two-pass:
@@ -32680,55 +33082,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32680
33082
  * accessory-spawn flow handles via the parent's
32681
33083
  * `getAccessoryChildren()`. Override only when the default doesn't
32682
33084
  * fit.
33085
+ *
33086
+ * A row that fails either pass is NOT terminal (D347): it is handed
33087
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33088
+ * Only after the bound is exhausted is the device marked permanently
33089
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33090
+ * `getStatus().error`.
32683
33091
  */
32684
33092
  async onRestoreDevices(savedDevices) {
32685
33093
  const restored = /* @__PURE__ */ new Set();
33094
+ const failures = [];
33095
+ const attemptRestore = async (saved) => {
33096
+ if (restored.has(saved.id)) return;
33097
+ const Class = this.deviceClasses[saved.type];
33098
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33099
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33100
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33101
+ restored.add(saved.id);
33102
+ };
32686
33103
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32687
33104
  const restoreOne = async (saved) => {
32688
- const Class = this.deviceClasses[saved.type];
32689
- if (!Class) {
33105
+ if (!this.deviceClasses[saved.type]) {
32690
33106
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32691
- tags: { stableId: saved.stableId },
33107
+ tags: {
33108
+ deviceId: saved.id,
33109
+ stableId: saved.stableId
33110
+ },
32692
33111
  meta: { type: saved.type }
32693
33112
  });
32694
33113
  return;
32695
33114
  }
32696
33115
  try {
32697
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32698
- restored.add(saved.id);
33116
+ await attemptRestore(saved);
32699
33117
  } catch (err) {
32700
- this.ctx.logger.warn("Failed to restore device", {
32701
- tags: { stableId: saved.stableId },
33118
+ const error = err instanceof Error ? err.message : String(err);
33119
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33120
+ tags: {
33121
+ deviceId: saved.id,
33122
+ stableId: saved.stableId
33123
+ },
32702
33124
  meta: {
32703
33125
  type: saved.type,
32704
- error: err instanceof Error ? err.message : String(err)
33126
+ attempt: 1,
33127
+ error
32705
33128
  }
32706
33129
  });
33130
+ failures.push({
33131
+ saved,
33132
+ error
33133
+ });
32707
33134
  }
32708
33135
  };
32709
33136
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33137
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32710
33138
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32711
33139
  for (const saved of childRows) {
32712
- const Class = this.deviceClasses[saved.type];
32713
- if (!Class) continue;
33140
+ if (!this.deviceClasses[saved.type]) continue;
32714
33141
  if (saved.parentDeviceId === null) continue;
32715
- if (!restored.has(saved.parentDeviceId)) continue;
32716
- try {
32717
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32718
- restored.add(saved.id);
32719
- } catch (err) {
32720
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33142
+ if (restored.has(saved.parentDeviceId)) {
33143
+ try {
33144
+ await attemptRestore(saved);
33145
+ } catch (err) {
33146
+ const error = err instanceof Error ? err.message : String(err);
33147
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33148
+ tags: {
33149
+ deviceId: saved.id,
33150
+ stableId: saved.stableId,
33151
+ parentDeviceId: saved.parentDeviceId
33152
+ },
33153
+ meta: {
33154
+ type: saved.type,
33155
+ attempt: 1,
33156
+ error
33157
+ }
33158
+ });
33159
+ failures.push({
33160
+ saved,
33161
+ error
33162
+ });
33163
+ }
33164
+ continue;
33165
+ }
33166
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33167
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32721
33168
  tags: {
33169
+ deviceId: saved.id,
32722
33170
  stableId: saved.stableId,
32723
33171
  parentDeviceId: saved.parentDeviceId
32724
33172
  },
32725
- meta: {
32726
- type: saved.type,
32727
- error: err instanceof Error ? err.message : String(err)
32728
- }
33173
+ meta: { type: saved.type }
33174
+ });
33175
+ failures.push({
33176
+ saved,
33177
+ error: `parent device ${saved.parentDeviceId} not restored`
32729
33178
  });
33179
+ continue;
32730
33180
  }
32731
33181
  }
33182
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33183
+ return {
33184
+ restoredCount: restored.size,
33185
+ failedCount: failures.length
33186
+ };
32732
33187
  }
32733
33188
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32734
33189
  toSummary(device) {
@@ -33237,6 +33692,12 @@ Object.freeze({
33237
33692
  addonId: null,
33238
33693
  access: "create"
33239
33694
  },
33695
+ "backup.cancel": {
33696
+ capName: "backup",
33697
+ capScope: "system",
33698
+ addonId: null,
33699
+ access: "create"
33700
+ },
33240
33701
  "backup.delete": {
33241
33702
  capName: "backup",
33242
33703
  capScope: "system",
@@ -33279,6 +33740,12 @@ Object.freeze({
33279
33740
  addonId: null,
33280
33741
  access: "view"
33281
33742
  },
33743
+ "backup.listRuns": {
33744
+ capName: "backup",
33745
+ capScope: "system",
33746
+ addonId: null,
33747
+ access: "view"
33748
+ },
33282
33749
  "backup.listSchedules": {
33283
33750
  capName: "backup",
33284
33751
  capScope: "system",
@@ -34479,6 +34946,12 @@ Object.freeze({
34479
34946
  addonId: null,
34480
34947
  access: "view"
34481
34948
  },
34949
+ "deviceProvider.reloadDevice": {
34950
+ capName: "device-provider",
34951
+ capScope: "system",
34952
+ addonId: null,
34953
+ access: "create"
34954
+ },
34482
34955
  "deviceProvider.start": {
34483
34956
  capName: "device-provider",
34484
34957
  capScope: "system",
@@ -37845,6 +38318,12 @@ Object.freeze({
37845
38318
  addonId: null,
37846
38319
  access: "create"
37847
38320
  },
38321
+ "streamBroker.forgetDeviceHardware": {
38322
+ capName: "stream-broker",
38323
+ capScope: "system",
38324
+ addonId: null,
38325
+ access: "delete"
38326
+ },
37848
38327
  "streamBroker.getAllRtspEntries": {
37849
38328
  capName: "stream-broker",
37850
38329
  capScope: "system",
@@ -40303,6 +40782,11 @@ Object.freeze({
40303
40782
  form: "single",
40304
40783
  optional: false
40305
40784
  }],
40785
+ "streamBroker.forgetDeviceHardware": [{
40786
+ name: "deviceId",
40787
+ form: "single",
40788
+ optional: false
40789
+ }],
40306
40790
  "streamBroker.getDeviceAudioMute": [{
40307
40791
  name: "deviceId",
40308
40792
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-matter-broker",
3
- "version": "0.2.57",
3
+ "version": "0.2.59",
4
4
  "description": "Matter broker addon for CamStack — owns a Matter fabric (commissioning + the long-lived controller) via the matter.js controller and brokers commissioned Matter nodes into CamStack",
5
5
  "keywords": [
6
6
  "camstack",