@camstack/addon-provider-unifi 0.2.59 → 0.2.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +391 -24
  2. package/dist/addon.mjs +391 -24
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -12812,6 +12812,35 @@ var deviceProviderCapability = {
12812
12812
  name: string(),
12813
12813
  type: string()
12814
12814
  }))),
12815
+ /**
12816
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12817
+ * touching no other device this provider owns.
12818
+ *
12819
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12820
+ * migrated numbers: after `swapIds` the runner's live instance still
12821
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12822
+ * registrations and its log tags), and a live object cannot be renumbered.
12823
+ * Before this method the only flush was restarting the whole owning addon
12824
+ * — which took every camera the provider owns down with it (28 devices
12825
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12826
+ * same day ~27 devices' native caps did not come back on their own).
12827
+ *
12828
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12829
+ * that changes. The reply carries the id the device answers on NOW.
12830
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12831
+ * instance (if any), then re-create from the persisted row: the same
12832
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12833
+ * An RPC, never an event: a dropped event would leave the runner writing
12834
+ * against the wrong camera (D8).
12835
+ *
12836
+ * Construction can dial hardware, and the migrated source is
12837
+ * characteristically dead — the timeout covers a full activate window
12838
+ * rather than the 60 s default.
12839
+ */
12840
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12841
+ kind: "mutation",
12842
+ timeoutMs: 3 * 6e4
12843
+ }),
12815
12844
  supportsDiscovery: method(object({}), boolean()),
12816
12845
  /**
12817
12846
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13139,7 +13168,8 @@ method(object({
13139
13168
  targetId: number()
13140
13169
  }), MigrateDeviceResultSchema, {
13141
13170
  kind: "mutation",
13142
- auth: "admin"
13171
+ auth: "admin",
13172
+ timeoutMs: 12 * 6e4
13143
13173
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13144
13174
  deviceId: number(),
13145
13175
  name: string()
@@ -32512,6 +32542,147 @@ var BaseDevice = class {
32512
32542
  }
32513
32543
  };
32514
32544
  /**
32545
+ * Delays before retry rounds 1..N — the round count IS the bound.
32546
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32547
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32548
+ * per attempt) covers a device-manager lock held for minutes — the
32549
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32550
+ */
32551
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32552
+ 1e4,
32553
+ 3e4,
32554
+ 9e4
32555
+ ];
32556
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32557
+ function sleep$1(ms, signal) {
32558
+ return new Promise((resolve) => {
32559
+ if (signal.aborted) {
32560
+ resolve();
32561
+ return;
32562
+ }
32563
+ const onAbort = () => {
32564
+ clearTimeout(timer);
32565
+ resolve();
32566
+ };
32567
+ const timer = setTimeout(() => {
32568
+ signal.removeEventListener("abort", onAbort);
32569
+ resolve();
32570
+ }, ms);
32571
+ timer.unref?.();
32572
+ signal.addEventListener("abort", onAbort, { once: true });
32573
+ });
32574
+ }
32575
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32576
+ * not reject (callers wrap their own try/catch). */
32577
+ async function runWithConcurrency(items, width, fn) {
32578
+ const queue = [...items];
32579
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32580
+ const lane = async () => {
32581
+ for (;;) {
32582
+ const item = queue.shift();
32583
+ if (item === void 0) return;
32584
+ await fn(item);
32585
+ }
32586
+ };
32587
+ await Promise.all(Array.from({ length: laneCount }, lane));
32588
+ }
32589
+ var DeviceRestoreRetryScheduler = class {
32590
+ #logger;
32591
+ #attempt;
32592
+ #onPermanentFailure;
32593
+ #delaysMs;
32594
+ #concurrency;
32595
+ #now;
32596
+ #abort = new AbortController();
32597
+ constructor(options) {
32598
+ this.#logger = options.logger;
32599
+ this.#attempt = options.attempt;
32600
+ this.#onPermanentFailure = options.onPermanentFailure;
32601
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32602
+ this.#concurrency = options.concurrency ?? 4;
32603
+ this.#now = options.now ?? Date.now;
32604
+ }
32605
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32606
+ * permanently failed — the next boot restores them from disk. */
32607
+ cancel() {
32608
+ this.#abort.abort();
32609
+ }
32610
+ /**
32611
+ * Run the bounded retry rounds. Resolves when every entry has either
32612
+ * restored, been marked permanently failed, or the scheduler was
32613
+ * cancelled. Never rejects.
32614
+ */
32615
+ async run(initialFailures) {
32616
+ let pending = initialFailures.map((failure) => ({
32617
+ saved: failure.saved,
32618
+ lastError: failure.error,
32619
+ attempts: 1
32620
+ }));
32621
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32622
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32623
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32624
+ if (this.#abort.signal.aborted) break;
32625
+ pending = await this.#runRound(pending, round);
32626
+ }
32627
+ if (this.#abort.signal.aborted) return [];
32628
+ const terminal = pending.map((entry) => ({
32629
+ deviceId: entry.saved.id,
32630
+ stableId: entry.saved.stableId,
32631
+ type: String(entry.saved.type),
32632
+ attempts: entry.attempts,
32633
+ lastError: entry.lastError,
32634
+ failedAt: this.#now()
32635
+ }));
32636
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32637
+ return terminal;
32638
+ }
32639
+ /** One retry round: parents first (phase 0), then hub-adopted
32640
+ * children (phase 1) — a child's attempt depends on its parent
32641
+ * having landed, exactly like the initial two-pass restore. */
32642
+ async #runRound(pending, round) {
32643
+ const next = [];
32644
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32645
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32646
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32647
+ if (this.#abort.signal.aborted) {
32648
+ next.push(entry);
32649
+ return;
32650
+ }
32651
+ const attemptNo = entry.attempts + 1;
32652
+ try {
32653
+ await this.#attempt(entry.saved);
32654
+ this.#logger.info("Device restored on retry", {
32655
+ tags: {
32656
+ deviceId: entry.saved.id,
32657
+ stableId: entry.saved.stableId
32658
+ },
32659
+ meta: { attempt: attemptNo }
32660
+ });
32661
+ } catch (err) {
32662
+ const lastError = err instanceof Error ? err.message : String(err);
32663
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32664
+ this.#logger.warn("Device restore retry failed", {
32665
+ tags: {
32666
+ deviceId: entry.saved.id,
32667
+ stableId: entry.saved.stableId
32668
+ },
32669
+ meta: {
32670
+ attempt: attemptNo,
32671
+ remainingRetries,
32672
+ error: lastError
32673
+ }
32674
+ });
32675
+ next.push({
32676
+ saved: entry.saved,
32677
+ lastError,
32678
+ attempts: attemptNo
32679
+ });
32680
+ }
32681
+ });
32682
+ return next;
32683
+ }
32684
+ };
32685
+ /**
32515
32686
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32516
32687
  * device-provider cap router. Shared across all providers.
32517
32688
  */
@@ -32560,6 +32731,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32560
32731
  }];
32561
32732
  }
32562
32733
  async onShutdown() {
32734
+ this.cancelRestoreRetries();
32563
32735
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32564
32736
  for (const device of devices) try {
32565
32737
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32577,9 +32749,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32577
32749
  async start() {}
32578
32750
  async stop() {}
32579
32751
  async getStatus() {
32752
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32753
+ const summary = this.restoreFailureSummary();
32754
+ if (summary === null) return {
32755
+ connected: true,
32756
+ deviceCount: all.length
32757
+ };
32580
32758
  return {
32581
32759
  connected: true,
32582
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32760
+ deviceCount: all.length,
32761
+ error: summary
32583
32762
  };
32584
32763
  }
32585
32764
  async getDevices() {
@@ -32669,8 +32848,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32669
32848
  };
32670
32849
  }
32671
32850
  async restoreDevices(savedDevices) {
32672
- await this.onRestoreDevices(savedDevices);
32673
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32851
+ const report = await this.onRestoreDevices(savedDevices);
32852
+ if (savedDevices.length === 0) return;
32853
+ if (report && report.failedCount > 0) {
32854
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32855
+ return;
32856
+ }
32857
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32858
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32859
+ }
32860
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32861
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32862
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32863
+ * never re-stampede full-width while the initial pass does (D167). */
32864
+ restoreRetryConcurrency = 4;
32865
+ _restoreRetryScheduler = null;
32866
+ _restoreRetryCompletion = null;
32867
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32868
+ /** Settles when the background retry rounds finish (or `null` when
32869
+ * nothing failed). Exposed for tests and subclass diagnostics —
32870
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32871
+ * with the devices that restored, and a late success is announced
32872
+ * through the `native-cap-change` → `updateCaps` path. */
32873
+ get restoreRetryCompletion() {
32874
+ return this._restoreRetryCompletion;
32875
+ }
32876
+ /** Devices that exhausted the retry bound this process lifetime. */
32877
+ get permanentRestoreFailures() {
32878
+ return [...this._permanentRestoreFailures.values()];
32879
+ }
32880
+ /** One-line operator-facing summary for `getStatus().error`, or
32881
+ * `null` when every device restored. */
32882
+ restoreFailureSummary() {
32883
+ if (this._permanentRestoreFailures.size === 0) return null;
32884
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32885
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32886
+ }
32887
+ cancelRestoreRetries() {
32888
+ this._restoreRetryScheduler?.cancel();
32889
+ this._restoreRetryScheduler = null;
32890
+ }
32891
+ recordPermanentRestoreFailure(failure) {
32892
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32893
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32894
+ tags: {
32895
+ deviceId: failure.deviceId,
32896
+ stableId: failure.stableId
32897
+ },
32898
+ meta: {
32899
+ type: failure.type,
32900
+ attempts: failure.attempts,
32901
+ error: failure.lastError
32902
+ }
32903
+ });
32904
+ }
32905
+ scheduleRestoreRetries(failures, attempt) {
32906
+ const scheduler = new DeviceRestoreRetryScheduler({
32907
+ logger: this.ctx.logger,
32908
+ delaysMs: this.restoreRetryDelaysMs,
32909
+ concurrency: this.restoreRetryConcurrency,
32910
+ attempt,
32911
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32912
+ });
32913
+ this._restoreRetryScheduler = scheduler;
32914
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32915
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32916
+ });
32917
+ }
32918
+ /**
32919
+ * Tear down and reconstruct ONE device from its persisted rows — the
32920
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32921
+ * and no other device this provider owns is disturbed.
32922
+ *
32923
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32924
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32925
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32926
+ * whatever number the row carries NOW. The teardown is `decommission` —
32927
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32928
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32929
+ * the boot restore's own `create()` path, including its pass 2: first-class
32930
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32931
+ * parent by the cascade and must be re-created explicitly, because only
32932
+ * accessory children come back through `getAccessoryChildren()`.
32933
+ *
32934
+ * Reloading an accessory child directly is refused (no device class) —
32935
+ * reload its parent instead.
32936
+ */
32937
+ async reloadDevice(input) {
32938
+ const { stableId } = input;
32939
+ const devices = this.ctx.kernel.devices;
32940
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32941
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32942
+ if (live) await devices.decommission(live.id);
32943
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32944
+ addonId: this.addonId,
32945
+ stableId
32946
+ });
32947
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32948
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32949
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32950
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32951
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32952
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32953
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32954
+ for (const row of rows) {
32955
+ if (row.parentDeviceId !== id) continue;
32956
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32957
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32958
+ if (!ChildClass) continue;
32959
+ try {
32960
+ await devices.create(row.stableId, ChildClass, {}, id);
32961
+ } catch (err) {
32962
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32963
+ tags: {
32964
+ deviceId: row.id,
32965
+ stableId: row.stableId
32966
+ },
32967
+ meta: {
32968
+ parentDeviceId: id,
32969
+ error: err instanceof Error ? err.message : String(err)
32970
+ }
32971
+ });
32972
+ }
32973
+ }
32974
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32975
+ tags: { deviceId: id },
32976
+ meta: {
32977
+ stableId,
32978
+ type: meta.type
32979
+ }
32980
+ });
32981
+ return { deviceId: id };
32674
32982
  }
32675
32983
  /**
32676
32984
  * Restore devices from persisted state. Two-pass:
@@ -32696,55 +33004,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32696
33004
  * accessory-spawn flow handles via the parent's
32697
33005
  * `getAccessoryChildren()`. Override only when the default doesn't
32698
33006
  * fit.
33007
+ *
33008
+ * A row that fails either pass is NOT terminal (D347): it is handed
33009
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33010
+ * Only after the bound is exhausted is the device marked permanently
33011
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33012
+ * `getStatus().error`.
32699
33013
  */
32700
33014
  async onRestoreDevices(savedDevices) {
32701
33015
  const restored = /* @__PURE__ */ new Set();
33016
+ const failures = [];
33017
+ const attemptRestore = async (saved) => {
33018
+ if (restored.has(saved.id)) return;
33019
+ const Class = this.deviceClasses[saved.type];
33020
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33021
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33022
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33023
+ restored.add(saved.id);
33024
+ };
32702
33025
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32703
33026
  const restoreOne = async (saved) => {
32704
- const Class = this.deviceClasses[saved.type];
32705
- if (!Class) {
33027
+ if (!this.deviceClasses[saved.type]) {
32706
33028
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32707
- tags: { stableId: saved.stableId },
33029
+ tags: {
33030
+ deviceId: saved.id,
33031
+ stableId: saved.stableId
33032
+ },
32708
33033
  meta: { type: saved.type }
32709
33034
  });
32710
33035
  return;
32711
33036
  }
32712
33037
  try {
32713
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32714
- restored.add(saved.id);
33038
+ await attemptRestore(saved);
32715
33039
  } catch (err) {
32716
- this.ctx.logger.warn("Failed to restore device", {
32717
- tags: { stableId: saved.stableId },
33040
+ const error = err instanceof Error ? err.message : String(err);
33041
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33042
+ tags: {
33043
+ deviceId: saved.id,
33044
+ stableId: saved.stableId
33045
+ },
32718
33046
  meta: {
32719
33047
  type: saved.type,
32720
- error: err instanceof Error ? err.message : String(err)
33048
+ attempt: 1,
33049
+ error
32721
33050
  }
32722
33051
  });
33052
+ failures.push({
33053
+ saved,
33054
+ error
33055
+ });
32723
33056
  }
32724
33057
  };
32725
33058
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33059
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32726
33060
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32727
33061
  for (const saved of childRows) {
32728
- const Class = this.deviceClasses[saved.type];
32729
- if (!Class) continue;
33062
+ if (!this.deviceClasses[saved.type]) continue;
32730
33063
  if (saved.parentDeviceId === null) continue;
32731
- if (!restored.has(saved.parentDeviceId)) continue;
32732
- try {
32733
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32734
- restored.add(saved.id);
32735
- } catch (err) {
32736
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33064
+ if (restored.has(saved.parentDeviceId)) {
33065
+ try {
33066
+ await attemptRestore(saved);
33067
+ } catch (err) {
33068
+ const error = err instanceof Error ? err.message : String(err);
33069
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33070
+ tags: {
33071
+ deviceId: saved.id,
33072
+ stableId: saved.stableId,
33073
+ parentDeviceId: saved.parentDeviceId
33074
+ },
33075
+ meta: {
33076
+ type: saved.type,
33077
+ attempt: 1,
33078
+ error
33079
+ }
33080
+ });
33081
+ failures.push({
33082
+ saved,
33083
+ error
33084
+ });
33085
+ }
33086
+ continue;
33087
+ }
33088
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33089
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32737
33090
  tags: {
33091
+ deviceId: saved.id,
32738
33092
  stableId: saved.stableId,
32739
33093
  parentDeviceId: saved.parentDeviceId
32740
33094
  },
32741
- meta: {
32742
- type: saved.type,
32743
- error: err instanceof Error ? err.message : String(err)
32744
- }
33095
+ meta: { type: saved.type }
32745
33096
  });
33097
+ failures.push({
33098
+ saved,
33099
+ error: `parent device ${saved.parentDeviceId} not restored`
33100
+ });
33101
+ continue;
32746
33102
  }
32747
33103
  }
33104
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33105
+ return {
33106
+ restoredCount: restored.size,
33107
+ failedCount: failures.length
33108
+ };
32748
33109
  }
32749
33110
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32750
33111
  toSummary(device) {
@@ -34507,6 +34868,12 @@ Object.freeze({
34507
34868
  addonId: null,
34508
34869
  access: "view"
34509
34870
  },
34871
+ "deviceProvider.reloadDevice": {
34872
+ capName: "device-provider",
34873
+ capScope: "system",
34874
+ addonId: null,
34875
+ access: "create"
34876
+ },
34510
34877
  "deviceProvider.start": {
34511
34878
  capName: "device-provider",
34512
34879
  capScope: "system",
package/dist/addon.mjs CHANGED
@@ -12811,6 +12811,35 @@ var deviceProviderCapability = {
12811
12811
  name: string(),
12812
12812
  type: string()
12813
12813
  }))),
12814
+ /**
12815
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12816
+ * touching no other device this provider owns.
12817
+ *
12818
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12819
+ * migrated numbers: after `swapIds` the runner's live instance still
12820
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12821
+ * registrations and its log tags), and a live object cannot be renumbered.
12822
+ * Before this method the only flush was restarting the whole owning addon
12823
+ * — which took every camera the provider owns down with it (28 devices
12824
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12825
+ * same day ~27 devices' native caps did not come back on their own).
12826
+ *
12827
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12828
+ * that changes. The reply carries the id the device answers on NOW.
12829
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12830
+ * instance (if any), then re-create from the persisted row: the same
12831
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12832
+ * An RPC, never an event: a dropped event would leave the runner writing
12833
+ * against the wrong camera (D8).
12834
+ *
12835
+ * Construction can dial hardware, and the migrated source is
12836
+ * characteristically dead — the timeout covers a full activate window
12837
+ * rather than the 60 s default.
12838
+ */
12839
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12840
+ kind: "mutation",
12841
+ timeoutMs: 3 * 6e4
12842
+ }),
12814
12843
  supportsDiscovery: method(object({}), boolean()),
12815
12844
  /**
12816
12845
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13138,7 +13167,8 @@ method(object({
13138
13167
  targetId: number()
13139
13168
  }), MigrateDeviceResultSchema, {
13140
13169
  kind: "mutation",
13141
- auth: "admin"
13170
+ auth: "admin",
13171
+ timeoutMs: 12 * 6e4
13142
13172
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13143
13173
  deviceId: number(),
13144
13174
  name: string()
@@ -32511,6 +32541,147 @@ var BaseDevice = class {
32511
32541
  }
32512
32542
  };
32513
32543
  /**
32544
+ * Delays before retry rounds 1..N — the round count IS the bound.
32545
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32546
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32547
+ * per attempt) covers a device-manager lock held for minutes — the
32548
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32549
+ */
32550
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32551
+ 1e4,
32552
+ 3e4,
32553
+ 9e4
32554
+ ];
32555
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32556
+ function sleep$1(ms, signal) {
32557
+ return new Promise((resolve) => {
32558
+ if (signal.aborted) {
32559
+ resolve();
32560
+ return;
32561
+ }
32562
+ const onAbort = () => {
32563
+ clearTimeout(timer);
32564
+ resolve();
32565
+ };
32566
+ const timer = setTimeout(() => {
32567
+ signal.removeEventListener("abort", onAbort);
32568
+ resolve();
32569
+ }, ms);
32570
+ timer.unref?.();
32571
+ signal.addEventListener("abort", onAbort, { once: true });
32572
+ });
32573
+ }
32574
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32575
+ * not reject (callers wrap their own try/catch). */
32576
+ async function runWithConcurrency(items, width, fn) {
32577
+ const queue = [...items];
32578
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32579
+ const lane = async () => {
32580
+ for (;;) {
32581
+ const item = queue.shift();
32582
+ if (item === void 0) return;
32583
+ await fn(item);
32584
+ }
32585
+ };
32586
+ await Promise.all(Array.from({ length: laneCount }, lane));
32587
+ }
32588
+ var DeviceRestoreRetryScheduler = class {
32589
+ #logger;
32590
+ #attempt;
32591
+ #onPermanentFailure;
32592
+ #delaysMs;
32593
+ #concurrency;
32594
+ #now;
32595
+ #abort = new AbortController();
32596
+ constructor(options) {
32597
+ this.#logger = options.logger;
32598
+ this.#attempt = options.attempt;
32599
+ this.#onPermanentFailure = options.onPermanentFailure;
32600
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32601
+ this.#concurrency = options.concurrency ?? 4;
32602
+ this.#now = options.now ?? Date.now;
32603
+ }
32604
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32605
+ * permanently failed — the next boot restores them from disk. */
32606
+ cancel() {
32607
+ this.#abort.abort();
32608
+ }
32609
+ /**
32610
+ * Run the bounded retry rounds. Resolves when every entry has either
32611
+ * restored, been marked permanently failed, or the scheduler was
32612
+ * cancelled. Never rejects.
32613
+ */
32614
+ async run(initialFailures) {
32615
+ let pending = initialFailures.map((failure) => ({
32616
+ saved: failure.saved,
32617
+ lastError: failure.error,
32618
+ attempts: 1
32619
+ }));
32620
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32621
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32622
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32623
+ if (this.#abort.signal.aborted) break;
32624
+ pending = await this.#runRound(pending, round);
32625
+ }
32626
+ if (this.#abort.signal.aborted) return [];
32627
+ const terminal = pending.map((entry) => ({
32628
+ deviceId: entry.saved.id,
32629
+ stableId: entry.saved.stableId,
32630
+ type: String(entry.saved.type),
32631
+ attempts: entry.attempts,
32632
+ lastError: entry.lastError,
32633
+ failedAt: this.#now()
32634
+ }));
32635
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32636
+ return terminal;
32637
+ }
32638
+ /** One retry round: parents first (phase 0), then hub-adopted
32639
+ * children (phase 1) — a child's attempt depends on its parent
32640
+ * having landed, exactly like the initial two-pass restore. */
32641
+ async #runRound(pending, round) {
32642
+ const next = [];
32643
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32644
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32645
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32646
+ if (this.#abort.signal.aborted) {
32647
+ next.push(entry);
32648
+ return;
32649
+ }
32650
+ const attemptNo = entry.attempts + 1;
32651
+ try {
32652
+ await this.#attempt(entry.saved);
32653
+ this.#logger.info("Device restored on retry", {
32654
+ tags: {
32655
+ deviceId: entry.saved.id,
32656
+ stableId: entry.saved.stableId
32657
+ },
32658
+ meta: { attempt: attemptNo }
32659
+ });
32660
+ } catch (err) {
32661
+ const lastError = err instanceof Error ? err.message : String(err);
32662
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32663
+ this.#logger.warn("Device restore retry failed", {
32664
+ tags: {
32665
+ deviceId: entry.saved.id,
32666
+ stableId: entry.saved.stableId
32667
+ },
32668
+ meta: {
32669
+ attempt: attemptNo,
32670
+ remainingRetries,
32671
+ error: lastError
32672
+ }
32673
+ });
32674
+ next.push({
32675
+ saved: entry.saved,
32676
+ lastError,
32677
+ attempts: attemptNo
32678
+ });
32679
+ }
32680
+ });
32681
+ return next;
32682
+ }
32683
+ };
32684
+ /**
32514
32685
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32515
32686
  * device-provider cap router. Shared across all providers.
32516
32687
  */
@@ -32559,6 +32730,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32559
32730
  }];
32560
32731
  }
32561
32732
  async onShutdown() {
32733
+ this.cancelRestoreRetries();
32562
32734
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32563
32735
  for (const device of devices) try {
32564
32736
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32576,9 +32748,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32576
32748
  async start() {}
32577
32749
  async stop() {}
32578
32750
  async getStatus() {
32751
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32752
+ const summary = this.restoreFailureSummary();
32753
+ if (summary === null) return {
32754
+ connected: true,
32755
+ deviceCount: all.length
32756
+ };
32579
32757
  return {
32580
32758
  connected: true,
32581
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32759
+ deviceCount: all.length,
32760
+ error: summary
32582
32761
  };
32583
32762
  }
32584
32763
  async getDevices() {
@@ -32668,8 +32847,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32668
32847
  };
32669
32848
  }
32670
32849
  async restoreDevices(savedDevices) {
32671
- await this.onRestoreDevices(savedDevices);
32672
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32850
+ const report = await this.onRestoreDevices(savedDevices);
32851
+ if (savedDevices.length === 0) return;
32852
+ if (report && report.failedCount > 0) {
32853
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32854
+ return;
32855
+ }
32856
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32857
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32858
+ }
32859
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32860
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32861
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32862
+ * never re-stampede full-width while the initial pass does (D167). */
32863
+ restoreRetryConcurrency = 4;
32864
+ _restoreRetryScheduler = null;
32865
+ _restoreRetryCompletion = null;
32866
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32867
+ /** Settles when the background retry rounds finish (or `null` when
32868
+ * nothing failed). Exposed for tests and subclass diagnostics —
32869
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32870
+ * with the devices that restored, and a late success is announced
32871
+ * through the `native-cap-change` → `updateCaps` path. */
32872
+ get restoreRetryCompletion() {
32873
+ return this._restoreRetryCompletion;
32874
+ }
32875
+ /** Devices that exhausted the retry bound this process lifetime. */
32876
+ get permanentRestoreFailures() {
32877
+ return [...this._permanentRestoreFailures.values()];
32878
+ }
32879
+ /** One-line operator-facing summary for `getStatus().error`, or
32880
+ * `null` when every device restored. */
32881
+ restoreFailureSummary() {
32882
+ if (this._permanentRestoreFailures.size === 0) return null;
32883
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32884
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32885
+ }
32886
+ cancelRestoreRetries() {
32887
+ this._restoreRetryScheduler?.cancel();
32888
+ this._restoreRetryScheduler = null;
32889
+ }
32890
+ recordPermanentRestoreFailure(failure) {
32891
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32892
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32893
+ tags: {
32894
+ deviceId: failure.deviceId,
32895
+ stableId: failure.stableId
32896
+ },
32897
+ meta: {
32898
+ type: failure.type,
32899
+ attempts: failure.attempts,
32900
+ error: failure.lastError
32901
+ }
32902
+ });
32903
+ }
32904
+ scheduleRestoreRetries(failures, attempt) {
32905
+ const scheduler = new DeviceRestoreRetryScheduler({
32906
+ logger: this.ctx.logger,
32907
+ delaysMs: this.restoreRetryDelaysMs,
32908
+ concurrency: this.restoreRetryConcurrency,
32909
+ attempt,
32910
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32911
+ });
32912
+ this._restoreRetryScheduler = scheduler;
32913
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32914
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32915
+ });
32916
+ }
32917
+ /**
32918
+ * Tear down and reconstruct ONE device from its persisted rows — the
32919
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32920
+ * and no other device this provider owns is disturbed.
32921
+ *
32922
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32923
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32924
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32925
+ * whatever number the row carries NOW. The teardown is `decommission` —
32926
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32927
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32928
+ * the boot restore's own `create()` path, including its pass 2: first-class
32929
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32930
+ * parent by the cascade and must be re-created explicitly, because only
32931
+ * accessory children come back through `getAccessoryChildren()`.
32932
+ *
32933
+ * Reloading an accessory child directly is refused (no device class) —
32934
+ * reload its parent instead.
32935
+ */
32936
+ async reloadDevice(input) {
32937
+ const { stableId } = input;
32938
+ const devices = this.ctx.kernel.devices;
32939
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32940
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32941
+ if (live) await devices.decommission(live.id);
32942
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32943
+ addonId: this.addonId,
32944
+ stableId
32945
+ });
32946
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32947
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32948
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32949
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32950
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32951
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32952
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32953
+ for (const row of rows) {
32954
+ if (row.parentDeviceId !== id) continue;
32955
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32956
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32957
+ if (!ChildClass) continue;
32958
+ try {
32959
+ await devices.create(row.stableId, ChildClass, {}, id);
32960
+ } catch (err) {
32961
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32962
+ tags: {
32963
+ deviceId: row.id,
32964
+ stableId: row.stableId
32965
+ },
32966
+ meta: {
32967
+ parentDeviceId: id,
32968
+ error: err instanceof Error ? err.message : String(err)
32969
+ }
32970
+ });
32971
+ }
32972
+ }
32973
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32974
+ tags: { deviceId: id },
32975
+ meta: {
32976
+ stableId,
32977
+ type: meta.type
32978
+ }
32979
+ });
32980
+ return { deviceId: id };
32673
32981
  }
32674
32982
  /**
32675
32983
  * Restore devices from persisted state. Two-pass:
@@ -32695,55 +33003,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32695
33003
  * accessory-spawn flow handles via the parent's
32696
33004
  * `getAccessoryChildren()`. Override only when the default doesn't
32697
33005
  * fit.
33006
+ *
33007
+ * A row that fails either pass is NOT terminal (D347): it is handed
33008
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33009
+ * Only after the bound is exhausted is the device marked permanently
33010
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33011
+ * `getStatus().error`.
32698
33012
  */
32699
33013
  async onRestoreDevices(savedDevices) {
32700
33014
  const restored = /* @__PURE__ */ new Set();
33015
+ const failures = [];
33016
+ const attemptRestore = async (saved) => {
33017
+ if (restored.has(saved.id)) return;
33018
+ const Class = this.deviceClasses[saved.type];
33019
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33020
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33021
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33022
+ restored.add(saved.id);
33023
+ };
32701
33024
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32702
33025
  const restoreOne = async (saved) => {
32703
- const Class = this.deviceClasses[saved.type];
32704
- if (!Class) {
33026
+ if (!this.deviceClasses[saved.type]) {
32705
33027
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32706
- tags: { stableId: saved.stableId },
33028
+ tags: {
33029
+ deviceId: saved.id,
33030
+ stableId: saved.stableId
33031
+ },
32707
33032
  meta: { type: saved.type }
32708
33033
  });
32709
33034
  return;
32710
33035
  }
32711
33036
  try {
32712
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32713
- restored.add(saved.id);
33037
+ await attemptRestore(saved);
32714
33038
  } catch (err) {
32715
- this.ctx.logger.warn("Failed to restore device", {
32716
- tags: { stableId: saved.stableId },
33039
+ const error = err instanceof Error ? err.message : String(err);
33040
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33041
+ tags: {
33042
+ deviceId: saved.id,
33043
+ stableId: saved.stableId
33044
+ },
32717
33045
  meta: {
32718
33046
  type: saved.type,
32719
- error: err instanceof Error ? err.message : String(err)
33047
+ attempt: 1,
33048
+ error
32720
33049
  }
32721
33050
  });
33051
+ failures.push({
33052
+ saved,
33053
+ error
33054
+ });
32722
33055
  }
32723
33056
  };
32724
33057
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33058
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32725
33059
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32726
33060
  for (const saved of childRows) {
32727
- const Class = this.deviceClasses[saved.type];
32728
- if (!Class) continue;
33061
+ if (!this.deviceClasses[saved.type]) continue;
32729
33062
  if (saved.parentDeviceId === null) continue;
32730
- if (!restored.has(saved.parentDeviceId)) continue;
32731
- try {
32732
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32733
- restored.add(saved.id);
32734
- } catch (err) {
32735
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33063
+ if (restored.has(saved.parentDeviceId)) {
33064
+ try {
33065
+ await attemptRestore(saved);
33066
+ } catch (err) {
33067
+ const error = err instanceof Error ? err.message : String(err);
33068
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33069
+ tags: {
33070
+ deviceId: saved.id,
33071
+ stableId: saved.stableId,
33072
+ parentDeviceId: saved.parentDeviceId
33073
+ },
33074
+ meta: {
33075
+ type: saved.type,
33076
+ attempt: 1,
33077
+ error
33078
+ }
33079
+ });
33080
+ failures.push({
33081
+ saved,
33082
+ error
33083
+ });
33084
+ }
33085
+ continue;
33086
+ }
33087
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33088
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32736
33089
  tags: {
33090
+ deviceId: saved.id,
32737
33091
  stableId: saved.stableId,
32738
33092
  parentDeviceId: saved.parentDeviceId
32739
33093
  },
32740
- meta: {
32741
- type: saved.type,
32742
- error: err instanceof Error ? err.message : String(err)
32743
- }
33094
+ meta: { type: saved.type }
32744
33095
  });
33096
+ failures.push({
33097
+ saved,
33098
+ error: `parent device ${saved.parentDeviceId} not restored`
33099
+ });
33100
+ continue;
32745
33101
  }
32746
33102
  }
33103
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33104
+ return {
33105
+ restoredCount: restored.size,
33106
+ failedCount: failures.length
33107
+ };
32747
33108
  }
32748
33109
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32749
33110
  toSummary(device) {
@@ -34506,6 +34867,12 @@ Object.freeze({
34506
34867
  addonId: null,
34507
34868
  access: "view"
34508
34869
  },
34870
+ "deviceProvider.reloadDevice": {
34871
+ capName: "device-provider",
34872
+ capScope: "system",
34873
+ addonId: null,
34874
+ access: "create"
34875
+ },
34509
34876
  "deviceProvider.start": {
34510
34877
  capName: "device-provider",
34511
34878
  capScope: "system",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-unifi",
3
- "version": "0.2.59",
3
+ "version": "0.2.60",
4
4
  "description": "UniFi Network controller device-provider addon for CamStack — local-controller infra switches/APs (as containers) + network-client presence. NO cameras/Protect.",
5
5
  "keywords": [
6
6
  "camstack",