@camstack/addon-provider-rademacher 0.2.58 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +391 -24
  2. package/dist/addon.mjs +391 -24
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -13797,6 +13797,35 @@ var deviceProviderCapability = {
13797
13797
  name: string(),
13798
13798
  type: string()
13799
13799
  }))),
13800
+ /**
13801
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13802
+ * touching no other device this provider owns.
13803
+ *
13804
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13805
+ * migrated numbers: after `swapIds` the runner's live instance still
13806
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13807
+ * registrations and its log tags), and a live object cannot be renumbered.
13808
+ * Before this method the only flush was restarting the whole owning addon
13809
+ * — which took every camera the provider owns down with it (28 devices
13810
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13811
+ * same day ~27 devices' native caps did not come back on their own).
13812
+ *
13813
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13814
+ * that changes. The reply carries the id the device answers on NOW.
13815
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13816
+ * instance (if any), then re-create from the persisted row: the same
13817
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13818
+ * An RPC, never an event: a dropped event would leave the runner writing
13819
+ * against the wrong camera (D8).
13820
+ *
13821
+ * Construction can dial hardware, and the migrated source is
13822
+ * characteristically dead — the timeout covers a full activate window
13823
+ * rather than the 60 s default.
13824
+ */
13825
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13826
+ kind: "mutation",
13827
+ timeoutMs: 3 * 6e4
13828
+ }),
13800
13829
  supportsDiscovery: method(object({}), boolean()),
13801
13830
  /**
13802
13831
  * Run a network scan. `params` carries optional provider-specific scan
@@ -14124,7 +14153,8 @@ method(object({
14124
14153
  targetId: number()
14125
14154
  }), MigrateDeviceResultSchema, {
14126
14155
  kind: "mutation",
14127
- auth: "admin"
14156
+ auth: "admin",
14157
+ timeoutMs: 12 * 6e4
14128
14158
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
14129
14159
  deviceId: number(),
14130
14160
  name: string()
@@ -33480,6 +33510,147 @@ var BaseDevice = class {
33480
33510
  }
33481
33511
  };
33482
33512
  /**
33513
+ * Delays before retry rounds 1..N — the round count IS the bound.
33514
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33515
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33516
+ * per attempt) covers a device-manager lock held for minutes — the
33517
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33518
+ */
33519
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33520
+ 1e4,
33521
+ 3e4,
33522
+ 9e4
33523
+ ];
33524
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33525
+ function sleep$1(ms, signal) {
33526
+ return new Promise((resolve) => {
33527
+ if (signal.aborted) {
33528
+ resolve();
33529
+ return;
33530
+ }
33531
+ const onAbort = () => {
33532
+ clearTimeout(timer);
33533
+ resolve();
33534
+ };
33535
+ const timer = setTimeout(() => {
33536
+ signal.removeEventListener("abort", onAbort);
33537
+ resolve();
33538
+ }, ms);
33539
+ timer.unref?.();
33540
+ signal.addEventListener("abort", onAbort, { once: true });
33541
+ });
33542
+ }
33543
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33544
+ * not reject (callers wrap their own try/catch). */
33545
+ async function runWithConcurrency(items, width, fn) {
33546
+ const queue = [...items];
33547
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33548
+ const lane = async () => {
33549
+ for (;;) {
33550
+ const item = queue.shift();
33551
+ if (item === void 0) return;
33552
+ await fn(item);
33553
+ }
33554
+ };
33555
+ await Promise.all(Array.from({ length: laneCount }, lane));
33556
+ }
33557
+ var DeviceRestoreRetryScheduler = class {
33558
+ #logger;
33559
+ #attempt;
33560
+ #onPermanentFailure;
33561
+ #delaysMs;
33562
+ #concurrency;
33563
+ #now;
33564
+ #abort = new AbortController();
33565
+ constructor(options) {
33566
+ this.#logger = options.logger;
33567
+ this.#attempt = options.attempt;
33568
+ this.#onPermanentFailure = options.onPermanentFailure;
33569
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33570
+ this.#concurrency = options.concurrency ?? 4;
33571
+ this.#now = options.now ?? Date.now;
33572
+ }
33573
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33574
+ * permanently failed — the next boot restores them from disk. */
33575
+ cancel() {
33576
+ this.#abort.abort();
33577
+ }
33578
+ /**
33579
+ * Run the bounded retry rounds. Resolves when every entry has either
33580
+ * restored, been marked permanently failed, or the scheduler was
33581
+ * cancelled. Never rejects.
33582
+ */
33583
+ async run(initialFailures) {
33584
+ let pending = initialFailures.map((failure) => ({
33585
+ saved: failure.saved,
33586
+ lastError: failure.error,
33587
+ attempts: 1
33588
+ }));
33589
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33590
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33591
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33592
+ if (this.#abort.signal.aborted) break;
33593
+ pending = await this.#runRound(pending, round);
33594
+ }
33595
+ if (this.#abort.signal.aborted) return [];
33596
+ const terminal = pending.map((entry) => ({
33597
+ deviceId: entry.saved.id,
33598
+ stableId: entry.saved.stableId,
33599
+ type: String(entry.saved.type),
33600
+ attempts: entry.attempts,
33601
+ lastError: entry.lastError,
33602
+ failedAt: this.#now()
33603
+ }));
33604
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33605
+ return terminal;
33606
+ }
33607
+ /** One retry round: parents first (phase 0), then hub-adopted
33608
+ * children (phase 1) — a child's attempt depends on its parent
33609
+ * having landed, exactly like the initial two-pass restore. */
33610
+ async #runRound(pending, round) {
33611
+ const next = [];
33612
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33613
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33614
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33615
+ if (this.#abort.signal.aborted) {
33616
+ next.push(entry);
33617
+ return;
33618
+ }
33619
+ const attemptNo = entry.attempts + 1;
33620
+ try {
33621
+ await this.#attempt(entry.saved);
33622
+ this.#logger.info("Device restored on retry", {
33623
+ tags: {
33624
+ deviceId: entry.saved.id,
33625
+ stableId: entry.saved.stableId
33626
+ },
33627
+ meta: { attempt: attemptNo }
33628
+ });
33629
+ } catch (err) {
33630
+ const lastError = err instanceof Error ? err.message : String(err);
33631
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33632
+ this.#logger.warn("Device restore retry failed", {
33633
+ tags: {
33634
+ deviceId: entry.saved.id,
33635
+ stableId: entry.saved.stableId
33636
+ },
33637
+ meta: {
33638
+ attempt: attemptNo,
33639
+ remainingRetries,
33640
+ error: lastError
33641
+ }
33642
+ });
33643
+ next.push({
33644
+ saved: entry.saved,
33645
+ lastError,
33646
+ attempts: attemptNo
33647
+ });
33648
+ }
33649
+ });
33650
+ return next;
33651
+ }
33652
+ };
33653
+ /**
33483
33654
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33484
33655
  * device-provider cap router. Shared across all providers.
33485
33656
  */
@@ -33528,6 +33699,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33528
33699
  }];
33529
33700
  }
33530
33701
  async onShutdown() {
33702
+ this.cancelRestoreRetries();
33531
33703
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33532
33704
  for (const device of devices) try {
33533
33705
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33545,9 +33717,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33545
33717
  async start() {}
33546
33718
  async stop() {}
33547
33719
  async getStatus() {
33720
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33721
+ const summary = this.restoreFailureSummary();
33722
+ if (summary === null) return {
33723
+ connected: true,
33724
+ deviceCount: all.length
33725
+ };
33548
33726
  return {
33549
33727
  connected: true,
33550
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33728
+ deviceCount: all.length,
33729
+ error: summary
33551
33730
  };
33552
33731
  }
33553
33732
  async getDevices() {
@@ -33637,8 +33816,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33637
33816
  };
33638
33817
  }
33639
33818
  async restoreDevices(savedDevices) {
33640
- await this.onRestoreDevices(savedDevices);
33641
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33819
+ const report = await this.onRestoreDevices(savedDevices);
33820
+ if (savedDevices.length === 0) return;
33821
+ if (report && report.failedCount > 0) {
33822
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33823
+ return;
33824
+ }
33825
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33826
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33827
+ }
33828
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33829
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33830
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33831
+ * never re-stampede full-width while the initial pass does (D167). */
33832
+ restoreRetryConcurrency = 4;
33833
+ _restoreRetryScheduler = null;
33834
+ _restoreRetryCompletion = null;
33835
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33836
+ /** Settles when the background retry rounds finish (or `null` when
33837
+ * nothing failed). Exposed for tests and subclass diagnostics —
33838
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33839
+ * with the devices that restored, and a late success is announced
33840
+ * through the `native-cap-change` → `updateCaps` path. */
33841
+ get restoreRetryCompletion() {
33842
+ return this._restoreRetryCompletion;
33843
+ }
33844
+ /** Devices that exhausted the retry bound this process lifetime. */
33845
+ get permanentRestoreFailures() {
33846
+ return [...this._permanentRestoreFailures.values()];
33847
+ }
33848
+ /** One-line operator-facing summary for `getStatus().error`, or
33849
+ * `null` when every device restored. */
33850
+ restoreFailureSummary() {
33851
+ if (this._permanentRestoreFailures.size === 0) return null;
33852
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33853
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33854
+ }
33855
+ cancelRestoreRetries() {
33856
+ this._restoreRetryScheduler?.cancel();
33857
+ this._restoreRetryScheduler = null;
33858
+ }
33859
+ recordPermanentRestoreFailure(failure) {
33860
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33861
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33862
+ tags: {
33863
+ deviceId: failure.deviceId,
33864
+ stableId: failure.stableId
33865
+ },
33866
+ meta: {
33867
+ type: failure.type,
33868
+ attempts: failure.attempts,
33869
+ error: failure.lastError
33870
+ }
33871
+ });
33872
+ }
33873
+ scheduleRestoreRetries(failures, attempt) {
33874
+ const scheduler = new DeviceRestoreRetryScheduler({
33875
+ logger: this.ctx.logger,
33876
+ delaysMs: this.restoreRetryDelaysMs,
33877
+ concurrency: this.restoreRetryConcurrency,
33878
+ attempt,
33879
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33880
+ });
33881
+ this._restoreRetryScheduler = scheduler;
33882
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33883
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33884
+ });
33885
+ }
33886
+ /**
33887
+ * Tear down and reconstruct ONE device from its persisted rows — the
33888
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33889
+ * and no other device this provider owns is disturbed.
33890
+ *
33891
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33892
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33893
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33894
+ * whatever number the row carries NOW. The teardown is `decommission` —
33895
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33896
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33897
+ * the boot restore's own `create()` path, including its pass 2: first-class
33898
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33899
+ * parent by the cascade and must be re-created explicitly, because only
33900
+ * accessory children come back through `getAccessoryChildren()`.
33901
+ *
33902
+ * Reloading an accessory child directly is refused (no device class) —
33903
+ * reload its parent instead.
33904
+ */
33905
+ async reloadDevice(input) {
33906
+ const { stableId } = input;
33907
+ const devices = this.ctx.kernel.devices;
33908
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33909
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33910
+ if (live) await devices.decommission(live.id);
33911
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33912
+ addonId: this.addonId,
33913
+ stableId
33914
+ });
33915
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33916
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33917
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33918
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33919
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33920
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33921
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33922
+ for (const row of rows) {
33923
+ if (row.parentDeviceId !== id) continue;
33924
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33925
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33926
+ if (!ChildClass) continue;
33927
+ try {
33928
+ await devices.create(row.stableId, ChildClass, {}, id);
33929
+ } catch (err) {
33930
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33931
+ tags: {
33932
+ deviceId: row.id,
33933
+ stableId: row.stableId
33934
+ },
33935
+ meta: {
33936
+ parentDeviceId: id,
33937
+ error: err instanceof Error ? err.message : String(err)
33938
+ }
33939
+ });
33940
+ }
33941
+ }
33942
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33943
+ tags: { deviceId: id },
33944
+ meta: {
33945
+ stableId,
33946
+ type: meta.type
33947
+ }
33948
+ });
33949
+ return { deviceId: id };
33642
33950
  }
33643
33951
  /**
33644
33952
  * Restore devices from persisted state. Two-pass:
@@ -33664,55 +33972,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33664
33972
  * accessory-spawn flow handles via the parent's
33665
33973
  * `getAccessoryChildren()`. Override only when the default doesn't
33666
33974
  * fit.
33975
+ *
33976
+ * A row that fails either pass is NOT terminal (D347): it is handed
33977
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33978
+ * Only after the bound is exhausted is the device marked permanently
33979
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33980
+ * `getStatus().error`.
33667
33981
  */
33668
33982
  async onRestoreDevices(savedDevices) {
33669
33983
  const restored = /* @__PURE__ */ new Set();
33984
+ const failures = [];
33985
+ const attemptRestore = async (saved) => {
33986
+ if (restored.has(saved.id)) return;
33987
+ const Class = this.deviceClasses[saved.type];
33988
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33989
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33990
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33991
+ restored.add(saved.id);
33992
+ };
33670
33993
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33671
33994
  const restoreOne = async (saved) => {
33672
- const Class = this.deviceClasses[saved.type];
33673
- if (!Class) {
33995
+ if (!this.deviceClasses[saved.type]) {
33674
33996
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33675
- tags: { stableId: saved.stableId },
33997
+ tags: {
33998
+ deviceId: saved.id,
33999
+ stableId: saved.stableId
34000
+ },
33676
34001
  meta: { type: saved.type }
33677
34002
  });
33678
34003
  return;
33679
34004
  }
33680
34005
  try {
33681
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33682
- restored.add(saved.id);
34006
+ await attemptRestore(saved);
33683
34007
  } catch (err) {
33684
- this.ctx.logger.warn("Failed to restore device", {
33685
- tags: { stableId: saved.stableId },
34008
+ const error = err instanceof Error ? err.message : String(err);
34009
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
34010
+ tags: {
34011
+ deviceId: saved.id,
34012
+ stableId: saved.stableId
34013
+ },
33686
34014
  meta: {
33687
34015
  type: saved.type,
33688
- error: err instanceof Error ? err.message : String(err)
34016
+ attempt: 1,
34017
+ error
33689
34018
  }
33690
34019
  });
34020
+ failures.push({
34021
+ saved,
34022
+ error
34023
+ });
33691
34024
  }
33692
34025
  };
33693
34026
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
34027
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33694
34028
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33695
34029
  for (const saved of childRows) {
33696
- const Class = this.deviceClasses[saved.type];
33697
- if (!Class) continue;
34030
+ if (!this.deviceClasses[saved.type]) continue;
33698
34031
  if (saved.parentDeviceId === null) continue;
33699
- if (!restored.has(saved.parentDeviceId)) continue;
33700
- try {
33701
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33702
- restored.add(saved.id);
33703
- } catch (err) {
33704
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
34032
+ if (restored.has(saved.parentDeviceId)) {
34033
+ try {
34034
+ await attemptRestore(saved);
34035
+ } catch (err) {
34036
+ const error = err instanceof Error ? err.message : String(err);
34037
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
34038
+ tags: {
34039
+ deviceId: saved.id,
34040
+ stableId: saved.stableId,
34041
+ parentDeviceId: saved.parentDeviceId
34042
+ },
34043
+ meta: {
34044
+ type: saved.type,
34045
+ attempt: 1,
34046
+ error
34047
+ }
34048
+ });
34049
+ failures.push({
34050
+ saved,
34051
+ error
34052
+ });
34053
+ }
34054
+ continue;
34055
+ }
34056
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
34057
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33705
34058
  tags: {
34059
+ deviceId: saved.id,
33706
34060
  stableId: saved.stableId,
33707
34061
  parentDeviceId: saved.parentDeviceId
33708
34062
  },
33709
- meta: {
33710
- type: saved.type,
33711
- error: err instanceof Error ? err.message : String(err)
33712
- }
34063
+ meta: { type: saved.type }
33713
34064
  });
34065
+ failures.push({
34066
+ saved,
34067
+ error: `parent device ${saved.parentDeviceId} not restored`
34068
+ });
34069
+ continue;
33714
34070
  }
33715
34071
  }
34072
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
34073
+ return {
34074
+ restoredCount: restored.size,
34075
+ failedCount: failures.length
34076
+ };
33716
34077
  }
33717
34078
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33718
34079
  toSummary(device) {
@@ -35475,6 +35836,12 @@ Object.freeze({
35475
35836
  addonId: null,
35476
35837
  access: "view"
35477
35838
  },
35839
+ "deviceProvider.reloadDevice": {
35840
+ capName: "device-provider",
35841
+ capScope: "system",
35842
+ addonId: null,
35843
+ access: "create"
35844
+ },
35478
35845
  "deviceProvider.start": {
35479
35846
  capName: "device-provider",
35480
35847
  capScope: "system",
package/dist/addon.mjs CHANGED
@@ -13796,6 +13796,35 @@ var deviceProviderCapability = {
13796
13796
  name: string(),
13797
13797
  type: string()
13798
13798
  }))),
13799
+ /**
13800
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13801
+ * touching no other device this provider owns.
13802
+ *
13803
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13804
+ * migrated numbers: after `swapIds` the runner's live instance still
13805
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13806
+ * registrations and its log tags), and a live object cannot be renumbered.
13807
+ * Before this method the only flush was restarting the whole owning addon
13808
+ * — which took every camera the provider owns down with it (28 devices
13809
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13810
+ * same day ~27 devices' native caps did not come back on their own).
13811
+ *
13812
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13813
+ * that changes. The reply carries the id the device answers on NOW.
13814
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13815
+ * instance (if any), then re-create from the persisted row: the same
13816
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13817
+ * An RPC, never an event: a dropped event would leave the runner writing
13818
+ * against the wrong camera (D8).
13819
+ *
13820
+ * Construction can dial hardware, and the migrated source is
13821
+ * characteristically dead — the timeout covers a full activate window
13822
+ * rather than the 60 s default.
13823
+ */
13824
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13825
+ kind: "mutation",
13826
+ timeoutMs: 3 * 6e4
13827
+ }),
13799
13828
  supportsDiscovery: method(object({}), boolean()),
13800
13829
  /**
13801
13830
  * Run a network scan. `params` carries optional provider-specific scan
@@ -14123,7 +14152,8 @@ method(object({
14123
14152
  targetId: number()
14124
14153
  }), MigrateDeviceResultSchema, {
14125
14154
  kind: "mutation",
14126
- auth: "admin"
14155
+ auth: "admin",
14156
+ timeoutMs: 12 * 6e4
14127
14157
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
14128
14158
  deviceId: number(),
14129
14159
  name: string()
@@ -33479,6 +33509,147 @@ var BaseDevice = class {
33479
33509
  }
33480
33510
  };
33481
33511
  /**
33512
+ * Delays before retry rounds 1..N — the round count IS the bound.
33513
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33514
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33515
+ * per attempt) covers a device-manager lock held for minutes — the
33516
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33517
+ */
33518
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33519
+ 1e4,
33520
+ 3e4,
33521
+ 9e4
33522
+ ];
33523
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33524
+ function sleep$1(ms, signal) {
33525
+ return new Promise((resolve) => {
33526
+ if (signal.aborted) {
33527
+ resolve();
33528
+ return;
33529
+ }
33530
+ const onAbort = () => {
33531
+ clearTimeout(timer);
33532
+ resolve();
33533
+ };
33534
+ const timer = setTimeout(() => {
33535
+ signal.removeEventListener("abort", onAbort);
33536
+ resolve();
33537
+ }, ms);
33538
+ timer.unref?.();
33539
+ signal.addEventListener("abort", onAbort, { once: true });
33540
+ });
33541
+ }
33542
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33543
+ * not reject (callers wrap their own try/catch). */
33544
+ async function runWithConcurrency(items, width, fn) {
33545
+ const queue = [...items];
33546
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33547
+ const lane = async () => {
33548
+ for (;;) {
33549
+ const item = queue.shift();
33550
+ if (item === void 0) return;
33551
+ await fn(item);
33552
+ }
33553
+ };
33554
+ await Promise.all(Array.from({ length: laneCount }, lane));
33555
+ }
33556
+ var DeviceRestoreRetryScheduler = class {
33557
+ #logger;
33558
+ #attempt;
33559
+ #onPermanentFailure;
33560
+ #delaysMs;
33561
+ #concurrency;
33562
+ #now;
33563
+ #abort = new AbortController();
33564
+ constructor(options) {
33565
+ this.#logger = options.logger;
33566
+ this.#attempt = options.attempt;
33567
+ this.#onPermanentFailure = options.onPermanentFailure;
33568
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33569
+ this.#concurrency = options.concurrency ?? 4;
33570
+ this.#now = options.now ?? Date.now;
33571
+ }
33572
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33573
+ * permanently failed — the next boot restores them from disk. */
33574
+ cancel() {
33575
+ this.#abort.abort();
33576
+ }
33577
+ /**
33578
+ * Run the bounded retry rounds. Resolves when every entry has either
33579
+ * restored, been marked permanently failed, or the scheduler was
33580
+ * cancelled. Never rejects.
33581
+ */
33582
+ async run(initialFailures) {
33583
+ let pending = initialFailures.map((failure) => ({
33584
+ saved: failure.saved,
33585
+ lastError: failure.error,
33586
+ attempts: 1
33587
+ }));
33588
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33589
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33590
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33591
+ if (this.#abort.signal.aborted) break;
33592
+ pending = await this.#runRound(pending, round);
33593
+ }
33594
+ if (this.#abort.signal.aborted) return [];
33595
+ const terminal = pending.map((entry) => ({
33596
+ deviceId: entry.saved.id,
33597
+ stableId: entry.saved.stableId,
33598
+ type: String(entry.saved.type),
33599
+ attempts: entry.attempts,
33600
+ lastError: entry.lastError,
33601
+ failedAt: this.#now()
33602
+ }));
33603
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33604
+ return terminal;
33605
+ }
33606
+ /** One retry round: parents first (phase 0), then hub-adopted
33607
+ * children (phase 1) — a child's attempt depends on its parent
33608
+ * having landed, exactly like the initial two-pass restore. */
33609
+ async #runRound(pending, round) {
33610
+ const next = [];
33611
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33612
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33613
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33614
+ if (this.#abort.signal.aborted) {
33615
+ next.push(entry);
33616
+ return;
33617
+ }
33618
+ const attemptNo = entry.attempts + 1;
33619
+ try {
33620
+ await this.#attempt(entry.saved);
33621
+ this.#logger.info("Device restored on retry", {
33622
+ tags: {
33623
+ deviceId: entry.saved.id,
33624
+ stableId: entry.saved.stableId
33625
+ },
33626
+ meta: { attempt: attemptNo }
33627
+ });
33628
+ } catch (err) {
33629
+ const lastError = err instanceof Error ? err.message : String(err);
33630
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33631
+ this.#logger.warn("Device restore retry failed", {
33632
+ tags: {
33633
+ deviceId: entry.saved.id,
33634
+ stableId: entry.saved.stableId
33635
+ },
33636
+ meta: {
33637
+ attempt: attemptNo,
33638
+ remainingRetries,
33639
+ error: lastError
33640
+ }
33641
+ });
33642
+ next.push({
33643
+ saved: entry.saved,
33644
+ lastError,
33645
+ attempts: attemptNo
33646
+ });
33647
+ }
33648
+ });
33649
+ return next;
33650
+ }
33651
+ };
33652
+ /**
33482
33653
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33483
33654
  * device-provider cap router. Shared across all providers.
33484
33655
  */
@@ -33527,6 +33698,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33527
33698
  }];
33528
33699
  }
33529
33700
  async onShutdown() {
33701
+ this.cancelRestoreRetries();
33530
33702
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33531
33703
  for (const device of devices) try {
33532
33704
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33544,9 +33716,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33544
33716
  async start() {}
33545
33717
  async stop() {}
33546
33718
  async getStatus() {
33719
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33720
+ const summary = this.restoreFailureSummary();
33721
+ if (summary === null) return {
33722
+ connected: true,
33723
+ deviceCount: all.length
33724
+ };
33547
33725
  return {
33548
33726
  connected: true,
33549
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33727
+ deviceCount: all.length,
33728
+ error: summary
33550
33729
  };
33551
33730
  }
33552
33731
  async getDevices() {
@@ -33636,8 +33815,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33636
33815
  };
33637
33816
  }
33638
33817
  async restoreDevices(savedDevices) {
33639
- await this.onRestoreDevices(savedDevices);
33640
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33818
+ const report = await this.onRestoreDevices(savedDevices);
33819
+ if (savedDevices.length === 0) return;
33820
+ if (report && report.failedCount > 0) {
33821
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33822
+ return;
33823
+ }
33824
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33825
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33826
+ }
33827
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33828
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33829
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33830
+ * never re-stampede full-width while the initial pass does (D167). */
33831
+ restoreRetryConcurrency = 4;
33832
+ _restoreRetryScheduler = null;
33833
+ _restoreRetryCompletion = null;
33834
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33835
+ /** Settles when the background retry rounds finish (or `null` when
33836
+ * nothing failed). Exposed for tests and subclass diagnostics —
33837
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33838
+ * with the devices that restored, and a late success is announced
33839
+ * through the `native-cap-change` → `updateCaps` path. */
33840
+ get restoreRetryCompletion() {
33841
+ return this._restoreRetryCompletion;
33842
+ }
33843
+ /** Devices that exhausted the retry bound this process lifetime. */
33844
+ get permanentRestoreFailures() {
33845
+ return [...this._permanentRestoreFailures.values()];
33846
+ }
33847
+ /** One-line operator-facing summary for `getStatus().error`, or
33848
+ * `null` when every device restored. */
33849
+ restoreFailureSummary() {
33850
+ if (this._permanentRestoreFailures.size === 0) return null;
33851
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33852
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33853
+ }
33854
+ cancelRestoreRetries() {
33855
+ this._restoreRetryScheduler?.cancel();
33856
+ this._restoreRetryScheduler = null;
33857
+ }
33858
+ recordPermanentRestoreFailure(failure) {
33859
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33860
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33861
+ tags: {
33862
+ deviceId: failure.deviceId,
33863
+ stableId: failure.stableId
33864
+ },
33865
+ meta: {
33866
+ type: failure.type,
33867
+ attempts: failure.attempts,
33868
+ error: failure.lastError
33869
+ }
33870
+ });
33871
+ }
33872
+ scheduleRestoreRetries(failures, attempt) {
33873
+ const scheduler = new DeviceRestoreRetryScheduler({
33874
+ logger: this.ctx.logger,
33875
+ delaysMs: this.restoreRetryDelaysMs,
33876
+ concurrency: this.restoreRetryConcurrency,
33877
+ attempt,
33878
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33879
+ });
33880
+ this._restoreRetryScheduler = scheduler;
33881
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33882
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33883
+ });
33884
+ }
33885
+ /**
33886
+ * Tear down and reconstruct ONE device from its persisted rows — the
33887
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33888
+ * and no other device this provider owns is disturbed.
33889
+ *
33890
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33891
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33892
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33893
+ * whatever number the row carries NOW. The teardown is `decommission` —
33894
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33895
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33896
+ * the boot restore's own `create()` path, including its pass 2: first-class
33897
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33898
+ * parent by the cascade and must be re-created explicitly, because only
33899
+ * accessory children come back through `getAccessoryChildren()`.
33900
+ *
33901
+ * Reloading an accessory child directly is refused (no device class) —
33902
+ * reload its parent instead.
33903
+ */
33904
+ async reloadDevice(input) {
33905
+ const { stableId } = input;
33906
+ const devices = this.ctx.kernel.devices;
33907
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33908
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33909
+ if (live) await devices.decommission(live.id);
33910
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33911
+ addonId: this.addonId,
33912
+ stableId
33913
+ });
33914
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33915
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33916
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33917
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33918
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33919
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33920
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33921
+ for (const row of rows) {
33922
+ if (row.parentDeviceId !== id) continue;
33923
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33924
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33925
+ if (!ChildClass) continue;
33926
+ try {
33927
+ await devices.create(row.stableId, ChildClass, {}, id);
33928
+ } catch (err) {
33929
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33930
+ tags: {
33931
+ deviceId: row.id,
33932
+ stableId: row.stableId
33933
+ },
33934
+ meta: {
33935
+ parentDeviceId: id,
33936
+ error: err instanceof Error ? err.message : String(err)
33937
+ }
33938
+ });
33939
+ }
33940
+ }
33941
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33942
+ tags: { deviceId: id },
33943
+ meta: {
33944
+ stableId,
33945
+ type: meta.type
33946
+ }
33947
+ });
33948
+ return { deviceId: id };
33641
33949
  }
33642
33950
  /**
33643
33951
  * Restore devices from persisted state. Two-pass:
@@ -33663,55 +33971,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33663
33971
  * accessory-spawn flow handles via the parent's
33664
33972
  * `getAccessoryChildren()`. Override only when the default doesn't
33665
33973
  * fit.
33974
+ *
33975
+ * A row that fails either pass is NOT terminal (D347): it is handed
33976
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33977
+ * Only after the bound is exhausted is the device marked permanently
33978
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33979
+ * `getStatus().error`.
33666
33980
  */
33667
33981
  async onRestoreDevices(savedDevices) {
33668
33982
  const restored = /* @__PURE__ */ new Set();
33983
+ const failures = [];
33984
+ const attemptRestore = async (saved) => {
33985
+ if (restored.has(saved.id)) return;
33986
+ const Class = this.deviceClasses[saved.type];
33987
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33988
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33989
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33990
+ restored.add(saved.id);
33991
+ };
33669
33992
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33670
33993
  const restoreOne = async (saved) => {
33671
- const Class = this.deviceClasses[saved.type];
33672
- if (!Class) {
33994
+ if (!this.deviceClasses[saved.type]) {
33673
33995
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33674
- tags: { stableId: saved.stableId },
33996
+ tags: {
33997
+ deviceId: saved.id,
33998
+ stableId: saved.stableId
33999
+ },
33675
34000
  meta: { type: saved.type }
33676
34001
  });
33677
34002
  return;
33678
34003
  }
33679
34004
  try {
33680
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33681
- restored.add(saved.id);
34005
+ await attemptRestore(saved);
33682
34006
  } catch (err) {
33683
- this.ctx.logger.warn("Failed to restore device", {
33684
- tags: { stableId: saved.stableId },
34007
+ const error = err instanceof Error ? err.message : String(err);
34008
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
34009
+ tags: {
34010
+ deviceId: saved.id,
34011
+ stableId: saved.stableId
34012
+ },
33685
34013
  meta: {
33686
34014
  type: saved.type,
33687
- error: err instanceof Error ? err.message : String(err)
34015
+ attempt: 1,
34016
+ error
33688
34017
  }
33689
34018
  });
34019
+ failures.push({
34020
+ saved,
34021
+ error
34022
+ });
33690
34023
  }
33691
34024
  };
33692
34025
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
34026
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33693
34027
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33694
34028
  for (const saved of childRows) {
33695
- const Class = this.deviceClasses[saved.type];
33696
- if (!Class) continue;
34029
+ if (!this.deviceClasses[saved.type]) continue;
33697
34030
  if (saved.parentDeviceId === null) continue;
33698
- if (!restored.has(saved.parentDeviceId)) continue;
33699
- try {
33700
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33701
- restored.add(saved.id);
33702
- } catch (err) {
33703
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
34031
+ if (restored.has(saved.parentDeviceId)) {
34032
+ try {
34033
+ await attemptRestore(saved);
34034
+ } catch (err) {
34035
+ const error = err instanceof Error ? err.message : String(err);
34036
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
34037
+ tags: {
34038
+ deviceId: saved.id,
34039
+ stableId: saved.stableId,
34040
+ parentDeviceId: saved.parentDeviceId
34041
+ },
34042
+ meta: {
34043
+ type: saved.type,
34044
+ attempt: 1,
34045
+ error
34046
+ }
34047
+ });
34048
+ failures.push({
34049
+ saved,
34050
+ error
34051
+ });
34052
+ }
34053
+ continue;
34054
+ }
34055
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
34056
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33704
34057
  tags: {
34058
+ deviceId: saved.id,
33705
34059
  stableId: saved.stableId,
33706
34060
  parentDeviceId: saved.parentDeviceId
33707
34061
  },
33708
- meta: {
33709
- type: saved.type,
33710
- error: err instanceof Error ? err.message : String(err)
33711
- }
34062
+ meta: { type: saved.type }
33712
34063
  });
34064
+ failures.push({
34065
+ saved,
34066
+ error: `parent device ${saved.parentDeviceId} not restored`
34067
+ });
34068
+ continue;
33713
34069
  }
33714
34070
  }
34071
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
34072
+ return {
34073
+ restoredCount: restored.size,
34074
+ failedCount: failures.length
34075
+ };
33715
34076
  }
33716
34077
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33717
34078
  toSummary(device) {
@@ -35474,6 +35835,12 @@ Object.freeze({
35474
35835
  addonId: null,
35475
35836
  access: "view"
35476
35837
  },
35838
+ "deviceProvider.reloadDevice": {
35839
+ capName: "device-provider",
35840
+ capScope: "system",
35841
+ addonId: null,
35842
+ access: "create"
35843
+ },
35477
35844
  "deviceProvider.start": {
35478
35845
  capName: "device-provider",
35479
35846
  capScope: "system",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-rademacher",
3
- "version": "0.2.58",
3
+ "version": "0.2.59",
4
4
  "description": "Rademacher HomePilot device-provider addon for CamStack — wraps the @apocaliss92/noderademacher local-hub client (roller shutters over the cover cap)",
5
5
  "keywords": [
6
6
  "camstack",