@camstack/addon-provider-gree 0.2.58 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +391 -24
  2. package/dist/addon.mjs +391 -24
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -12815,6 +12815,35 @@ var deviceProviderCapability = {
12815
12815
  name: string(),
12816
12816
  type: string()
12817
12817
  }))),
12818
+ /**
12819
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12820
+ * touching no other device this provider owns.
12821
+ *
12822
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12823
+ * migrated numbers: after `swapIds` the runner's live instance still
12824
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12825
+ * registrations and its log tags), and a live object cannot be renumbered.
12826
+ * Before this method the only flush was restarting the whole owning addon
12827
+ * — which took every camera the provider owns down with it (28 devices
12828
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12829
+ * same day ~27 devices' native caps did not come back on their own).
12830
+ *
12831
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12832
+ * that changes. The reply carries the id the device answers on NOW.
12833
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12834
+ * instance (if any), then re-create from the persisted row: the same
12835
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12836
+ * An RPC, never an event: a dropped event would leave the runner writing
12837
+ * against the wrong camera (D8).
12838
+ *
12839
+ * Construction can dial hardware, and the migrated source is
12840
+ * characteristically dead — the timeout covers a full activate window
12841
+ * rather than the 60 s default.
12842
+ */
12843
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12844
+ kind: "mutation",
12845
+ timeoutMs: 3 * 6e4
12846
+ }),
12818
12847
  supportsDiscovery: method(object({}), boolean()),
12819
12848
  /**
12820
12849
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13142,7 +13171,8 @@ method(object({
13142
13171
  targetId: number()
13143
13172
  }), MigrateDeviceResultSchema, {
13144
13173
  kind: "mutation",
13145
- auth: "admin"
13174
+ auth: "admin",
13175
+ timeoutMs: 12 * 6e4
13146
13176
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13147
13177
  deviceId: number(),
13148
13178
  name: string()
@@ -32498,6 +32528,147 @@ var BaseDevice$1 = class {
32498
32528
  }
32499
32529
  };
32500
32530
  /**
32531
+ * Delays before retry rounds 1..N — the round count IS the bound.
32532
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32533
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32534
+ * per attempt) covers a device-manager lock held for minutes — the
32535
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32536
+ */
32537
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32538
+ 1e4,
32539
+ 3e4,
32540
+ 9e4
32541
+ ];
32542
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32543
+ function sleep$1(ms, signal) {
32544
+ return new Promise((resolve) => {
32545
+ if (signal.aborted) {
32546
+ resolve();
32547
+ return;
32548
+ }
32549
+ const onAbort = () => {
32550
+ clearTimeout(timer);
32551
+ resolve();
32552
+ };
32553
+ const timer = setTimeout(() => {
32554
+ signal.removeEventListener("abort", onAbort);
32555
+ resolve();
32556
+ }, ms);
32557
+ timer.unref?.();
32558
+ signal.addEventListener("abort", onAbort, { once: true });
32559
+ });
32560
+ }
32561
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32562
+ * not reject (callers wrap their own try/catch). */
32563
+ async function runWithConcurrency(items, width, fn) {
32564
+ const queue = [...items];
32565
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32566
+ const lane = async () => {
32567
+ for (;;) {
32568
+ const item = queue.shift();
32569
+ if (item === void 0) return;
32570
+ await fn(item);
32571
+ }
32572
+ };
32573
+ await Promise.all(Array.from({ length: laneCount }, lane));
32574
+ }
32575
+ var DeviceRestoreRetryScheduler = class {
32576
+ #logger;
32577
+ #attempt;
32578
+ #onPermanentFailure;
32579
+ #delaysMs;
32580
+ #concurrency;
32581
+ #now;
32582
+ #abort = new AbortController();
32583
+ constructor(options) {
32584
+ this.#logger = options.logger;
32585
+ this.#attempt = options.attempt;
32586
+ this.#onPermanentFailure = options.onPermanentFailure;
32587
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32588
+ this.#concurrency = options.concurrency ?? 4;
32589
+ this.#now = options.now ?? Date.now;
32590
+ }
32591
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32592
+ * permanently failed — the next boot restores them from disk. */
32593
+ cancel() {
32594
+ this.#abort.abort();
32595
+ }
32596
+ /**
32597
+ * Run the bounded retry rounds. Resolves when every entry has either
32598
+ * restored, been marked permanently failed, or the scheduler was
32599
+ * cancelled. Never rejects.
32600
+ */
32601
+ async run(initialFailures) {
32602
+ let pending = initialFailures.map((failure) => ({
32603
+ saved: failure.saved,
32604
+ lastError: failure.error,
32605
+ attempts: 1
32606
+ }));
32607
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32608
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32609
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32610
+ if (this.#abort.signal.aborted) break;
32611
+ pending = await this.#runRound(pending, round);
32612
+ }
32613
+ if (this.#abort.signal.aborted) return [];
32614
+ const terminal = pending.map((entry) => ({
32615
+ deviceId: entry.saved.id,
32616
+ stableId: entry.saved.stableId,
32617
+ type: String(entry.saved.type),
32618
+ attempts: entry.attempts,
32619
+ lastError: entry.lastError,
32620
+ failedAt: this.#now()
32621
+ }));
32622
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32623
+ return terminal;
32624
+ }
32625
+ /** One retry round: parents first (phase 0), then hub-adopted
32626
+ * children (phase 1) — a child's attempt depends on its parent
32627
+ * having landed, exactly like the initial two-pass restore. */
32628
+ async #runRound(pending, round) {
32629
+ const next = [];
32630
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32631
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32632
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32633
+ if (this.#abort.signal.aborted) {
32634
+ next.push(entry);
32635
+ return;
32636
+ }
32637
+ const attemptNo = entry.attempts + 1;
32638
+ try {
32639
+ await this.#attempt(entry.saved);
32640
+ this.#logger.info("Device restored on retry", {
32641
+ tags: {
32642
+ deviceId: entry.saved.id,
32643
+ stableId: entry.saved.stableId
32644
+ },
32645
+ meta: { attempt: attemptNo }
32646
+ });
32647
+ } catch (err) {
32648
+ const lastError = err instanceof Error ? err.message : String(err);
32649
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32650
+ this.#logger.warn("Device restore retry failed", {
32651
+ tags: {
32652
+ deviceId: entry.saved.id,
32653
+ stableId: entry.saved.stableId
32654
+ },
32655
+ meta: {
32656
+ attempt: attemptNo,
32657
+ remainingRetries,
32658
+ error: lastError
32659
+ }
32660
+ });
32661
+ next.push({
32662
+ saved: entry.saved,
32663
+ lastError,
32664
+ attempts: attemptNo
32665
+ });
32666
+ }
32667
+ });
32668
+ return next;
32669
+ }
32670
+ };
32671
+ /**
32501
32672
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32502
32673
  * device-provider cap router. Shared across all providers.
32503
32674
  */
@@ -32546,6 +32717,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32546
32717
  }];
32547
32718
  }
32548
32719
  async onShutdown() {
32720
+ this.cancelRestoreRetries();
32549
32721
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32550
32722
  for (const device of devices) try {
32551
32723
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32563,9 +32735,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32563
32735
  async start() {}
32564
32736
  async stop() {}
32565
32737
  async getStatus() {
32738
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32739
+ const summary = this.restoreFailureSummary();
32740
+ if (summary === null) return {
32741
+ connected: true,
32742
+ deviceCount: all.length
32743
+ };
32566
32744
  return {
32567
32745
  connected: true,
32568
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32746
+ deviceCount: all.length,
32747
+ error: summary
32569
32748
  };
32570
32749
  }
32571
32750
  async getDevices() {
@@ -32655,8 +32834,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32655
32834
  };
32656
32835
  }
32657
32836
  async restoreDevices(savedDevices) {
32658
- await this.onRestoreDevices(savedDevices);
32659
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32837
+ const report = await this.onRestoreDevices(savedDevices);
32838
+ if (savedDevices.length === 0) return;
32839
+ if (report && report.failedCount > 0) {
32840
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32841
+ return;
32842
+ }
32843
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32844
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32845
+ }
32846
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32847
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32848
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32849
+ * never re-stampede full-width while the initial pass does (D167). */
32850
+ restoreRetryConcurrency = 4;
32851
+ _restoreRetryScheduler = null;
32852
+ _restoreRetryCompletion = null;
32853
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32854
+ /** Settles when the background retry rounds finish (or `null` when
32855
+ * nothing failed). Exposed for tests and subclass diagnostics —
32856
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32857
+ * with the devices that restored, and a late success is announced
32858
+ * through the `native-cap-change` → `updateCaps` path. */
32859
+ get restoreRetryCompletion() {
32860
+ return this._restoreRetryCompletion;
32861
+ }
32862
+ /** Devices that exhausted the retry bound this process lifetime. */
32863
+ get permanentRestoreFailures() {
32864
+ return [...this._permanentRestoreFailures.values()];
32865
+ }
32866
+ /** One-line operator-facing summary for `getStatus().error`, or
32867
+ * `null` when every device restored. */
32868
+ restoreFailureSummary() {
32869
+ if (this._permanentRestoreFailures.size === 0) return null;
32870
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32871
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32872
+ }
32873
+ cancelRestoreRetries() {
32874
+ this._restoreRetryScheduler?.cancel();
32875
+ this._restoreRetryScheduler = null;
32876
+ }
32877
+ recordPermanentRestoreFailure(failure) {
32878
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32879
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32880
+ tags: {
32881
+ deviceId: failure.deviceId,
32882
+ stableId: failure.stableId
32883
+ },
32884
+ meta: {
32885
+ type: failure.type,
32886
+ attempts: failure.attempts,
32887
+ error: failure.lastError
32888
+ }
32889
+ });
32890
+ }
32891
+ scheduleRestoreRetries(failures, attempt) {
32892
+ const scheduler = new DeviceRestoreRetryScheduler({
32893
+ logger: this.ctx.logger,
32894
+ delaysMs: this.restoreRetryDelaysMs,
32895
+ concurrency: this.restoreRetryConcurrency,
32896
+ attempt,
32897
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32898
+ });
32899
+ this._restoreRetryScheduler = scheduler;
32900
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32901
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32902
+ });
32903
+ }
32904
+ /**
32905
+ * Tear down and reconstruct ONE device from its persisted rows — the
32906
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32907
+ * and no other device this provider owns is disturbed.
32908
+ *
32909
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32910
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32911
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32912
+ * whatever number the row carries NOW. The teardown is `decommission` —
32913
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32914
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32915
+ * the boot restore's own `create()` path, including its pass 2: first-class
32916
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32917
+ * parent by the cascade and must be re-created explicitly, because only
32918
+ * accessory children come back through `getAccessoryChildren()`.
32919
+ *
32920
+ * Reloading an accessory child directly is refused (no device class) —
32921
+ * reload its parent instead.
32922
+ */
32923
+ async reloadDevice(input) {
32924
+ const { stableId } = input;
32925
+ const devices = this.ctx.kernel.devices;
32926
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32927
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32928
+ if (live) await devices.decommission(live.id);
32929
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32930
+ addonId: this.addonId,
32931
+ stableId
32932
+ });
32933
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32934
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32935
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32936
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32937
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32938
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32939
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32940
+ for (const row of rows) {
32941
+ if (row.parentDeviceId !== id) continue;
32942
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32943
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32944
+ if (!ChildClass) continue;
32945
+ try {
32946
+ await devices.create(row.stableId, ChildClass, {}, id);
32947
+ } catch (err) {
32948
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32949
+ tags: {
32950
+ deviceId: row.id,
32951
+ stableId: row.stableId
32952
+ },
32953
+ meta: {
32954
+ parentDeviceId: id,
32955
+ error: err instanceof Error ? err.message : String(err)
32956
+ }
32957
+ });
32958
+ }
32959
+ }
32960
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32961
+ tags: { deviceId: id },
32962
+ meta: {
32963
+ stableId,
32964
+ type: meta.type
32965
+ }
32966
+ });
32967
+ return { deviceId: id };
32660
32968
  }
32661
32969
  /**
32662
32970
  * Restore devices from persisted state. Two-pass:
@@ -32682,55 +32990,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32682
32990
  * accessory-spawn flow handles via the parent's
32683
32991
  * `getAccessoryChildren()`. Override only when the default doesn't
32684
32992
  * fit.
32993
+ *
32994
+ * A row that fails either pass is NOT terminal (D347): it is handed
32995
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
32996
+ * Only after the bound is exhausted is the device marked permanently
32997
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
32998
+ * `getStatus().error`.
32685
32999
  */
32686
33000
  async onRestoreDevices(savedDevices) {
32687
33001
  const restored = /* @__PURE__ */ new Set();
33002
+ const failures = [];
33003
+ const attemptRestore = async (saved) => {
33004
+ if (restored.has(saved.id)) return;
33005
+ const Class = this.deviceClasses[saved.type];
33006
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33007
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33008
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33009
+ restored.add(saved.id);
33010
+ };
32688
33011
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32689
33012
  const restoreOne = async (saved) => {
32690
- const Class = this.deviceClasses[saved.type];
32691
- if (!Class) {
33013
+ if (!this.deviceClasses[saved.type]) {
32692
33014
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32693
- tags: { stableId: saved.stableId },
33015
+ tags: {
33016
+ deviceId: saved.id,
33017
+ stableId: saved.stableId
33018
+ },
32694
33019
  meta: { type: saved.type }
32695
33020
  });
32696
33021
  return;
32697
33022
  }
32698
33023
  try {
32699
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32700
- restored.add(saved.id);
33024
+ await attemptRestore(saved);
32701
33025
  } catch (err) {
32702
- this.ctx.logger.warn("Failed to restore device", {
32703
- tags: { stableId: saved.stableId },
33026
+ const error = err instanceof Error ? err.message : String(err);
33027
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33028
+ tags: {
33029
+ deviceId: saved.id,
33030
+ stableId: saved.stableId
33031
+ },
32704
33032
  meta: {
32705
33033
  type: saved.type,
32706
- error: err instanceof Error ? err.message : String(err)
33034
+ attempt: 1,
33035
+ error
32707
33036
  }
32708
33037
  });
33038
+ failures.push({
33039
+ saved,
33040
+ error
33041
+ });
32709
33042
  }
32710
33043
  };
32711
33044
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33045
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32712
33046
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32713
33047
  for (const saved of childRows) {
32714
- const Class = this.deviceClasses[saved.type];
32715
- if (!Class) continue;
33048
+ if (!this.deviceClasses[saved.type]) continue;
32716
33049
  if (saved.parentDeviceId === null) continue;
32717
- if (!restored.has(saved.parentDeviceId)) continue;
32718
- try {
32719
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32720
- restored.add(saved.id);
32721
- } catch (err) {
32722
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33050
+ if (restored.has(saved.parentDeviceId)) {
33051
+ try {
33052
+ await attemptRestore(saved);
33053
+ } catch (err) {
33054
+ const error = err instanceof Error ? err.message : String(err);
33055
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33056
+ tags: {
33057
+ deviceId: saved.id,
33058
+ stableId: saved.stableId,
33059
+ parentDeviceId: saved.parentDeviceId
33060
+ },
33061
+ meta: {
33062
+ type: saved.type,
33063
+ attempt: 1,
33064
+ error
33065
+ }
33066
+ });
33067
+ failures.push({
33068
+ saved,
33069
+ error
33070
+ });
33071
+ }
33072
+ continue;
33073
+ }
33074
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33075
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32723
33076
  tags: {
33077
+ deviceId: saved.id,
32724
33078
  stableId: saved.stableId,
32725
33079
  parentDeviceId: saved.parentDeviceId
32726
33080
  },
32727
- meta: {
32728
- type: saved.type,
32729
- error: err instanceof Error ? err.message : String(err)
32730
- }
33081
+ meta: { type: saved.type }
32731
33082
  });
33083
+ failures.push({
33084
+ saved,
33085
+ error: `parent device ${saved.parentDeviceId} not restored`
33086
+ });
33087
+ continue;
32732
33088
  }
32733
33089
  }
33090
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33091
+ return {
33092
+ restoredCount: restored.size,
33093
+ failedCount: failures.length
33094
+ };
32734
33095
  }
32735
33096
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32736
33097
  toSummary(device) {
@@ -34493,6 +34854,12 @@ Object.freeze({
34493
34854
  addonId: null,
34494
34855
  access: "view"
34495
34856
  },
34857
+ "deviceProvider.reloadDevice": {
34858
+ capName: "device-provider",
34859
+ capScope: "system",
34860
+ addonId: null,
34861
+ access: "create"
34862
+ },
34496
34863
  "deviceProvider.start": {
34497
34864
  capName: "device-provider",
34498
34865
  capScope: "system",
package/dist/addon.mjs CHANGED
@@ -12814,6 +12814,35 @@ var deviceProviderCapability = {
12814
12814
  name: string(),
12815
12815
  type: string()
12816
12816
  }))),
12817
+ /**
12818
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12819
+ * touching no other device this provider owns.
12820
+ *
12821
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12822
+ * migrated numbers: after `swapIds` the runner's live instance still
12823
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12824
+ * registrations and its log tags), and a live object cannot be renumbered.
12825
+ * Before this method the only flush was restarting the whole owning addon
12826
+ * — which took every camera the provider owns down with it (28 devices
12827
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12828
+ * same day ~27 devices' native caps did not come back on their own).
12829
+ *
12830
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12831
+ * that changes. The reply carries the id the device answers on NOW.
12832
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12833
+ * instance (if any), then re-create from the persisted row: the same
12834
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12835
+ * An RPC, never an event: a dropped event would leave the runner writing
12836
+ * against the wrong camera (D8).
12837
+ *
12838
+ * Construction can dial hardware, and the migrated source is
12839
+ * characteristically dead — the timeout covers a full activate window
12840
+ * rather than the 60 s default.
12841
+ */
12842
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12843
+ kind: "mutation",
12844
+ timeoutMs: 3 * 6e4
12845
+ }),
12817
12846
  supportsDiscovery: method(object({}), boolean()),
12818
12847
  /**
12819
12848
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13141,7 +13170,8 @@ method(object({
13141
13170
  targetId: number()
13142
13171
  }), MigrateDeviceResultSchema, {
13143
13172
  kind: "mutation",
13144
- auth: "admin"
13173
+ auth: "admin",
13174
+ timeoutMs: 12 * 6e4
13145
13175
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13146
13176
  deviceId: number(),
13147
13177
  name: string()
@@ -32497,6 +32527,147 @@ var BaseDevice$1 = class {
32497
32527
  }
32498
32528
  };
32499
32529
  /**
32530
+ * Delays before retry rounds 1..N — the round count IS the bound.
32531
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32532
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32533
+ * per attempt) covers a device-manager lock held for minutes — the
32534
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32535
+ */
32536
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32537
+ 1e4,
32538
+ 3e4,
32539
+ 9e4
32540
+ ];
32541
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32542
+ function sleep$1(ms, signal) {
32543
+ return new Promise((resolve) => {
32544
+ if (signal.aborted) {
32545
+ resolve();
32546
+ return;
32547
+ }
32548
+ const onAbort = () => {
32549
+ clearTimeout(timer);
32550
+ resolve();
32551
+ };
32552
+ const timer = setTimeout(() => {
32553
+ signal.removeEventListener("abort", onAbort);
32554
+ resolve();
32555
+ }, ms);
32556
+ timer.unref?.();
32557
+ signal.addEventListener("abort", onAbort, { once: true });
32558
+ });
32559
+ }
32560
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32561
+ * not reject (callers wrap their own try/catch). */
32562
+ async function runWithConcurrency(items, width, fn) {
32563
+ const queue = [...items];
32564
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32565
+ const lane = async () => {
32566
+ for (;;) {
32567
+ const item = queue.shift();
32568
+ if (item === void 0) return;
32569
+ await fn(item);
32570
+ }
32571
+ };
32572
+ await Promise.all(Array.from({ length: laneCount }, lane));
32573
+ }
32574
+ var DeviceRestoreRetryScheduler = class {
32575
+ #logger;
32576
+ #attempt;
32577
+ #onPermanentFailure;
32578
+ #delaysMs;
32579
+ #concurrency;
32580
+ #now;
32581
+ #abort = new AbortController();
32582
+ constructor(options) {
32583
+ this.#logger = options.logger;
32584
+ this.#attempt = options.attempt;
32585
+ this.#onPermanentFailure = options.onPermanentFailure;
32586
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32587
+ this.#concurrency = options.concurrency ?? 4;
32588
+ this.#now = options.now ?? Date.now;
32589
+ }
32590
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32591
+ * permanently failed — the next boot restores them from disk. */
32592
+ cancel() {
32593
+ this.#abort.abort();
32594
+ }
32595
+ /**
32596
+ * Run the bounded retry rounds. Resolves when every entry has either
32597
+ * restored, been marked permanently failed, or the scheduler was
32598
+ * cancelled. Never rejects.
32599
+ */
32600
+ async run(initialFailures) {
32601
+ let pending = initialFailures.map((failure) => ({
32602
+ saved: failure.saved,
32603
+ lastError: failure.error,
32604
+ attempts: 1
32605
+ }));
32606
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32607
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32608
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32609
+ if (this.#abort.signal.aborted) break;
32610
+ pending = await this.#runRound(pending, round);
32611
+ }
32612
+ if (this.#abort.signal.aborted) return [];
32613
+ const terminal = pending.map((entry) => ({
32614
+ deviceId: entry.saved.id,
32615
+ stableId: entry.saved.stableId,
32616
+ type: String(entry.saved.type),
32617
+ attempts: entry.attempts,
32618
+ lastError: entry.lastError,
32619
+ failedAt: this.#now()
32620
+ }));
32621
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32622
+ return terminal;
32623
+ }
32624
+ /** One retry round: parents first (phase 0), then hub-adopted
32625
+ * children (phase 1) — a child's attempt depends on its parent
32626
+ * having landed, exactly like the initial two-pass restore. */
32627
+ async #runRound(pending, round) {
32628
+ const next = [];
32629
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32630
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32631
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32632
+ if (this.#abort.signal.aborted) {
32633
+ next.push(entry);
32634
+ return;
32635
+ }
32636
+ const attemptNo = entry.attempts + 1;
32637
+ try {
32638
+ await this.#attempt(entry.saved);
32639
+ this.#logger.info("Device restored on retry", {
32640
+ tags: {
32641
+ deviceId: entry.saved.id,
32642
+ stableId: entry.saved.stableId
32643
+ },
32644
+ meta: { attempt: attemptNo }
32645
+ });
32646
+ } catch (err) {
32647
+ const lastError = err instanceof Error ? err.message : String(err);
32648
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32649
+ this.#logger.warn("Device restore retry failed", {
32650
+ tags: {
32651
+ deviceId: entry.saved.id,
32652
+ stableId: entry.saved.stableId
32653
+ },
32654
+ meta: {
32655
+ attempt: attemptNo,
32656
+ remainingRetries,
32657
+ error: lastError
32658
+ }
32659
+ });
32660
+ next.push({
32661
+ saved: entry.saved,
32662
+ lastError,
32663
+ attempts: attemptNo
32664
+ });
32665
+ }
32666
+ });
32667
+ return next;
32668
+ }
32669
+ };
32670
+ /**
32500
32671
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32501
32672
  * device-provider cap router. Shared across all providers.
32502
32673
  */
@@ -32545,6 +32716,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32545
32716
  }];
32546
32717
  }
32547
32718
  async onShutdown() {
32719
+ this.cancelRestoreRetries();
32548
32720
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32549
32721
  for (const device of devices) try {
32550
32722
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32562,9 +32734,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32562
32734
  async start() {}
32563
32735
  async stop() {}
32564
32736
  async getStatus() {
32737
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32738
+ const summary = this.restoreFailureSummary();
32739
+ if (summary === null) return {
32740
+ connected: true,
32741
+ deviceCount: all.length
32742
+ };
32565
32743
  return {
32566
32744
  connected: true,
32567
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32745
+ deviceCount: all.length,
32746
+ error: summary
32568
32747
  };
32569
32748
  }
32570
32749
  async getDevices() {
@@ -32654,8 +32833,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32654
32833
  };
32655
32834
  }
32656
32835
  async restoreDevices(savedDevices) {
32657
- await this.onRestoreDevices(savedDevices);
32658
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32836
+ const report = await this.onRestoreDevices(savedDevices);
32837
+ if (savedDevices.length === 0) return;
32838
+ if (report && report.failedCount > 0) {
32839
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32840
+ return;
32841
+ }
32842
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32843
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32844
+ }
32845
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32846
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32847
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32848
+ * never re-stampede full-width while the initial pass does (D167). */
32849
+ restoreRetryConcurrency = 4;
32850
+ _restoreRetryScheduler = null;
32851
+ _restoreRetryCompletion = null;
32852
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32853
+ /** Settles when the background retry rounds finish (or `null` when
32854
+ * nothing failed). Exposed for tests and subclass diagnostics —
32855
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32856
+ * with the devices that restored, and a late success is announced
32857
+ * through the `native-cap-change` → `updateCaps` path. */
32858
+ get restoreRetryCompletion() {
32859
+ return this._restoreRetryCompletion;
32860
+ }
32861
+ /** Devices that exhausted the retry bound this process lifetime. */
32862
+ get permanentRestoreFailures() {
32863
+ return [...this._permanentRestoreFailures.values()];
32864
+ }
32865
+ /** One-line operator-facing summary for `getStatus().error`, or
32866
+ * `null` when every device restored. */
32867
+ restoreFailureSummary() {
32868
+ if (this._permanentRestoreFailures.size === 0) return null;
32869
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32870
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32871
+ }
32872
+ cancelRestoreRetries() {
32873
+ this._restoreRetryScheduler?.cancel();
32874
+ this._restoreRetryScheduler = null;
32875
+ }
32876
+ recordPermanentRestoreFailure(failure) {
32877
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32878
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32879
+ tags: {
32880
+ deviceId: failure.deviceId,
32881
+ stableId: failure.stableId
32882
+ },
32883
+ meta: {
32884
+ type: failure.type,
32885
+ attempts: failure.attempts,
32886
+ error: failure.lastError
32887
+ }
32888
+ });
32889
+ }
32890
+ scheduleRestoreRetries(failures, attempt) {
32891
+ const scheduler = new DeviceRestoreRetryScheduler({
32892
+ logger: this.ctx.logger,
32893
+ delaysMs: this.restoreRetryDelaysMs,
32894
+ concurrency: this.restoreRetryConcurrency,
32895
+ attempt,
32896
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32897
+ });
32898
+ this._restoreRetryScheduler = scheduler;
32899
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32900
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32901
+ });
32902
+ }
32903
+ /**
32904
+ * Tear down and reconstruct ONE device from its persisted rows — the
32905
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32906
+ * and no other device this provider owns is disturbed.
32907
+ *
32908
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32909
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32910
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32911
+ * whatever number the row carries NOW. The teardown is `decommission` —
32912
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32913
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32914
+ * the boot restore's own `create()` path, including its pass 2: first-class
32915
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32916
+ * parent by the cascade and must be re-created explicitly, because only
32917
+ * accessory children come back through `getAccessoryChildren()`.
32918
+ *
32919
+ * Reloading an accessory child directly is refused (no device class) —
32920
+ * reload its parent instead.
32921
+ */
32922
+ async reloadDevice(input) {
32923
+ const { stableId } = input;
32924
+ const devices = this.ctx.kernel.devices;
32925
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32926
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32927
+ if (live) await devices.decommission(live.id);
32928
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32929
+ addonId: this.addonId,
32930
+ stableId
32931
+ });
32932
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32933
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32934
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32935
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32936
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32937
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32938
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32939
+ for (const row of rows) {
32940
+ if (row.parentDeviceId !== id) continue;
32941
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32942
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32943
+ if (!ChildClass) continue;
32944
+ try {
32945
+ await devices.create(row.stableId, ChildClass, {}, id);
32946
+ } catch (err) {
32947
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32948
+ tags: {
32949
+ deviceId: row.id,
32950
+ stableId: row.stableId
32951
+ },
32952
+ meta: {
32953
+ parentDeviceId: id,
32954
+ error: err instanceof Error ? err.message : String(err)
32955
+ }
32956
+ });
32957
+ }
32958
+ }
32959
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32960
+ tags: { deviceId: id },
32961
+ meta: {
32962
+ stableId,
32963
+ type: meta.type
32964
+ }
32965
+ });
32966
+ return { deviceId: id };
32659
32967
  }
32660
32968
  /**
32661
32969
  * Restore devices from persisted state. Two-pass:
@@ -32681,55 +32989,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32681
32989
  * accessory-spawn flow handles via the parent's
32682
32990
  * `getAccessoryChildren()`. Override only when the default doesn't
32683
32991
  * fit.
32992
+ *
32993
+ * A row that fails either pass is NOT terminal (D347): it is handed
32994
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
32995
+ * Only after the bound is exhausted is the device marked permanently
32996
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
32997
+ * `getStatus().error`.
32684
32998
  */
32685
32999
  async onRestoreDevices(savedDevices) {
32686
33000
  const restored = /* @__PURE__ */ new Set();
33001
+ const failures = [];
33002
+ const attemptRestore = async (saved) => {
33003
+ if (restored.has(saved.id)) return;
33004
+ const Class = this.deviceClasses[saved.type];
33005
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33006
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33007
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33008
+ restored.add(saved.id);
33009
+ };
32687
33010
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32688
33011
  const restoreOne = async (saved) => {
32689
- const Class = this.deviceClasses[saved.type];
32690
- if (!Class) {
33012
+ if (!this.deviceClasses[saved.type]) {
32691
33013
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32692
- tags: { stableId: saved.stableId },
33014
+ tags: {
33015
+ deviceId: saved.id,
33016
+ stableId: saved.stableId
33017
+ },
32693
33018
  meta: { type: saved.type }
32694
33019
  });
32695
33020
  return;
32696
33021
  }
32697
33022
  try {
32698
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32699
- restored.add(saved.id);
33023
+ await attemptRestore(saved);
32700
33024
  } catch (err) {
32701
- this.ctx.logger.warn("Failed to restore device", {
32702
- tags: { stableId: saved.stableId },
33025
+ const error = err instanceof Error ? err.message : String(err);
33026
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33027
+ tags: {
33028
+ deviceId: saved.id,
33029
+ stableId: saved.stableId
33030
+ },
32703
33031
  meta: {
32704
33032
  type: saved.type,
32705
- error: err instanceof Error ? err.message : String(err)
33033
+ attempt: 1,
33034
+ error
32706
33035
  }
32707
33036
  });
33037
+ failures.push({
33038
+ saved,
33039
+ error
33040
+ });
32708
33041
  }
32709
33042
  };
32710
33043
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33044
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32711
33045
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32712
33046
  for (const saved of childRows) {
32713
- const Class = this.deviceClasses[saved.type];
32714
- if (!Class) continue;
33047
+ if (!this.deviceClasses[saved.type]) continue;
32715
33048
  if (saved.parentDeviceId === null) continue;
32716
- if (!restored.has(saved.parentDeviceId)) continue;
32717
- try {
32718
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32719
- restored.add(saved.id);
32720
- } catch (err) {
32721
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33049
+ if (restored.has(saved.parentDeviceId)) {
33050
+ try {
33051
+ await attemptRestore(saved);
33052
+ } catch (err) {
33053
+ const error = err instanceof Error ? err.message : String(err);
33054
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33055
+ tags: {
33056
+ deviceId: saved.id,
33057
+ stableId: saved.stableId,
33058
+ parentDeviceId: saved.parentDeviceId
33059
+ },
33060
+ meta: {
33061
+ type: saved.type,
33062
+ attempt: 1,
33063
+ error
33064
+ }
33065
+ });
33066
+ failures.push({
33067
+ saved,
33068
+ error
33069
+ });
33070
+ }
33071
+ continue;
33072
+ }
33073
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33074
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32722
33075
  tags: {
33076
+ deviceId: saved.id,
32723
33077
  stableId: saved.stableId,
32724
33078
  parentDeviceId: saved.parentDeviceId
32725
33079
  },
32726
- meta: {
32727
- type: saved.type,
32728
- error: err instanceof Error ? err.message : String(err)
32729
- }
33080
+ meta: { type: saved.type }
32730
33081
  });
33082
+ failures.push({
33083
+ saved,
33084
+ error: `parent device ${saved.parentDeviceId} not restored`
33085
+ });
33086
+ continue;
32731
33087
  }
32732
33088
  }
33089
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33090
+ return {
33091
+ restoredCount: restored.size,
33092
+ failedCount: failures.length
33093
+ };
32733
33094
  }
32734
33095
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32735
33096
  toSummary(device) {
@@ -34492,6 +34853,12 @@ Object.freeze({
34492
34853
  addonId: null,
34493
34854
  access: "view"
34494
34855
  },
34856
+ "deviceProvider.reloadDevice": {
34857
+ capName: "device-provider",
34858
+ capScope: "system",
34859
+ addonId: null,
34860
+ access: "create"
34861
+ },
34495
34862
  "deviceProvider.start": {
34496
34863
  capName: "device-provider",
34497
34864
  capScope: "system",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-gree",
3
- "version": "0.2.58",
3
+ "version": "0.2.59",
4
4
  "description": "Gree air-conditioner device-provider addon for CamStack — wraps the @apocaliss92/nodegree local-UDP client (LAN discovery + AES control), exposing climate-control and fan-control",
5
5
  "keywords": [
6
6
  "camstack",