@camstack/addon-provider-tuya 0.2.59 → 0.2.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +391 -24
  2. package/dist/addon.mjs +391 -24
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -13649,6 +13649,35 @@ var deviceProviderCapability = {
13649
13649
  name: string(),
13650
13650
  type: string()
13651
13651
  }))),
13652
+ /**
13653
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13654
+ * touching no other device this provider owns.
13655
+ *
13656
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13657
+ * migrated numbers: after `swapIds` the runner's live instance still
13658
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13659
+ * registrations and its log tags), and a live object cannot be renumbered.
13660
+ * Before this method the only flush was restarting the whole owning addon
13661
+ * — which took every camera the provider owns down with it (28 devices
13662
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13663
+ * same day ~27 devices' native caps did not come back on their own).
13664
+ *
13665
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13666
+ * that changes. The reply carries the id the device answers on NOW.
13667
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13668
+ * instance (if any), then re-create from the persisted row: the same
13669
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13670
+ * An RPC, never an event: a dropped event would leave the runner writing
13671
+ * against the wrong camera (D8).
13672
+ *
13673
+ * Construction can dial hardware, and the migrated source is
13674
+ * characteristically dead — the timeout covers a full activate window
13675
+ * rather than the 60 s default.
13676
+ */
13677
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13678
+ kind: "mutation",
13679
+ timeoutMs: 3 * 6e4
13680
+ }),
13652
13681
  supportsDiscovery: method(object({}), boolean()),
13653
13682
  /**
13654
13683
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13976,7 +14005,8 @@ method(object({
13976
14005
  targetId: number()
13977
14006
  }), MigrateDeviceResultSchema, {
13978
14007
  kind: "mutation",
13979
- auth: "admin"
14008
+ auth: "admin",
14009
+ timeoutMs: 12 * 6e4
13980
14010
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13981
14011
  deviceId: number(),
13982
14012
  name: string()
@@ -33340,6 +33370,147 @@ var BaseDevice = class {
33340
33370
  }
33341
33371
  };
33342
33372
  /**
33373
+ * Delays before retry rounds 1..N — the round count IS the bound.
33374
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33375
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33376
+ * per attempt) covers a device-manager lock held for minutes — the
33377
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33378
+ */
33379
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33380
+ 1e4,
33381
+ 3e4,
33382
+ 9e4
33383
+ ];
33384
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33385
+ function sleep$1(ms, signal) {
33386
+ return new Promise((resolve) => {
33387
+ if (signal.aborted) {
33388
+ resolve();
33389
+ return;
33390
+ }
33391
+ const onAbort = () => {
33392
+ clearTimeout(timer);
33393
+ resolve();
33394
+ };
33395
+ const timer = setTimeout(() => {
33396
+ signal.removeEventListener("abort", onAbort);
33397
+ resolve();
33398
+ }, ms);
33399
+ timer.unref?.();
33400
+ signal.addEventListener("abort", onAbort, { once: true });
33401
+ });
33402
+ }
33403
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33404
+ * not reject (callers wrap their own try/catch). */
33405
+ async function runWithConcurrency(items, width, fn) {
33406
+ const queue = [...items];
33407
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33408
+ const lane = async () => {
33409
+ for (;;) {
33410
+ const item = queue.shift();
33411
+ if (item === void 0) return;
33412
+ await fn(item);
33413
+ }
33414
+ };
33415
+ await Promise.all(Array.from({ length: laneCount }, lane));
33416
+ }
33417
+ var DeviceRestoreRetryScheduler = class {
33418
+ #logger;
33419
+ #attempt;
33420
+ #onPermanentFailure;
33421
+ #delaysMs;
33422
+ #concurrency;
33423
+ #now;
33424
+ #abort = new AbortController();
33425
+ constructor(options) {
33426
+ this.#logger = options.logger;
33427
+ this.#attempt = options.attempt;
33428
+ this.#onPermanentFailure = options.onPermanentFailure;
33429
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33430
+ this.#concurrency = options.concurrency ?? 4;
33431
+ this.#now = options.now ?? Date.now;
33432
+ }
33433
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33434
+ * permanently failed — the next boot restores them from disk. */
33435
+ cancel() {
33436
+ this.#abort.abort();
33437
+ }
33438
+ /**
33439
+ * Run the bounded retry rounds. Resolves when every entry has either
33440
+ * restored, been marked permanently failed, or the scheduler was
33441
+ * cancelled. Never rejects.
33442
+ */
33443
+ async run(initialFailures) {
33444
+ let pending = initialFailures.map((failure) => ({
33445
+ saved: failure.saved,
33446
+ lastError: failure.error,
33447
+ attempts: 1
33448
+ }));
33449
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33450
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33451
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33452
+ if (this.#abort.signal.aborted) break;
33453
+ pending = await this.#runRound(pending, round);
33454
+ }
33455
+ if (this.#abort.signal.aborted) return [];
33456
+ const terminal = pending.map((entry) => ({
33457
+ deviceId: entry.saved.id,
33458
+ stableId: entry.saved.stableId,
33459
+ type: String(entry.saved.type),
33460
+ attempts: entry.attempts,
33461
+ lastError: entry.lastError,
33462
+ failedAt: this.#now()
33463
+ }));
33464
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33465
+ return terminal;
33466
+ }
33467
+ /** One retry round: parents first (phase 0), then hub-adopted
33468
+ * children (phase 1) — a child's attempt depends on its parent
33469
+ * having landed, exactly like the initial two-pass restore. */
33470
+ async #runRound(pending, round) {
33471
+ const next = [];
33472
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33473
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33474
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33475
+ if (this.#abort.signal.aborted) {
33476
+ next.push(entry);
33477
+ return;
33478
+ }
33479
+ const attemptNo = entry.attempts + 1;
33480
+ try {
33481
+ await this.#attempt(entry.saved);
33482
+ this.#logger.info("Device restored on retry", {
33483
+ tags: {
33484
+ deviceId: entry.saved.id,
33485
+ stableId: entry.saved.stableId
33486
+ },
33487
+ meta: { attempt: attemptNo }
33488
+ });
33489
+ } catch (err) {
33490
+ const lastError = err instanceof Error ? err.message : String(err);
33491
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33492
+ this.#logger.warn("Device restore retry failed", {
33493
+ tags: {
33494
+ deviceId: entry.saved.id,
33495
+ stableId: entry.saved.stableId
33496
+ },
33497
+ meta: {
33498
+ attempt: attemptNo,
33499
+ remainingRetries,
33500
+ error: lastError
33501
+ }
33502
+ });
33503
+ next.push({
33504
+ saved: entry.saved,
33505
+ lastError,
33506
+ attempts: attemptNo
33507
+ });
33508
+ }
33509
+ });
33510
+ return next;
33511
+ }
33512
+ };
33513
+ /**
33343
33514
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33344
33515
  * device-provider cap router. Shared across all providers.
33345
33516
  */
@@ -33388,6 +33559,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33388
33559
  }];
33389
33560
  }
33390
33561
  async onShutdown() {
33562
+ this.cancelRestoreRetries();
33391
33563
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33392
33564
  for (const device of devices) try {
33393
33565
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33405,9 +33577,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33405
33577
  async start() {}
33406
33578
  async stop() {}
33407
33579
  async getStatus() {
33580
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33581
+ const summary = this.restoreFailureSummary();
33582
+ if (summary === null) return {
33583
+ connected: true,
33584
+ deviceCount: all.length
33585
+ };
33408
33586
  return {
33409
33587
  connected: true,
33410
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33588
+ deviceCount: all.length,
33589
+ error: summary
33411
33590
  };
33412
33591
  }
33413
33592
  async getDevices() {
@@ -33497,8 +33676,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33497
33676
  };
33498
33677
  }
33499
33678
  async restoreDevices(savedDevices) {
33500
- await this.onRestoreDevices(savedDevices);
33501
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33679
+ const report = await this.onRestoreDevices(savedDevices);
33680
+ if (savedDevices.length === 0) return;
33681
+ if (report && report.failedCount > 0) {
33682
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33683
+ return;
33684
+ }
33685
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33686
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33687
+ }
33688
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33689
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33690
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33691
+ * never re-stampede full-width while the initial pass does (D167). */
33692
+ restoreRetryConcurrency = 4;
33693
+ _restoreRetryScheduler = null;
33694
+ _restoreRetryCompletion = null;
33695
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33696
+ /** Settles when the background retry rounds finish (or `null` when
33697
+ * nothing failed). Exposed for tests and subclass diagnostics —
33698
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33699
+ * with the devices that restored, and a late success is announced
33700
+ * through the `native-cap-change` → `updateCaps` path. */
33701
+ get restoreRetryCompletion() {
33702
+ return this._restoreRetryCompletion;
33703
+ }
33704
+ /** Devices that exhausted the retry bound this process lifetime. */
33705
+ get permanentRestoreFailures() {
33706
+ return [...this._permanentRestoreFailures.values()];
33707
+ }
33708
+ /** One-line operator-facing summary for `getStatus().error`, or
33709
+ * `null` when every device restored. */
33710
+ restoreFailureSummary() {
33711
+ if (this._permanentRestoreFailures.size === 0) return null;
33712
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33713
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33714
+ }
33715
+ cancelRestoreRetries() {
33716
+ this._restoreRetryScheduler?.cancel();
33717
+ this._restoreRetryScheduler = null;
33718
+ }
33719
+ recordPermanentRestoreFailure(failure) {
33720
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33721
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33722
+ tags: {
33723
+ deviceId: failure.deviceId,
33724
+ stableId: failure.stableId
33725
+ },
33726
+ meta: {
33727
+ type: failure.type,
33728
+ attempts: failure.attempts,
33729
+ error: failure.lastError
33730
+ }
33731
+ });
33732
+ }
33733
+ scheduleRestoreRetries(failures, attempt) {
33734
+ const scheduler = new DeviceRestoreRetryScheduler({
33735
+ logger: this.ctx.logger,
33736
+ delaysMs: this.restoreRetryDelaysMs,
33737
+ concurrency: this.restoreRetryConcurrency,
33738
+ attempt,
33739
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33740
+ });
33741
+ this._restoreRetryScheduler = scheduler;
33742
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33743
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33744
+ });
33745
+ }
33746
+ /**
33747
+ * Tear down and reconstruct ONE device from its persisted rows — the
33748
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33749
+ * and no other device this provider owns is disturbed.
33750
+ *
33751
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33752
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33753
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33754
+ * whatever number the row carries NOW. The teardown is `decommission` —
33755
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33756
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33757
+ * the boot restore's own `create()` path, including its pass 2: first-class
33758
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33759
+ * parent by the cascade and must be re-created explicitly, because only
33760
+ * accessory children come back through `getAccessoryChildren()`.
33761
+ *
33762
+ * Reloading an accessory child directly is refused (no device class) —
33763
+ * reload its parent instead.
33764
+ */
33765
+ async reloadDevice(input) {
33766
+ const { stableId } = input;
33767
+ const devices = this.ctx.kernel.devices;
33768
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33769
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33770
+ if (live) await devices.decommission(live.id);
33771
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33772
+ addonId: this.addonId,
33773
+ stableId
33774
+ });
33775
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33776
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33777
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33778
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33779
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33780
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33781
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33782
+ for (const row of rows) {
33783
+ if (row.parentDeviceId !== id) continue;
33784
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33785
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33786
+ if (!ChildClass) continue;
33787
+ try {
33788
+ await devices.create(row.stableId, ChildClass, {}, id);
33789
+ } catch (err) {
33790
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33791
+ tags: {
33792
+ deviceId: row.id,
33793
+ stableId: row.stableId
33794
+ },
33795
+ meta: {
33796
+ parentDeviceId: id,
33797
+ error: err instanceof Error ? err.message : String(err)
33798
+ }
33799
+ });
33800
+ }
33801
+ }
33802
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33803
+ tags: { deviceId: id },
33804
+ meta: {
33805
+ stableId,
33806
+ type: meta.type
33807
+ }
33808
+ });
33809
+ return { deviceId: id };
33502
33810
  }
33503
33811
  /**
33504
33812
  * Restore devices from persisted state. Two-pass:
@@ -33524,55 +33832,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33524
33832
  * accessory-spawn flow handles via the parent's
33525
33833
  * `getAccessoryChildren()`. Override only when the default doesn't
33526
33834
  * fit.
33835
+ *
33836
+ * A row that fails either pass is NOT terminal (D347): it is handed
33837
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33838
+ * Only after the bound is exhausted is the device marked permanently
33839
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33840
+ * `getStatus().error`.
33527
33841
  */
33528
33842
  async onRestoreDevices(savedDevices) {
33529
33843
  const restored = /* @__PURE__ */ new Set();
33844
+ const failures = [];
33845
+ const attemptRestore = async (saved) => {
33846
+ if (restored.has(saved.id)) return;
33847
+ const Class = this.deviceClasses[saved.type];
33848
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33849
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33850
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33851
+ restored.add(saved.id);
33852
+ };
33530
33853
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33531
33854
  const restoreOne = async (saved) => {
33532
- const Class = this.deviceClasses[saved.type];
33533
- if (!Class) {
33855
+ if (!this.deviceClasses[saved.type]) {
33534
33856
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33535
- tags: { stableId: saved.stableId },
33857
+ tags: {
33858
+ deviceId: saved.id,
33859
+ stableId: saved.stableId
33860
+ },
33536
33861
  meta: { type: saved.type }
33537
33862
  });
33538
33863
  return;
33539
33864
  }
33540
33865
  try {
33541
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33542
- restored.add(saved.id);
33866
+ await attemptRestore(saved);
33543
33867
  } catch (err) {
33544
- this.ctx.logger.warn("Failed to restore device", {
33545
- tags: { stableId: saved.stableId },
33868
+ const error = err instanceof Error ? err.message : String(err);
33869
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33870
+ tags: {
33871
+ deviceId: saved.id,
33872
+ stableId: saved.stableId
33873
+ },
33546
33874
  meta: {
33547
33875
  type: saved.type,
33548
- error: err instanceof Error ? err.message : String(err)
33876
+ attempt: 1,
33877
+ error
33549
33878
  }
33550
33879
  });
33880
+ failures.push({
33881
+ saved,
33882
+ error
33883
+ });
33551
33884
  }
33552
33885
  };
33553
33886
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33887
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33554
33888
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33555
33889
  for (const saved of childRows) {
33556
- const Class = this.deviceClasses[saved.type];
33557
- if (!Class) continue;
33890
+ if (!this.deviceClasses[saved.type]) continue;
33558
33891
  if (saved.parentDeviceId === null) continue;
33559
- if (!restored.has(saved.parentDeviceId)) continue;
33560
- try {
33561
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33562
- restored.add(saved.id);
33563
- } catch (err) {
33564
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33892
+ if (restored.has(saved.parentDeviceId)) {
33893
+ try {
33894
+ await attemptRestore(saved);
33895
+ } catch (err) {
33896
+ const error = err instanceof Error ? err.message : String(err);
33897
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33898
+ tags: {
33899
+ deviceId: saved.id,
33900
+ stableId: saved.stableId,
33901
+ parentDeviceId: saved.parentDeviceId
33902
+ },
33903
+ meta: {
33904
+ type: saved.type,
33905
+ attempt: 1,
33906
+ error
33907
+ }
33908
+ });
33909
+ failures.push({
33910
+ saved,
33911
+ error
33912
+ });
33913
+ }
33914
+ continue;
33915
+ }
33916
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33917
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33565
33918
  tags: {
33919
+ deviceId: saved.id,
33566
33920
  stableId: saved.stableId,
33567
33921
  parentDeviceId: saved.parentDeviceId
33568
33922
  },
33569
- meta: {
33570
- type: saved.type,
33571
- error: err instanceof Error ? err.message : String(err)
33572
- }
33923
+ meta: { type: saved.type }
33573
33924
  });
33925
+ failures.push({
33926
+ saved,
33927
+ error: `parent device ${saved.parentDeviceId} not restored`
33928
+ });
33929
+ continue;
33574
33930
  }
33575
33931
  }
33932
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33933
+ return {
33934
+ restoredCount: restored.size,
33935
+ failedCount: failures.length
33936
+ };
33576
33937
  }
33577
33938
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33578
33939
  toSummary(device) {
@@ -35335,6 +35696,12 @@ Object.freeze({
35335
35696
  addonId: null,
35336
35697
  access: "view"
35337
35698
  },
35699
+ "deviceProvider.reloadDevice": {
35700
+ capName: "device-provider",
35701
+ capScope: "system",
35702
+ addonId: null,
35703
+ access: "create"
35704
+ },
35338
35705
  "deviceProvider.start": {
35339
35706
  capName: "device-provider",
35340
35707
  capScope: "system",
package/dist/addon.mjs CHANGED
@@ -13648,6 +13648,35 @@ var deviceProviderCapability = {
13648
13648
  name: string(),
13649
13649
  type: string()
13650
13650
  }))),
13651
+ /**
13652
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13653
+ * touching no other device this provider owns.
13654
+ *
13655
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13656
+ * migrated numbers: after `swapIds` the runner's live instance still
13657
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13658
+ * registrations and its log tags), and a live object cannot be renumbered.
13659
+ * Before this method the only flush was restarting the whole owning addon
13660
+ * — which took every camera the provider owns down with it (28 devices
13661
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13662
+ * same day ~27 devices' native caps did not come back on their own).
13663
+ *
13664
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13665
+ * that changes. The reply carries the id the device answers on NOW.
13666
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13667
+ * instance (if any), then re-create from the persisted row: the same
13668
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13669
+ * An RPC, never an event: a dropped event would leave the runner writing
13670
+ * against the wrong camera (D8).
13671
+ *
13672
+ * Construction can dial hardware, and the migrated source is
13673
+ * characteristically dead — the timeout covers a full activate window
13674
+ * rather than the 60 s default.
13675
+ */
13676
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13677
+ kind: "mutation",
13678
+ timeoutMs: 3 * 6e4
13679
+ }),
13651
13680
  supportsDiscovery: method(object({}), boolean()),
13652
13681
  /**
13653
13682
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13975,7 +14004,8 @@ method(object({
13975
14004
  targetId: number()
13976
14005
  }), MigrateDeviceResultSchema, {
13977
14006
  kind: "mutation",
13978
- auth: "admin"
14007
+ auth: "admin",
14008
+ timeoutMs: 12 * 6e4
13979
14009
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13980
14010
  deviceId: number(),
13981
14011
  name: string()
@@ -33339,6 +33369,147 @@ var BaseDevice = class {
33339
33369
  }
33340
33370
  };
33341
33371
  /**
33372
+ * Delays before retry rounds 1..N — the round count IS the bound.
33373
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33374
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33375
+ * per attempt) covers a device-manager lock held for minutes — the
33376
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33377
+ */
33378
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33379
+ 1e4,
33380
+ 3e4,
33381
+ 9e4
33382
+ ];
33383
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33384
+ function sleep$1(ms, signal) {
33385
+ return new Promise((resolve) => {
33386
+ if (signal.aborted) {
33387
+ resolve();
33388
+ return;
33389
+ }
33390
+ const onAbort = () => {
33391
+ clearTimeout(timer);
33392
+ resolve();
33393
+ };
33394
+ const timer = setTimeout(() => {
33395
+ signal.removeEventListener("abort", onAbort);
33396
+ resolve();
33397
+ }, ms);
33398
+ timer.unref?.();
33399
+ signal.addEventListener("abort", onAbort, { once: true });
33400
+ });
33401
+ }
33402
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33403
+ * not reject (callers wrap their own try/catch). */
33404
+ async function runWithConcurrency(items, width, fn) {
33405
+ const queue = [...items];
33406
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33407
+ const lane = async () => {
33408
+ for (;;) {
33409
+ const item = queue.shift();
33410
+ if (item === void 0) return;
33411
+ await fn(item);
33412
+ }
33413
+ };
33414
+ await Promise.all(Array.from({ length: laneCount }, lane));
33415
+ }
33416
+ var DeviceRestoreRetryScheduler = class {
33417
+ #logger;
33418
+ #attempt;
33419
+ #onPermanentFailure;
33420
+ #delaysMs;
33421
+ #concurrency;
33422
+ #now;
33423
+ #abort = new AbortController();
33424
+ constructor(options) {
33425
+ this.#logger = options.logger;
33426
+ this.#attempt = options.attempt;
33427
+ this.#onPermanentFailure = options.onPermanentFailure;
33428
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33429
+ this.#concurrency = options.concurrency ?? 4;
33430
+ this.#now = options.now ?? Date.now;
33431
+ }
33432
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33433
+ * permanently failed — the next boot restores them from disk. */
33434
+ cancel() {
33435
+ this.#abort.abort();
33436
+ }
33437
+ /**
33438
+ * Run the bounded retry rounds. Resolves when every entry has either
33439
+ * restored, been marked permanently failed, or the scheduler was
33440
+ * cancelled. Never rejects.
33441
+ */
33442
+ async run(initialFailures) {
33443
+ let pending = initialFailures.map((failure) => ({
33444
+ saved: failure.saved,
33445
+ lastError: failure.error,
33446
+ attempts: 1
33447
+ }));
33448
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33449
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33450
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33451
+ if (this.#abort.signal.aborted) break;
33452
+ pending = await this.#runRound(pending, round);
33453
+ }
33454
+ if (this.#abort.signal.aborted) return [];
33455
+ const terminal = pending.map((entry) => ({
33456
+ deviceId: entry.saved.id,
33457
+ stableId: entry.saved.stableId,
33458
+ type: String(entry.saved.type),
33459
+ attempts: entry.attempts,
33460
+ lastError: entry.lastError,
33461
+ failedAt: this.#now()
33462
+ }));
33463
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33464
+ return terminal;
33465
+ }
33466
+ /** One retry round: parents first (phase 0), then hub-adopted
33467
+ * children (phase 1) — a child's attempt depends on its parent
33468
+ * having landed, exactly like the initial two-pass restore. */
33469
+ async #runRound(pending, round) {
33470
+ const next = [];
33471
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33472
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33473
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33474
+ if (this.#abort.signal.aborted) {
33475
+ next.push(entry);
33476
+ return;
33477
+ }
33478
+ const attemptNo = entry.attempts + 1;
33479
+ try {
33480
+ await this.#attempt(entry.saved);
33481
+ this.#logger.info("Device restored on retry", {
33482
+ tags: {
33483
+ deviceId: entry.saved.id,
33484
+ stableId: entry.saved.stableId
33485
+ },
33486
+ meta: { attempt: attemptNo }
33487
+ });
33488
+ } catch (err) {
33489
+ const lastError = err instanceof Error ? err.message : String(err);
33490
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33491
+ this.#logger.warn("Device restore retry failed", {
33492
+ tags: {
33493
+ deviceId: entry.saved.id,
33494
+ stableId: entry.saved.stableId
33495
+ },
33496
+ meta: {
33497
+ attempt: attemptNo,
33498
+ remainingRetries,
33499
+ error: lastError
33500
+ }
33501
+ });
33502
+ next.push({
33503
+ saved: entry.saved,
33504
+ lastError,
33505
+ attempts: attemptNo
33506
+ });
33507
+ }
33508
+ });
33509
+ return next;
33510
+ }
33511
+ };
33512
+ /**
33342
33513
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33343
33514
  * device-provider cap router. Shared across all providers.
33344
33515
  */
@@ -33387,6 +33558,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33387
33558
  }];
33388
33559
  }
33389
33560
  async onShutdown() {
33561
+ this.cancelRestoreRetries();
33390
33562
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33391
33563
  for (const device of devices) try {
33392
33564
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33404,9 +33576,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33404
33576
  async start() {}
33405
33577
  async stop() {}
33406
33578
  async getStatus() {
33579
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33580
+ const summary = this.restoreFailureSummary();
33581
+ if (summary === null) return {
33582
+ connected: true,
33583
+ deviceCount: all.length
33584
+ };
33407
33585
  return {
33408
33586
  connected: true,
33409
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33587
+ deviceCount: all.length,
33588
+ error: summary
33410
33589
  };
33411
33590
  }
33412
33591
  async getDevices() {
@@ -33496,8 +33675,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33496
33675
  };
33497
33676
  }
33498
33677
  async restoreDevices(savedDevices) {
33499
- await this.onRestoreDevices(savedDevices);
33500
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33678
+ const report = await this.onRestoreDevices(savedDevices);
33679
+ if (savedDevices.length === 0) return;
33680
+ if (report && report.failedCount > 0) {
33681
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33682
+ return;
33683
+ }
33684
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33685
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33686
+ }
33687
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33688
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33689
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33690
+ * never re-stampede full-width while the initial pass does (D167). */
33691
+ restoreRetryConcurrency = 4;
33692
+ _restoreRetryScheduler = null;
33693
+ _restoreRetryCompletion = null;
33694
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33695
+ /** Settles when the background retry rounds finish (or `null` when
33696
+ * nothing failed). Exposed for tests and subclass diagnostics —
33697
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33698
+ * with the devices that restored, and a late success is announced
33699
+ * through the `native-cap-change` → `updateCaps` path. */
33700
+ get restoreRetryCompletion() {
33701
+ return this._restoreRetryCompletion;
33702
+ }
33703
+ /** Devices that exhausted the retry bound this process lifetime. */
33704
+ get permanentRestoreFailures() {
33705
+ return [...this._permanentRestoreFailures.values()];
33706
+ }
33707
+ /** One-line operator-facing summary for `getStatus().error`, or
33708
+ * `null` when every device restored. */
33709
+ restoreFailureSummary() {
33710
+ if (this._permanentRestoreFailures.size === 0) return null;
33711
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33712
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33713
+ }
33714
+ cancelRestoreRetries() {
33715
+ this._restoreRetryScheduler?.cancel();
33716
+ this._restoreRetryScheduler = null;
33717
+ }
33718
+ recordPermanentRestoreFailure(failure) {
33719
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33720
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33721
+ tags: {
33722
+ deviceId: failure.deviceId,
33723
+ stableId: failure.stableId
33724
+ },
33725
+ meta: {
33726
+ type: failure.type,
33727
+ attempts: failure.attempts,
33728
+ error: failure.lastError
33729
+ }
33730
+ });
33731
+ }
33732
+ scheduleRestoreRetries(failures, attempt) {
33733
+ const scheduler = new DeviceRestoreRetryScheduler({
33734
+ logger: this.ctx.logger,
33735
+ delaysMs: this.restoreRetryDelaysMs,
33736
+ concurrency: this.restoreRetryConcurrency,
33737
+ attempt,
33738
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33739
+ });
33740
+ this._restoreRetryScheduler = scheduler;
33741
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33742
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33743
+ });
33744
+ }
33745
+ /**
33746
+ * Tear down and reconstruct ONE device from its persisted rows — the
33747
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33748
+ * and no other device this provider owns is disturbed.
33749
+ *
33750
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33751
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33752
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33753
+ * whatever number the row carries NOW. The teardown is `decommission` —
33754
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33755
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33756
+ * the boot restore's own `create()` path, including its pass 2: first-class
33757
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33758
+ * parent by the cascade and must be re-created explicitly, because only
33759
+ * accessory children come back through `getAccessoryChildren()`.
33760
+ *
33761
+ * Reloading an accessory child directly is refused (no device class) —
33762
+ * reload its parent instead.
33763
+ */
33764
+ async reloadDevice(input) {
33765
+ const { stableId } = input;
33766
+ const devices = this.ctx.kernel.devices;
33767
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33768
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33769
+ if (live) await devices.decommission(live.id);
33770
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33771
+ addonId: this.addonId,
33772
+ stableId
33773
+ });
33774
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33775
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33776
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33777
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33778
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33779
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33780
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33781
+ for (const row of rows) {
33782
+ if (row.parentDeviceId !== id) continue;
33783
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33784
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33785
+ if (!ChildClass) continue;
33786
+ try {
33787
+ await devices.create(row.stableId, ChildClass, {}, id);
33788
+ } catch (err) {
33789
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33790
+ tags: {
33791
+ deviceId: row.id,
33792
+ stableId: row.stableId
33793
+ },
33794
+ meta: {
33795
+ parentDeviceId: id,
33796
+ error: err instanceof Error ? err.message : String(err)
33797
+ }
33798
+ });
33799
+ }
33800
+ }
33801
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33802
+ tags: { deviceId: id },
33803
+ meta: {
33804
+ stableId,
33805
+ type: meta.type
33806
+ }
33807
+ });
33808
+ return { deviceId: id };
33501
33809
  }
33502
33810
  /**
33503
33811
  * Restore devices from persisted state. Two-pass:
@@ -33523,55 +33831,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33523
33831
  * accessory-spawn flow handles via the parent's
33524
33832
  * `getAccessoryChildren()`. Override only when the default doesn't
33525
33833
  * fit.
33834
+ *
33835
+ * A row that fails either pass is NOT terminal (D347): it is handed
33836
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33837
+ * Only after the bound is exhausted is the device marked permanently
33838
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33839
+ * `getStatus().error`.
33526
33840
  */
33527
33841
  async onRestoreDevices(savedDevices) {
33528
33842
  const restored = /* @__PURE__ */ new Set();
33843
+ const failures = [];
33844
+ const attemptRestore = async (saved) => {
33845
+ if (restored.has(saved.id)) return;
33846
+ const Class = this.deviceClasses[saved.type];
33847
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33848
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33849
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33850
+ restored.add(saved.id);
33851
+ };
33529
33852
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33530
33853
  const restoreOne = async (saved) => {
33531
- const Class = this.deviceClasses[saved.type];
33532
- if (!Class) {
33854
+ if (!this.deviceClasses[saved.type]) {
33533
33855
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33534
- tags: { stableId: saved.stableId },
33856
+ tags: {
33857
+ deviceId: saved.id,
33858
+ stableId: saved.stableId
33859
+ },
33535
33860
  meta: { type: saved.type }
33536
33861
  });
33537
33862
  return;
33538
33863
  }
33539
33864
  try {
33540
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33541
- restored.add(saved.id);
33865
+ await attemptRestore(saved);
33542
33866
  } catch (err) {
33543
- this.ctx.logger.warn("Failed to restore device", {
33544
- tags: { stableId: saved.stableId },
33867
+ const error = err instanceof Error ? err.message : String(err);
33868
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33869
+ tags: {
33870
+ deviceId: saved.id,
33871
+ stableId: saved.stableId
33872
+ },
33545
33873
  meta: {
33546
33874
  type: saved.type,
33547
- error: err instanceof Error ? err.message : String(err)
33875
+ attempt: 1,
33876
+ error
33548
33877
  }
33549
33878
  });
33879
+ failures.push({
33880
+ saved,
33881
+ error
33882
+ });
33550
33883
  }
33551
33884
  };
33552
33885
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33886
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33553
33887
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33554
33888
  for (const saved of childRows) {
33555
- const Class = this.deviceClasses[saved.type];
33556
- if (!Class) continue;
33889
+ if (!this.deviceClasses[saved.type]) continue;
33557
33890
  if (saved.parentDeviceId === null) continue;
33558
- if (!restored.has(saved.parentDeviceId)) continue;
33559
- try {
33560
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33561
- restored.add(saved.id);
33562
- } catch (err) {
33563
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33891
+ if (restored.has(saved.parentDeviceId)) {
33892
+ try {
33893
+ await attemptRestore(saved);
33894
+ } catch (err) {
33895
+ const error = err instanceof Error ? err.message : String(err);
33896
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33897
+ tags: {
33898
+ deviceId: saved.id,
33899
+ stableId: saved.stableId,
33900
+ parentDeviceId: saved.parentDeviceId
33901
+ },
33902
+ meta: {
33903
+ type: saved.type,
33904
+ attempt: 1,
33905
+ error
33906
+ }
33907
+ });
33908
+ failures.push({
33909
+ saved,
33910
+ error
33911
+ });
33912
+ }
33913
+ continue;
33914
+ }
33915
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33916
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33564
33917
  tags: {
33918
+ deviceId: saved.id,
33565
33919
  stableId: saved.stableId,
33566
33920
  parentDeviceId: saved.parentDeviceId
33567
33921
  },
33568
- meta: {
33569
- type: saved.type,
33570
- error: err instanceof Error ? err.message : String(err)
33571
- }
33922
+ meta: { type: saved.type }
33572
33923
  });
33924
+ failures.push({
33925
+ saved,
33926
+ error: `parent device ${saved.parentDeviceId} not restored`
33927
+ });
33928
+ continue;
33573
33929
  }
33574
33930
  }
33931
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33932
+ return {
33933
+ restoredCount: restored.size,
33934
+ failedCount: failures.length
33935
+ };
33575
33936
  }
33576
33937
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33577
33938
  toSummary(device) {
@@ -35334,6 +35695,12 @@ Object.freeze({
35334
35695
  addonId: null,
35335
35696
  access: "view"
35336
35697
  },
35698
+ "deviceProvider.reloadDevice": {
35699
+ capName: "device-provider",
35700
+ capScope: "system",
35701
+ addonId: null,
35702
+ access: "create"
35703
+ },
35337
35704
  "deviceProvider.start": {
35338
35705
  capName: "device-provider",
35339
35706
  capScope: "system",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-tuya",
3
- "version": "0.2.59",
3
+ "version": "0.2.60",
4
4
  "description": "Tuya / Smart Life device-provider addon for CamStack — account-onboarded (Tuya IoT cloud fetch of device localKeys) + LOCAL DP control via the @apocaliss92/nodetuya encrypted-LAN client, exposing switch / water-heater-family kettle entities",
5
5
  "keywords": [
6
6
  "camstack",