@camstack/addon-provider-tuya 0.2.58 → 0.2.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -11347,6 +11347,89 @@ var LocationStatSchema = object({
11347
11347
  fileCount: number(),
11348
11348
  present: boolean()
11349
11349
  });
11350
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
11351
+ var BackupRunStateSchema = _enum([
11352
+ "queued",
11353
+ "running",
11354
+ "succeeded",
11355
+ "failed",
11356
+ "cancelled"
11357
+ ]);
11358
+ /**
11359
+ * Where a running backup currently is. `queued` before it starts,
11360
+ * `building` while the tar.gz is being staged, `uploading` during the
11361
+ * per-destination fan-out, `done` once terminal.
11362
+ */
11363
+ var BackupRunPhaseSchema = _enum([
11364
+ "queued",
11365
+ "building",
11366
+ "uploading",
11367
+ "done"
11368
+ ]);
11369
+ /**
11370
+ * Observable state of one backup run — readable WHILE it runs via
11371
+ * `backup.listRuns`. This is what makes the execution queue and
11372
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
11373
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
11374
+ * diagnosable with `du` because nothing reported that runs existed or
11375
+ * how large the staged archive had grown.
11376
+ */
11377
+ var BackupRunSchema = object({
11378
+ /** Stable run id — the handle `backup.cancel` takes. */
11379
+ id: string(),
11380
+ state: BackupRunStateSchema,
11381
+ phase: BackupRunPhaseSchema,
11382
+ /**
11383
+ * Resolved destination location ids. Empty while queued (targets are
11384
+ * resolved when the run starts, against the then-current policies).
11385
+ */
11386
+ destinationIds: array(string()).readonly(),
11387
+ label: string().optional(),
11388
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
11389
+ requestedAt: number(),
11390
+ /** ms-epoch when the run left the queue and started building. */
11391
+ startedAt: number().optional(),
11392
+ /** ms-epoch when the run reached a terminal state. */
11393
+ finishedAt: number().optional(),
11394
+ /** Compressed bytes of the staging archive written so far. */
11395
+ stagedBytes: number(),
11396
+ /** Final staged archive size, once the build phase completes. */
11397
+ archiveSizeBytes: number().optional(),
11398
+ /** Bytes pushed to the destination currently uploading. */
11399
+ uploadedBytes: number(),
11400
+ /** Destinations where the archive fully landed (uploaded + indexed). */
11401
+ completedDestinationIds: array(string()).readonly(),
11402
+ /** Destinations that failed during the fan-out. */
11403
+ failedDestinationIds: array(string()).readonly(),
11404
+ /** Failure message when `state === 'failed'`. */
11405
+ error: string().optional(),
11406
+ /**
11407
+ * 1-based place in the execution queue — 1 = runs next. Present only
11408
+ * while `state === 'queued'`. Stamped by the orchestrator from the
11409
+ * queue's OWN pending order, never derived from timestamps, so the
11410
+ * UI cannot show an order the executor will not honour.
11411
+ */
11412
+ queuePosition: number().int().min(1).optional()
11413
+ });
11414
+ /**
11415
+ * Result of `backup.trigger`. The call still resolves when the run
11416
+ * terminates (compat with schedule-driven runs and the admin UI), but
11417
+ * it now names the run and says whether it had to WAIT: a trigger that
11418
+ * arrives while another run is in flight is enqueued (or joined onto
11419
+ * an identical already-queued run), never started concurrently.
11420
+ */
11421
+ var BackupTriggerResultSchema = object({
11422
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
11423
+ runId: string(),
11424
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
11425
+ queued: boolean(),
11426
+ /** True when this trigger was coalesced onto an identical already-queued run. */
11427
+ joined: boolean(),
11428
+ /** True when the run was cancelled before completing every destination. */
11429
+ cancelled: boolean(),
11430
+ /** One entry per destination the archive landed at (partial on cancel). */
11431
+ entries: array(BackupEntrySchema).readonly()
11432
+ });
11350
11433
  /**
11351
11434
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
11352
11435
  * SET of destination locations. Supersedes the per-location cron on
@@ -11394,7 +11477,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
11394
11477
  * retention (manual runs).
11395
11478
  */
11396
11479
  retentionCount: number().int().min(1).max(1e3).optional()
11397
- }).optional(), array(BackupEntrySchema).readonly(), {
11480
+ }).optional(), BackupTriggerResultSchema, {
11481
+ kind: "mutation",
11482
+ auth: "admin"
11483
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
11398
11484
  kind: "mutation",
11399
11485
  auth: "admin"
11400
11486
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -12111,6 +12197,14 @@ method(object({
12111
12197
  }), object({ success: literal(true) }), {
12112
12198
  kind: "mutation",
12113
12199
  auth: "admin"
12200
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
12201
+ derivedStreamsDeleted: array(string()).readonly(),
12202
+ assignmentsPurged: boolean(),
12203
+ probeSnapshotsDropped: number().int().nonnegative(),
12204
+ rtspTokenRowsDeleted: number().int().nonnegative()
12205
+ }), {
12206
+ kind: "mutation",
12207
+ auth: "admin"
12114
12208
  }), method(object({
12115
12209
  deviceId: number(),
12116
12210
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13555,6 +13649,35 @@ var deviceProviderCapability = {
13555
13649
  name: string(),
13556
13650
  type: string()
13557
13651
  }))),
13652
+ /**
13653
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13654
+ * touching no other device this provider owns.
13655
+ *
13656
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13657
+ * migrated numbers: after `swapIds` the runner's live instance still
13658
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13659
+ * registrations and its log tags), and a live object cannot be renumbered.
13660
+ * Before this method the only flush was restarting the whole owning addon
13661
+ * — which took every camera the provider owns down with it (28 devices
13662
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13663
+ * same day ~27 devices' native caps did not come back on their own).
13664
+ *
13665
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13666
+ * that changes. The reply carries the id the device answers on NOW.
13667
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13668
+ * instance (if any), then re-create from the persisted row: the same
13669
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13670
+ * An RPC, never an event: a dropped event would leave the runner writing
13671
+ * against the wrong camera (D8).
13672
+ *
13673
+ * Construction can dial hardware, and the migrated source is
13674
+ * characteristically dead — the timeout covers a full activate window
13675
+ * rather than the 60 s default.
13676
+ */
13677
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13678
+ kind: "mutation",
13679
+ timeoutMs: 3 * 6e4
13680
+ }),
13558
13681
  supportsDiscovery: method(object({}), boolean()),
13559
13682
  /**
13560
13683
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13882,7 +14005,8 @@ method(object({
13882
14005
  targetId: number()
13883
14006
  }), MigrateDeviceResultSchema, {
13884
14007
  kind: "mutation",
13885
- auth: "admin"
14008
+ auth: "admin",
14009
+ timeoutMs: 12 * 6e4
13886
14010
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13887
14011
  deviceId: number(),
13888
14012
  name: string()
@@ -33246,6 +33370,147 @@ var BaseDevice = class {
33246
33370
  }
33247
33371
  };
33248
33372
  /**
33373
+ * Delays before retry rounds 1..N — the round count IS the bound.
33374
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33375
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33376
+ * per attempt) covers a device-manager lock held for minutes — the
33377
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33378
+ */
33379
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33380
+ 1e4,
33381
+ 3e4,
33382
+ 9e4
33383
+ ];
33384
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33385
+ function sleep$1(ms, signal) {
33386
+ return new Promise((resolve) => {
33387
+ if (signal.aborted) {
33388
+ resolve();
33389
+ return;
33390
+ }
33391
+ const onAbort = () => {
33392
+ clearTimeout(timer);
33393
+ resolve();
33394
+ };
33395
+ const timer = setTimeout(() => {
33396
+ signal.removeEventListener("abort", onAbort);
33397
+ resolve();
33398
+ }, ms);
33399
+ timer.unref?.();
33400
+ signal.addEventListener("abort", onAbort, { once: true });
33401
+ });
33402
+ }
33403
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33404
+ * not reject (callers wrap their own try/catch). */
33405
+ async function runWithConcurrency(items, width, fn) {
33406
+ const queue = [...items];
33407
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33408
+ const lane = async () => {
33409
+ for (;;) {
33410
+ const item = queue.shift();
33411
+ if (item === void 0) return;
33412
+ await fn(item);
33413
+ }
33414
+ };
33415
+ await Promise.all(Array.from({ length: laneCount }, lane));
33416
+ }
33417
+ var DeviceRestoreRetryScheduler = class {
33418
+ #logger;
33419
+ #attempt;
33420
+ #onPermanentFailure;
33421
+ #delaysMs;
33422
+ #concurrency;
33423
+ #now;
33424
+ #abort = new AbortController();
33425
+ constructor(options) {
33426
+ this.#logger = options.logger;
33427
+ this.#attempt = options.attempt;
33428
+ this.#onPermanentFailure = options.onPermanentFailure;
33429
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33430
+ this.#concurrency = options.concurrency ?? 4;
33431
+ this.#now = options.now ?? Date.now;
33432
+ }
33433
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33434
+ * permanently failed — the next boot restores them from disk. */
33435
+ cancel() {
33436
+ this.#abort.abort();
33437
+ }
33438
+ /**
33439
+ * Run the bounded retry rounds. Resolves when every entry has either
33440
+ * restored, been marked permanently failed, or the scheduler was
33441
+ * cancelled. Never rejects.
33442
+ */
33443
+ async run(initialFailures) {
33444
+ let pending = initialFailures.map((failure) => ({
33445
+ saved: failure.saved,
33446
+ lastError: failure.error,
33447
+ attempts: 1
33448
+ }));
33449
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33450
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33451
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33452
+ if (this.#abort.signal.aborted) break;
33453
+ pending = await this.#runRound(pending, round);
33454
+ }
33455
+ if (this.#abort.signal.aborted) return [];
33456
+ const terminal = pending.map((entry) => ({
33457
+ deviceId: entry.saved.id,
33458
+ stableId: entry.saved.stableId,
33459
+ type: String(entry.saved.type),
33460
+ attempts: entry.attempts,
33461
+ lastError: entry.lastError,
33462
+ failedAt: this.#now()
33463
+ }));
33464
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33465
+ return terminal;
33466
+ }
33467
+ /** One retry round: parents first (phase 0), then hub-adopted
33468
+ * children (phase 1) — a child's attempt depends on its parent
33469
+ * having landed, exactly like the initial two-pass restore. */
33470
+ async #runRound(pending, round) {
33471
+ const next = [];
33472
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33473
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33474
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33475
+ if (this.#abort.signal.aborted) {
33476
+ next.push(entry);
33477
+ return;
33478
+ }
33479
+ const attemptNo = entry.attempts + 1;
33480
+ try {
33481
+ await this.#attempt(entry.saved);
33482
+ this.#logger.info("Device restored on retry", {
33483
+ tags: {
33484
+ deviceId: entry.saved.id,
33485
+ stableId: entry.saved.stableId
33486
+ },
33487
+ meta: { attempt: attemptNo }
33488
+ });
33489
+ } catch (err) {
33490
+ const lastError = err instanceof Error ? err.message : String(err);
33491
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33492
+ this.#logger.warn("Device restore retry failed", {
33493
+ tags: {
33494
+ deviceId: entry.saved.id,
33495
+ stableId: entry.saved.stableId
33496
+ },
33497
+ meta: {
33498
+ attempt: attemptNo,
33499
+ remainingRetries,
33500
+ error: lastError
33501
+ }
33502
+ });
33503
+ next.push({
33504
+ saved: entry.saved,
33505
+ lastError,
33506
+ attempts: attemptNo
33507
+ });
33508
+ }
33509
+ });
33510
+ return next;
33511
+ }
33512
+ };
33513
+ /**
33249
33514
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33250
33515
  * device-provider cap router. Shared across all providers.
33251
33516
  */
@@ -33294,6 +33559,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33294
33559
  }];
33295
33560
  }
33296
33561
  async onShutdown() {
33562
+ this.cancelRestoreRetries();
33297
33563
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33298
33564
  for (const device of devices) try {
33299
33565
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33311,9 +33577,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33311
33577
  async start() {}
33312
33578
  async stop() {}
33313
33579
  async getStatus() {
33580
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33581
+ const summary = this.restoreFailureSummary();
33582
+ if (summary === null) return {
33583
+ connected: true,
33584
+ deviceCount: all.length
33585
+ };
33314
33586
  return {
33315
33587
  connected: true,
33316
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33588
+ deviceCount: all.length,
33589
+ error: summary
33317
33590
  };
33318
33591
  }
33319
33592
  async getDevices() {
@@ -33403,8 +33676,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33403
33676
  };
33404
33677
  }
33405
33678
  async restoreDevices(savedDevices) {
33406
- await this.onRestoreDevices(savedDevices);
33407
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33679
+ const report = await this.onRestoreDevices(savedDevices);
33680
+ if (savedDevices.length === 0) return;
33681
+ if (report && report.failedCount > 0) {
33682
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33683
+ return;
33684
+ }
33685
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33686
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33687
+ }
33688
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33689
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33690
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33691
+ * never re-stampede full-width while the initial pass does (D167). */
33692
+ restoreRetryConcurrency = 4;
33693
+ _restoreRetryScheduler = null;
33694
+ _restoreRetryCompletion = null;
33695
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33696
+ /** Settles when the background retry rounds finish (or `null` when
33697
+ * nothing failed). Exposed for tests and subclass diagnostics —
33698
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33699
+ * with the devices that restored, and a late success is announced
33700
+ * through the `native-cap-change` → `updateCaps` path. */
33701
+ get restoreRetryCompletion() {
33702
+ return this._restoreRetryCompletion;
33703
+ }
33704
+ /** Devices that exhausted the retry bound this process lifetime. */
33705
+ get permanentRestoreFailures() {
33706
+ return [...this._permanentRestoreFailures.values()];
33707
+ }
33708
+ /** One-line operator-facing summary for `getStatus().error`, or
33709
+ * `null` when every device restored. */
33710
+ restoreFailureSummary() {
33711
+ if (this._permanentRestoreFailures.size === 0) return null;
33712
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33713
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33714
+ }
33715
+ cancelRestoreRetries() {
33716
+ this._restoreRetryScheduler?.cancel();
33717
+ this._restoreRetryScheduler = null;
33718
+ }
33719
+ recordPermanentRestoreFailure(failure) {
33720
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33721
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33722
+ tags: {
33723
+ deviceId: failure.deviceId,
33724
+ stableId: failure.stableId
33725
+ },
33726
+ meta: {
33727
+ type: failure.type,
33728
+ attempts: failure.attempts,
33729
+ error: failure.lastError
33730
+ }
33731
+ });
33732
+ }
33733
+ scheduleRestoreRetries(failures, attempt) {
33734
+ const scheduler = new DeviceRestoreRetryScheduler({
33735
+ logger: this.ctx.logger,
33736
+ delaysMs: this.restoreRetryDelaysMs,
33737
+ concurrency: this.restoreRetryConcurrency,
33738
+ attempt,
33739
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33740
+ });
33741
+ this._restoreRetryScheduler = scheduler;
33742
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33743
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33744
+ });
33745
+ }
33746
+ /**
33747
+ * Tear down and reconstruct ONE device from its persisted rows — the
33748
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33749
+ * and no other device this provider owns is disturbed.
33750
+ *
33751
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33752
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33753
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33754
+ * whatever number the row carries NOW. The teardown is `decommission` —
33755
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33756
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33757
+ * the boot restore's own `create()` path, including its pass 2: first-class
33758
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33759
+ * parent by the cascade and must be re-created explicitly, because only
33760
+ * accessory children come back through `getAccessoryChildren()`.
33761
+ *
33762
+ * Reloading an accessory child directly is refused (no device class) —
33763
+ * reload its parent instead.
33764
+ */
33765
+ async reloadDevice(input) {
33766
+ const { stableId } = input;
33767
+ const devices = this.ctx.kernel.devices;
33768
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33769
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33770
+ if (live) await devices.decommission(live.id);
33771
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33772
+ addonId: this.addonId,
33773
+ stableId
33774
+ });
33775
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33776
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33777
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33778
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33779
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33780
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33781
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33782
+ for (const row of rows) {
33783
+ if (row.parentDeviceId !== id) continue;
33784
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33785
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33786
+ if (!ChildClass) continue;
33787
+ try {
33788
+ await devices.create(row.stableId, ChildClass, {}, id);
33789
+ } catch (err) {
33790
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33791
+ tags: {
33792
+ deviceId: row.id,
33793
+ stableId: row.stableId
33794
+ },
33795
+ meta: {
33796
+ parentDeviceId: id,
33797
+ error: err instanceof Error ? err.message : String(err)
33798
+ }
33799
+ });
33800
+ }
33801
+ }
33802
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33803
+ tags: { deviceId: id },
33804
+ meta: {
33805
+ stableId,
33806
+ type: meta.type
33807
+ }
33808
+ });
33809
+ return { deviceId: id };
33408
33810
  }
33409
33811
  /**
33410
33812
  * Restore devices from persisted state. Two-pass:
@@ -33430,55 +33832,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33430
33832
  * accessory-spawn flow handles via the parent's
33431
33833
  * `getAccessoryChildren()`. Override only when the default doesn't
33432
33834
  * fit.
33835
+ *
33836
+ * A row that fails either pass is NOT terminal (D347): it is handed
33837
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33838
+ * Only after the bound is exhausted is the device marked permanently
33839
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33840
+ * `getStatus().error`.
33433
33841
  */
33434
33842
  async onRestoreDevices(savedDevices) {
33435
33843
  const restored = /* @__PURE__ */ new Set();
33844
+ const failures = [];
33845
+ const attemptRestore = async (saved) => {
33846
+ if (restored.has(saved.id)) return;
33847
+ const Class = this.deviceClasses[saved.type];
33848
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33849
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33850
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33851
+ restored.add(saved.id);
33852
+ };
33436
33853
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33437
33854
  const restoreOne = async (saved) => {
33438
- const Class = this.deviceClasses[saved.type];
33439
- if (!Class) {
33855
+ if (!this.deviceClasses[saved.type]) {
33440
33856
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33441
- tags: { stableId: saved.stableId },
33857
+ tags: {
33858
+ deviceId: saved.id,
33859
+ stableId: saved.stableId
33860
+ },
33442
33861
  meta: { type: saved.type }
33443
33862
  });
33444
33863
  return;
33445
33864
  }
33446
33865
  try {
33447
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33448
- restored.add(saved.id);
33866
+ await attemptRestore(saved);
33449
33867
  } catch (err) {
33450
- this.ctx.logger.warn("Failed to restore device", {
33451
- tags: { stableId: saved.stableId },
33868
+ const error = err instanceof Error ? err.message : String(err);
33869
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33870
+ tags: {
33871
+ deviceId: saved.id,
33872
+ stableId: saved.stableId
33873
+ },
33452
33874
  meta: {
33453
33875
  type: saved.type,
33454
- error: err instanceof Error ? err.message : String(err)
33876
+ attempt: 1,
33877
+ error
33455
33878
  }
33456
33879
  });
33880
+ failures.push({
33881
+ saved,
33882
+ error
33883
+ });
33457
33884
  }
33458
33885
  };
33459
33886
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33887
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33460
33888
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33461
33889
  for (const saved of childRows) {
33462
- const Class = this.deviceClasses[saved.type];
33463
- if (!Class) continue;
33890
+ if (!this.deviceClasses[saved.type]) continue;
33464
33891
  if (saved.parentDeviceId === null) continue;
33465
- if (!restored.has(saved.parentDeviceId)) continue;
33466
- try {
33467
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33468
- restored.add(saved.id);
33469
- } catch (err) {
33470
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33892
+ if (restored.has(saved.parentDeviceId)) {
33893
+ try {
33894
+ await attemptRestore(saved);
33895
+ } catch (err) {
33896
+ const error = err instanceof Error ? err.message : String(err);
33897
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33898
+ tags: {
33899
+ deviceId: saved.id,
33900
+ stableId: saved.stableId,
33901
+ parentDeviceId: saved.parentDeviceId
33902
+ },
33903
+ meta: {
33904
+ type: saved.type,
33905
+ attempt: 1,
33906
+ error
33907
+ }
33908
+ });
33909
+ failures.push({
33910
+ saved,
33911
+ error
33912
+ });
33913
+ }
33914
+ continue;
33915
+ }
33916
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33917
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33471
33918
  tags: {
33919
+ deviceId: saved.id,
33472
33920
  stableId: saved.stableId,
33473
33921
  parentDeviceId: saved.parentDeviceId
33474
33922
  },
33475
- meta: {
33476
- type: saved.type,
33477
- error: err instanceof Error ? err.message : String(err)
33478
- }
33923
+ meta: { type: saved.type }
33479
33924
  });
33925
+ failures.push({
33926
+ saved,
33927
+ error: `parent device ${saved.parentDeviceId} not restored`
33928
+ });
33929
+ continue;
33480
33930
  }
33481
33931
  }
33932
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33933
+ return {
33934
+ restoredCount: restored.size,
33935
+ failedCount: failures.length
33936
+ };
33482
33937
  }
33483
33938
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33484
33939
  toSummary(device) {
@@ -33987,6 +34442,12 @@ Object.freeze({
33987
34442
  addonId: null,
33988
34443
  access: "create"
33989
34444
  },
34445
+ "backup.cancel": {
34446
+ capName: "backup",
34447
+ capScope: "system",
34448
+ addonId: null,
34449
+ access: "create"
34450
+ },
33990
34451
  "backup.delete": {
33991
34452
  capName: "backup",
33992
34453
  capScope: "system",
@@ -34029,6 +34490,12 @@ Object.freeze({
34029
34490
  addonId: null,
34030
34491
  access: "view"
34031
34492
  },
34493
+ "backup.listRuns": {
34494
+ capName: "backup",
34495
+ capScope: "system",
34496
+ addonId: null,
34497
+ access: "view"
34498
+ },
34032
34499
  "backup.listSchedules": {
34033
34500
  capName: "backup",
34034
34501
  capScope: "system",
@@ -35229,6 +35696,12 @@ Object.freeze({
35229
35696
  addonId: null,
35230
35697
  access: "view"
35231
35698
  },
35699
+ "deviceProvider.reloadDevice": {
35700
+ capName: "device-provider",
35701
+ capScope: "system",
35702
+ addonId: null,
35703
+ access: "create"
35704
+ },
35232
35705
  "deviceProvider.start": {
35233
35706
  capName: "device-provider",
35234
35707
  capScope: "system",
@@ -38595,6 +39068,12 @@ Object.freeze({
38595
39068
  addonId: null,
38596
39069
  access: "create"
38597
39070
  },
39071
+ "streamBroker.forgetDeviceHardware": {
39072
+ capName: "stream-broker",
39073
+ capScope: "system",
39074
+ addonId: null,
39075
+ access: "delete"
39076
+ },
38598
39077
  "streamBroker.getAllRtspEntries": {
38599
39078
  capName: "stream-broker",
38600
39079
  capScope: "system",
@@ -41053,6 +41532,11 @@ Object.freeze({
41053
41532
  form: "single",
41054
41533
  optional: false
41055
41534
  }],
41535
+ "streamBroker.forgetDeviceHardware": [{
41536
+ name: "deviceId",
41537
+ form: "single",
41538
+ optional: false
41539
+ }],
41056
41540
  "streamBroker.getDeviceAudioMute": [{
41057
41541
  name: "deviceId",
41058
41542
  form: "single",
package/dist/addon.mjs CHANGED
@@ -11346,6 +11346,89 @@ var LocationStatSchema = object({
11346
11346
  fileCount: number(),
11347
11347
  present: boolean()
11348
11348
  });
11349
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
11350
+ var BackupRunStateSchema = _enum([
11351
+ "queued",
11352
+ "running",
11353
+ "succeeded",
11354
+ "failed",
11355
+ "cancelled"
11356
+ ]);
11357
+ /**
11358
+ * Where a running backup currently is. `queued` before it starts,
11359
+ * `building` while the tar.gz is being staged, `uploading` during the
11360
+ * per-destination fan-out, `done` once terminal.
11361
+ */
11362
+ var BackupRunPhaseSchema = _enum([
11363
+ "queued",
11364
+ "building",
11365
+ "uploading",
11366
+ "done"
11367
+ ]);
11368
+ /**
11369
+ * Observable state of one backup run — readable WHILE it runs via
11370
+ * `backup.listRuns`. This is what makes the execution queue and
11371
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
11372
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
11373
+ * diagnosable with `du` because nothing reported that runs existed or
11374
+ * how large the staged archive had grown.
11375
+ */
11376
+ var BackupRunSchema = object({
11377
+ /** Stable run id — the handle `backup.cancel` takes. */
11378
+ id: string(),
11379
+ state: BackupRunStateSchema,
11380
+ phase: BackupRunPhaseSchema,
11381
+ /**
11382
+ * Resolved destination location ids. Empty while queued (targets are
11383
+ * resolved when the run starts, against the then-current policies).
11384
+ */
11385
+ destinationIds: array(string()).readonly(),
11386
+ label: string().optional(),
11387
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
11388
+ requestedAt: number(),
11389
+ /** ms-epoch when the run left the queue and started building. */
11390
+ startedAt: number().optional(),
11391
+ /** ms-epoch when the run reached a terminal state. */
11392
+ finishedAt: number().optional(),
11393
+ /** Compressed bytes of the staging archive written so far. */
11394
+ stagedBytes: number(),
11395
+ /** Final staged archive size, once the build phase completes. */
11396
+ archiveSizeBytes: number().optional(),
11397
+ /** Bytes pushed to the destination currently uploading. */
11398
+ uploadedBytes: number(),
11399
+ /** Destinations where the archive fully landed (uploaded + indexed). */
11400
+ completedDestinationIds: array(string()).readonly(),
11401
+ /** Destinations that failed during the fan-out. */
11402
+ failedDestinationIds: array(string()).readonly(),
11403
+ /** Failure message when `state === 'failed'`. */
11404
+ error: string().optional(),
11405
+ /**
11406
+ * 1-based place in the execution queue — 1 = runs next. Present only
11407
+ * while `state === 'queued'`. Stamped by the orchestrator from the
11408
+ * queue's OWN pending order, never derived from timestamps, so the
11409
+ * UI cannot show an order the executor will not honour.
11410
+ */
11411
+ queuePosition: number().int().min(1).optional()
11412
+ });
11413
+ /**
11414
+ * Result of `backup.trigger`. The call still resolves when the run
11415
+ * terminates (compat with schedule-driven runs and the admin UI), but
11416
+ * it now names the run and says whether it had to WAIT: a trigger that
11417
+ * arrives while another run is in flight is enqueued (or joined onto
11418
+ * an identical already-queued run), never started concurrently.
11419
+ */
11420
+ var BackupTriggerResultSchema = object({
11421
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
11422
+ runId: string(),
11423
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
11424
+ queued: boolean(),
11425
+ /** True when this trigger was coalesced onto an identical already-queued run. */
11426
+ joined: boolean(),
11427
+ /** True when the run was cancelled before completing every destination. */
11428
+ cancelled: boolean(),
11429
+ /** One entry per destination the archive landed at (partial on cancel). */
11430
+ entries: array(BackupEntrySchema).readonly()
11431
+ });
11349
11432
  /**
11350
11433
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
11351
11434
  * SET of destination locations. Supersedes the per-location cron on
@@ -11393,7 +11476,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
11393
11476
  * retention (manual runs).
11394
11477
  */
11395
11478
  retentionCount: number().int().min(1).max(1e3).optional()
11396
- }).optional(), array(BackupEntrySchema).readonly(), {
11479
+ }).optional(), BackupTriggerResultSchema, {
11480
+ kind: "mutation",
11481
+ auth: "admin"
11482
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
11397
11483
  kind: "mutation",
11398
11484
  auth: "admin"
11399
11485
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -12110,6 +12196,14 @@ method(object({
12110
12196
  }), object({ success: literal(true) }), {
12111
12197
  kind: "mutation",
12112
12198
  auth: "admin"
12199
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
12200
+ derivedStreamsDeleted: array(string()).readonly(),
12201
+ assignmentsPurged: boolean(),
12202
+ probeSnapshotsDropped: number().int().nonnegative(),
12203
+ rtspTokenRowsDeleted: number().int().nonnegative()
12204
+ }), {
12205
+ kind: "mutation",
12206
+ auth: "admin"
12113
12207
  }), method(object({
12114
12208
  deviceId: number(),
12115
12209
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13554,6 +13648,35 @@ var deviceProviderCapability = {
13554
13648
  name: string(),
13555
13649
  type: string()
13556
13650
  }))),
13651
+ /**
13652
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13653
+ * touching no other device this provider owns.
13654
+ *
13655
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13656
+ * migrated numbers: after `swapIds` the runner's live instance still
13657
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13658
+ * registrations and its log tags), and a live object cannot be renumbered.
13659
+ * Before this method the only flush was restarting the whole owning addon
13660
+ * — which took every camera the provider owns down with it (28 devices
13661
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13662
+ * same day ~27 devices' native caps did not come back on their own).
13663
+ *
13664
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13665
+ * that changes. The reply carries the id the device answers on NOW.
13666
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13667
+ * instance (if any), then re-create from the persisted row: the same
13668
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13669
+ * An RPC, never an event: a dropped event would leave the runner writing
13670
+ * against the wrong camera (D8).
13671
+ *
13672
+ * Construction can dial hardware, and the migrated source is
13673
+ * characteristically dead — the timeout covers a full activate window
13674
+ * rather than the 60 s default.
13675
+ */
13676
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13677
+ kind: "mutation",
13678
+ timeoutMs: 3 * 6e4
13679
+ }),
13557
13680
  supportsDiscovery: method(object({}), boolean()),
13558
13681
  /**
13559
13682
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13881,7 +14004,8 @@ method(object({
13881
14004
  targetId: number()
13882
14005
  }), MigrateDeviceResultSchema, {
13883
14006
  kind: "mutation",
13884
- auth: "admin"
14007
+ auth: "admin",
14008
+ timeoutMs: 12 * 6e4
13885
14009
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13886
14010
  deviceId: number(),
13887
14011
  name: string()
@@ -33245,6 +33369,147 @@ var BaseDevice = class {
33245
33369
  }
33246
33370
  };
33247
33371
  /**
33372
+ * Delays before retry rounds 1..N — the round count IS the bound.
33373
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33374
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33375
+ * per attempt) covers a device-manager lock held for minutes — the
33376
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33377
+ */
33378
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33379
+ 1e4,
33380
+ 3e4,
33381
+ 9e4
33382
+ ];
33383
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33384
+ function sleep$1(ms, signal) {
33385
+ return new Promise((resolve) => {
33386
+ if (signal.aborted) {
33387
+ resolve();
33388
+ return;
33389
+ }
33390
+ const onAbort = () => {
33391
+ clearTimeout(timer);
33392
+ resolve();
33393
+ };
33394
+ const timer = setTimeout(() => {
33395
+ signal.removeEventListener("abort", onAbort);
33396
+ resolve();
33397
+ }, ms);
33398
+ timer.unref?.();
33399
+ signal.addEventListener("abort", onAbort, { once: true });
33400
+ });
33401
+ }
33402
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33403
+ * not reject (callers wrap their own try/catch). */
33404
+ async function runWithConcurrency(items, width, fn) {
33405
+ const queue = [...items];
33406
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33407
+ const lane = async () => {
33408
+ for (;;) {
33409
+ const item = queue.shift();
33410
+ if (item === void 0) return;
33411
+ await fn(item);
33412
+ }
33413
+ };
33414
+ await Promise.all(Array.from({ length: laneCount }, lane));
33415
+ }
33416
+ var DeviceRestoreRetryScheduler = class {
33417
+ #logger;
33418
+ #attempt;
33419
+ #onPermanentFailure;
33420
+ #delaysMs;
33421
+ #concurrency;
33422
+ #now;
33423
+ #abort = new AbortController();
33424
+ constructor(options) {
33425
+ this.#logger = options.logger;
33426
+ this.#attempt = options.attempt;
33427
+ this.#onPermanentFailure = options.onPermanentFailure;
33428
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33429
+ this.#concurrency = options.concurrency ?? 4;
33430
+ this.#now = options.now ?? Date.now;
33431
+ }
33432
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33433
+ * permanently failed — the next boot restores them from disk. */
33434
+ cancel() {
33435
+ this.#abort.abort();
33436
+ }
33437
+ /**
33438
+ * Run the bounded retry rounds. Resolves when every entry has either
33439
+ * restored, been marked permanently failed, or the scheduler was
33440
+ * cancelled. Never rejects.
33441
+ */
33442
+ async run(initialFailures) {
33443
+ let pending = initialFailures.map((failure) => ({
33444
+ saved: failure.saved,
33445
+ lastError: failure.error,
33446
+ attempts: 1
33447
+ }));
33448
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33449
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33450
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33451
+ if (this.#abort.signal.aborted) break;
33452
+ pending = await this.#runRound(pending, round);
33453
+ }
33454
+ if (this.#abort.signal.aborted) return [];
33455
+ const terminal = pending.map((entry) => ({
33456
+ deviceId: entry.saved.id,
33457
+ stableId: entry.saved.stableId,
33458
+ type: String(entry.saved.type),
33459
+ attempts: entry.attempts,
33460
+ lastError: entry.lastError,
33461
+ failedAt: this.#now()
33462
+ }));
33463
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33464
+ return terminal;
33465
+ }
33466
+ /** One retry round: parents first (phase 0), then hub-adopted
33467
+ * children (phase 1) — a child's attempt depends on its parent
33468
+ * having landed, exactly like the initial two-pass restore. */
33469
+ async #runRound(pending, round) {
33470
+ const next = [];
33471
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33472
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33473
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33474
+ if (this.#abort.signal.aborted) {
33475
+ next.push(entry);
33476
+ return;
33477
+ }
33478
+ const attemptNo = entry.attempts + 1;
33479
+ try {
33480
+ await this.#attempt(entry.saved);
33481
+ this.#logger.info("Device restored on retry", {
33482
+ tags: {
33483
+ deviceId: entry.saved.id,
33484
+ stableId: entry.saved.stableId
33485
+ },
33486
+ meta: { attempt: attemptNo }
33487
+ });
33488
+ } catch (err) {
33489
+ const lastError = err instanceof Error ? err.message : String(err);
33490
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33491
+ this.#logger.warn("Device restore retry failed", {
33492
+ tags: {
33493
+ deviceId: entry.saved.id,
33494
+ stableId: entry.saved.stableId
33495
+ },
33496
+ meta: {
33497
+ attempt: attemptNo,
33498
+ remainingRetries,
33499
+ error: lastError
33500
+ }
33501
+ });
33502
+ next.push({
33503
+ saved: entry.saved,
33504
+ lastError,
33505
+ attempts: attemptNo
33506
+ });
33507
+ }
33508
+ });
33509
+ return next;
33510
+ }
33511
+ };
33512
+ /**
33248
33513
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33249
33514
  * device-provider cap router. Shared across all providers.
33250
33515
  */
@@ -33293,6 +33558,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33293
33558
  }];
33294
33559
  }
33295
33560
  async onShutdown() {
33561
+ this.cancelRestoreRetries();
33296
33562
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33297
33563
  for (const device of devices) try {
33298
33564
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33310,9 +33576,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33310
33576
  async start() {}
33311
33577
  async stop() {}
33312
33578
  async getStatus() {
33579
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33580
+ const summary = this.restoreFailureSummary();
33581
+ if (summary === null) return {
33582
+ connected: true,
33583
+ deviceCount: all.length
33584
+ };
33313
33585
  return {
33314
33586
  connected: true,
33315
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33587
+ deviceCount: all.length,
33588
+ error: summary
33316
33589
  };
33317
33590
  }
33318
33591
  async getDevices() {
@@ -33402,8 +33675,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33402
33675
  };
33403
33676
  }
33404
33677
  async restoreDevices(savedDevices) {
33405
- await this.onRestoreDevices(savedDevices);
33406
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33678
+ const report = await this.onRestoreDevices(savedDevices);
33679
+ if (savedDevices.length === 0) return;
33680
+ if (report && report.failedCount > 0) {
33681
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33682
+ return;
33683
+ }
33684
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33685
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33686
+ }
33687
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33688
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33689
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33690
+ * never re-stampede full-width while the initial pass does (D167). */
33691
+ restoreRetryConcurrency = 4;
33692
+ _restoreRetryScheduler = null;
33693
+ _restoreRetryCompletion = null;
33694
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33695
+ /** Settles when the background retry rounds finish (or `null` when
33696
+ * nothing failed). Exposed for tests and subclass diagnostics —
33697
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33698
+ * with the devices that restored, and a late success is announced
33699
+ * through the `native-cap-change` → `updateCaps` path. */
33700
+ get restoreRetryCompletion() {
33701
+ return this._restoreRetryCompletion;
33702
+ }
33703
+ /** Devices that exhausted the retry bound this process lifetime. */
33704
+ get permanentRestoreFailures() {
33705
+ return [...this._permanentRestoreFailures.values()];
33706
+ }
33707
+ /** One-line operator-facing summary for `getStatus().error`, or
33708
+ * `null` when every device restored. */
33709
+ restoreFailureSummary() {
33710
+ if (this._permanentRestoreFailures.size === 0) return null;
33711
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33712
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33713
+ }
33714
+ cancelRestoreRetries() {
33715
+ this._restoreRetryScheduler?.cancel();
33716
+ this._restoreRetryScheduler = null;
33717
+ }
33718
+ recordPermanentRestoreFailure(failure) {
33719
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33720
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33721
+ tags: {
33722
+ deviceId: failure.deviceId,
33723
+ stableId: failure.stableId
33724
+ },
33725
+ meta: {
33726
+ type: failure.type,
33727
+ attempts: failure.attempts,
33728
+ error: failure.lastError
33729
+ }
33730
+ });
33731
+ }
33732
+ scheduleRestoreRetries(failures, attempt) {
33733
+ const scheduler = new DeviceRestoreRetryScheduler({
33734
+ logger: this.ctx.logger,
33735
+ delaysMs: this.restoreRetryDelaysMs,
33736
+ concurrency: this.restoreRetryConcurrency,
33737
+ attempt,
33738
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33739
+ });
33740
+ this._restoreRetryScheduler = scheduler;
33741
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33742
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33743
+ });
33744
+ }
33745
+ /**
33746
+ * Tear down and reconstruct ONE device from its persisted rows — the
33747
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33748
+ * and no other device this provider owns is disturbed.
33749
+ *
33750
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33751
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33752
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33753
+ * whatever number the row carries NOW. The teardown is `decommission` —
33754
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33755
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33756
+ * the boot restore's own `create()` path, including its pass 2: first-class
33757
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33758
+ * parent by the cascade and must be re-created explicitly, because only
33759
+ * accessory children come back through `getAccessoryChildren()`.
33760
+ *
33761
+ * Reloading an accessory child directly is refused (no device class) —
33762
+ * reload its parent instead.
33763
+ */
33764
+ async reloadDevice(input) {
33765
+ const { stableId } = input;
33766
+ const devices = this.ctx.kernel.devices;
33767
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33768
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33769
+ if (live) await devices.decommission(live.id);
33770
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33771
+ addonId: this.addonId,
33772
+ stableId
33773
+ });
33774
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33775
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33776
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33777
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33778
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33779
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33780
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33781
+ for (const row of rows) {
33782
+ if (row.parentDeviceId !== id) continue;
33783
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33784
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33785
+ if (!ChildClass) continue;
33786
+ try {
33787
+ await devices.create(row.stableId, ChildClass, {}, id);
33788
+ } catch (err) {
33789
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33790
+ tags: {
33791
+ deviceId: row.id,
33792
+ stableId: row.stableId
33793
+ },
33794
+ meta: {
33795
+ parentDeviceId: id,
33796
+ error: err instanceof Error ? err.message : String(err)
33797
+ }
33798
+ });
33799
+ }
33800
+ }
33801
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33802
+ tags: { deviceId: id },
33803
+ meta: {
33804
+ stableId,
33805
+ type: meta.type
33806
+ }
33807
+ });
33808
+ return { deviceId: id };
33407
33809
  }
33408
33810
  /**
33409
33811
  * Restore devices from persisted state. Two-pass:
@@ -33429,55 +33831,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33429
33831
  * accessory-spawn flow handles via the parent's
33430
33832
  * `getAccessoryChildren()`. Override only when the default doesn't
33431
33833
  * fit.
33834
+ *
33835
+ * A row that fails either pass is NOT terminal (D347): it is handed
33836
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33837
+ * Only after the bound is exhausted is the device marked permanently
33838
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33839
+ * `getStatus().error`.
33432
33840
  */
33433
33841
  async onRestoreDevices(savedDevices) {
33434
33842
  const restored = /* @__PURE__ */ new Set();
33843
+ const failures = [];
33844
+ const attemptRestore = async (saved) => {
33845
+ if (restored.has(saved.id)) return;
33846
+ const Class = this.deviceClasses[saved.type];
33847
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33848
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33849
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33850
+ restored.add(saved.id);
33851
+ };
33435
33852
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33436
33853
  const restoreOne = async (saved) => {
33437
- const Class = this.deviceClasses[saved.type];
33438
- if (!Class) {
33854
+ if (!this.deviceClasses[saved.type]) {
33439
33855
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33440
- tags: { stableId: saved.stableId },
33856
+ tags: {
33857
+ deviceId: saved.id,
33858
+ stableId: saved.stableId
33859
+ },
33441
33860
  meta: { type: saved.type }
33442
33861
  });
33443
33862
  return;
33444
33863
  }
33445
33864
  try {
33446
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33447
- restored.add(saved.id);
33865
+ await attemptRestore(saved);
33448
33866
  } catch (err) {
33449
- this.ctx.logger.warn("Failed to restore device", {
33450
- tags: { stableId: saved.stableId },
33867
+ const error = err instanceof Error ? err.message : String(err);
33868
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33869
+ tags: {
33870
+ deviceId: saved.id,
33871
+ stableId: saved.stableId
33872
+ },
33451
33873
  meta: {
33452
33874
  type: saved.type,
33453
- error: err instanceof Error ? err.message : String(err)
33875
+ attempt: 1,
33876
+ error
33454
33877
  }
33455
33878
  });
33879
+ failures.push({
33880
+ saved,
33881
+ error
33882
+ });
33456
33883
  }
33457
33884
  };
33458
33885
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33886
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33459
33887
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33460
33888
  for (const saved of childRows) {
33461
- const Class = this.deviceClasses[saved.type];
33462
- if (!Class) continue;
33889
+ if (!this.deviceClasses[saved.type]) continue;
33463
33890
  if (saved.parentDeviceId === null) continue;
33464
- if (!restored.has(saved.parentDeviceId)) continue;
33465
- try {
33466
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33467
- restored.add(saved.id);
33468
- } catch (err) {
33469
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33891
+ if (restored.has(saved.parentDeviceId)) {
33892
+ try {
33893
+ await attemptRestore(saved);
33894
+ } catch (err) {
33895
+ const error = err instanceof Error ? err.message : String(err);
33896
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33897
+ tags: {
33898
+ deviceId: saved.id,
33899
+ stableId: saved.stableId,
33900
+ parentDeviceId: saved.parentDeviceId
33901
+ },
33902
+ meta: {
33903
+ type: saved.type,
33904
+ attempt: 1,
33905
+ error
33906
+ }
33907
+ });
33908
+ failures.push({
33909
+ saved,
33910
+ error
33911
+ });
33912
+ }
33913
+ continue;
33914
+ }
33915
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33916
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33470
33917
  tags: {
33918
+ deviceId: saved.id,
33471
33919
  stableId: saved.stableId,
33472
33920
  parentDeviceId: saved.parentDeviceId
33473
33921
  },
33474
- meta: {
33475
- type: saved.type,
33476
- error: err instanceof Error ? err.message : String(err)
33477
- }
33922
+ meta: { type: saved.type }
33478
33923
  });
33924
+ failures.push({
33925
+ saved,
33926
+ error: `parent device ${saved.parentDeviceId} not restored`
33927
+ });
33928
+ continue;
33479
33929
  }
33480
33930
  }
33931
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33932
+ return {
33933
+ restoredCount: restored.size,
33934
+ failedCount: failures.length
33935
+ };
33481
33936
  }
33482
33937
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33483
33938
  toSummary(device) {
@@ -33986,6 +34441,12 @@ Object.freeze({
33986
34441
  addonId: null,
33987
34442
  access: "create"
33988
34443
  },
34444
+ "backup.cancel": {
34445
+ capName: "backup",
34446
+ capScope: "system",
34447
+ addonId: null,
34448
+ access: "create"
34449
+ },
33989
34450
  "backup.delete": {
33990
34451
  capName: "backup",
33991
34452
  capScope: "system",
@@ -34028,6 +34489,12 @@ Object.freeze({
34028
34489
  addonId: null,
34029
34490
  access: "view"
34030
34491
  },
34492
+ "backup.listRuns": {
34493
+ capName: "backup",
34494
+ capScope: "system",
34495
+ addonId: null,
34496
+ access: "view"
34497
+ },
34031
34498
  "backup.listSchedules": {
34032
34499
  capName: "backup",
34033
34500
  capScope: "system",
@@ -35228,6 +35695,12 @@ Object.freeze({
35228
35695
  addonId: null,
35229
35696
  access: "view"
35230
35697
  },
35698
+ "deviceProvider.reloadDevice": {
35699
+ capName: "device-provider",
35700
+ capScope: "system",
35701
+ addonId: null,
35702
+ access: "create"
35703
+ },
35231
35704
  "deviceProvider.start": {
35232
35705
  capName: "device-provider",
35233
35706
  capScope: "system",
@@ -38594,6 +39067,12 @@ Object.freeze({
38594
39067
  addonId: null,
38595
39068
  access: "create"
38596
39069
  },
39070
+ "streamBroker.forgetDeviceHardware": {
39071
+ capName: "stream-broker",
39072
+ capScope: "system",
39073
+ addonId: null,
39074
+ access: "delete"
39075
+ },
38597
39076
  "streamBroker.getAllRtspEntries": {
38598
39077
  capName: "stream-broker",
38599
39078
  capScope: "system",
@@ -41052,6 +41531,11 @@ Object.freeze({
41052
41531
  form: "single",
41053
41532
  optional: false
41054
41533
  }],
41534
+ "streamBroker.forgetDeviceHardware": [{
41535
+ name: "deviceId",
41536
+ form: "single",
41537
+ optional: false
41538
+ }],
41055
41539
  "streamBroker.getDeviceAudioMute": [{
41056
41540
  name: "deviceId",
41057
41541
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-tuya",
3
- "version": "0.2.58",
3
+ "version": "0.2.60",
4
4
  "description": "Tuya / Smart Life device-provider addon for CamStack — account-onboarded (Tuya IoT cloud fetch of device localKeys) + LOCAL DP control via the @apocaliss92/nodetuya encrypted-LAN client, exposing switch / water-heater-family kettle entities",
5
5
  "keywords": [
6
6
  "camstack",