@camstack/addon-provider-unifi 0.2.58 → 0.2.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10527,6 +10527,89 @@ var LocationStatSchema = object({
10527
10527
  fileCount: number(),
10528
10528
  present: boolean()
10529
10529
  });
10530
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10531
+ var BackupRunStateSchema = _enum([
10532
+ "queued",
10533
+ "running",
10534
+ "succeeded",
10535
+ "failed",
10536
+ "cancelled"
10537
+ ]);
10538
+ /**
10539
+ * Where a running backup currently is. `queued` before it starts,
10540
+ * `building` while the tar.gz is being staged, `uploading` during the
10541
+ * per-destination fan-out, `done` once terminal.
10542
+ */
10543
+ var BackupRunPhaseSchema = _enum([
10544
+ "queued",
10545
+ "building",
10546
+ "uploading",
10547
+ "done"
10548
+ ]);
10549
+ /**
10550
+ * Observable state of one backup run — readable WHILE it runs via
10551
+ * `backup.listRuns`. This is what makes the execution queue and
10552
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10553
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10554
+ * diagnosable with `du` because nothing reported that runs existed or
10555
+ * how large the staged archive had grown.
10556
+ */
10557
+ var BackupRunSchema = object({
10558
+ /** Stable run id — the handle `backup.cancel` takes. */
10559
+ id: string(),
10560
+ state: BackupRunStateSchema,
10561
+ phase: BackupRunPhaseSchema,
10562
+ /**
10563
+ * Resolved destination location ids. Empty while queued (targets are
10564
+ * resolved when the run starts, against the then-current policies).
10565
+ */
10566
+ destinationIds: array(string()).readonly(),
10567
+ label: string().optional(),
10568
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10569
+ requestedAt: number(),
10570
+ /** ms-epoch when the run left the queue and started building. */
10571
+ startedAt: number().optional(),
10572
+ /** ms-epoch when the run reached a terminal state. */
10573
+ finishedAt: number().optional(),
10574
+ /** Compressed bytes of the staging archive written so far. */
10575
+ stagedBytes: number(),
10576
+ /** Final staged archive size, once the build phase completes. */
10577
+ archiveSizeBytes: number().optional(),
10578
+ /** Bytes pushed to the destination currently uploading. */
10579
+ uploadedBytes: number(),
10580
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10581
+ completedDestinationIds: array(string()).readonly(),
10582
+ /** Destinations that failed during the fan-out. */
10583
+ failedDestinationIds: array(string()).readonly(),
10584
+ /** Failure message when `state === 'failed'`. */
10585
+ error: string().optional(),
10586
+ /**
10587
+ * 1-based place in the execution queue — 1 = runs next. Present only
10588
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10589
+ * queue's OWN pending order, never derived from timestamps, so the
10590
+ * UI cannot show an order the executor will not honour.
10591
+ */
10592
+ queuePosition: number().int().min(1).optional()
10593
+ });
10594
+ /**
10595
+ * Result of `backup.trigger`. The call still resolves when the run
10596
+ * terminates (compat with schedule-driven runs and the admin UI), but
10597
+ * it now names the run and says whether it had to WAIT: a trigger that
10598
+ * arrives while another run is in flight is enqueued (or joined onto
10599
+ * an identical already-queued run), never started concurrently.
10600
+ */
10601
+ var BackupTriggerResultSchema = object({
10602
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10603
+ runId: string(),
10604
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10605
+ queued: boolean(),
10606
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10607
+ joined: boolean(),
10608
+ /** True when the run was cancelled before completing every destination. */
10609
+ cancelled: boolean(),
10610
+ /** One entry per destination the archive landed at (partial on cancel). */
10611
+ entries: array(BackupEntrySchema).readonly()
10612
+ });
10530
10613
  /**
10531
10614
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10532
10615
  * SET of destination locations. Supersedes the per-location cron on
@@ -10574,7 +10657,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10574
10657
  * retention (manual runs).
10575
10658
  */
10576
10659
  retentionCount: number().int().min(1).max(1e3).optional()
10577
- }).optional(), array(BackupEntrySchema).readonly(), {
10660
+ }).optional(), BackupTriggerResultSchema, {
10661
+ kind: "mutation",
10662
+ auth: "admin"
10663
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10578
10664
  kind: "mutation",
10579
10665
  auth: "admin"
10580
10666
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11291,6 +11377,14 @@ method(object({
11291
11377
  }), object({ success: literal(true) }), {
11292
11378
  kind: "mutation",
11293
11379
  auth: "admin"
11380
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11381
+ derivedStreamsDeleted: array(string()).readonly(),
11382
+ assignmentsPurged: boolean(),
11383
+ probeSnapshotsDropped: number().int().nonnegative(),
11384
+ rtspTokenRowsDeleted: number().int().nonnegative()
11385
+ }), {
11386
+ kind: "mutation",
11387
+ auth: "admin"
11294
11388
  }), method(object({
11295
11389
  deviceId: number(),
11296
11390
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12718,6 +12812,35 @@ var deviceProviderCapability = {
12718
12812
  name: string(),
12719
12813
  type: string()
12720
12814
  }))),
12815
+ /**
12816
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12817
+ * touching no other device this provider owns.
12818
+ *
12819
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12820
+ * migrated numbers: after `swapIds` the runner's live instance still
12821
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12822
+ * registrations and its log tags), and a live object cannot be renumbered.
12823
+ * Before this method the only flush was restarting the whole owning addon
12824
+ * — which took every camera the provider owns down with it (28 devices
12825
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12826
+ * same day ~27 devices' native caps did not come back on their own).
12827
+ *
12828
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12829
+ * that changes. The reply carries the id the device answers on NOW.
12830
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12831
+ * instance (if any), then re-create from the persisted row: the same
12832
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12833
+ * An RPC, never an event: a dropped event would leave the runner writing
12834
+ * against the wrong camera (D8).
12835
+ *
12836
+ * Construction can dial hardware, and the migrated source is
12837
+ * characteristically dead — the timeout covers a full activate window
12838
+ * rather than the 60 s default.
12839
+ */
12840
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12841
+ kind: "mutation",
12842
+ timeoutMs: 3 * 6e4
12843
+ }),
12721
12844
  supportsDiscovery: method(object({}), boolean()),
12722
12845
  /**
12723
12846
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13045,7 +13168,8 @@ method(object({
13045
13168
  targetId: number()
13046
13169
  }), MigrateDeviceResultSchema, {
13047
13170
  kind: "mutation",
13048
- auth: "admin"
13171
+ auth: "admin",
13172
+ timeoutMs: 12 * 6e4
13049
13173
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13050
13174
  deviceId: number(),
13051
13175
  name: string()
@@ -32418,6 +32542,147 @@ var BaseDevice = class {
32418
32542
  }
32419
32543
  };
32420
32544
  /**
32545
+ * Delays before retry rounds 1..N — the round count IS the bound.
32546
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32547
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32548
+ * per attempt) covers a device-manager lock held for minutes — the
32549
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32550
+ */
32551
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32552
+ 1e4,
32553
+ 3e4,
32554
+ 9e4
32555
+ ];
32556
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32557
+ function sleep$1(ms, signal) {
32558
+ return new Promise((resolve) => {
32559
+ if (signal.aborted) {
32560
+ resolve();
32561
+ return;
32562
+ }
32563
+ const onAbort = () => {
32564
+ clearTimeout(timer);
32565
+ resolve();
32566
+ };
32567
+ const timer = setTimeout(() => {
32568
+ signal.removeEventListener("abort", onAbort);
32569
+ resolve();
32570
+ }, ms);
32571
+ timer.unref?.();
32572
+ signal.addEventListener("abort", onAbort, { once: true });
32573
+ });
32574
+ }
32575
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32576
+ * not reject (callers wrap their own try/catch). */
32577
+ async function runWithConcurrency(items, width, fn) {
32578
+ const queue = [...items];
32579
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32580
+ const lane = async () => {
32581
+ for (;;) {
32582
+ const item = queue.shift();
32583
+ if (item === void 0) return;
32584
+ await fn(item);
32585
+ }
32586
+ };
32587
+ await Promise.all(Array.from({ length: laneCount }, lane));
32588
+ }
32589
+ var DeviceRestoreRetryScheduler = class {
32590
+ #logger;
32591
+ #attempt;
32592
+ #onPermanentFailure;
32593
+ #delaysMs;
32594
+ #concurrency;
32595
+ #now;
32596
+ #abort = new AbortController();
32597
+ constructor(options) {
32598
+ this.#logger = options.logger;
32599
+ this.#attempt = options.attempt;
32600
+ this.#onPermanentFailure = options.onPermanentFailure;
32601
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32602
+ this.#concurrency = options.concurrency ?? 4;
32603
+ this.#now = options.now ?? Date.now;
32604
+ }
32605
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32606
+ * permanently failed — the next boot restores them from disk. */
32607
+ cancel() {
32608
+ this.#abort.abort();
32609
+ }
32610
+ /**
32611
+ * Run the bounded retry rounds. Resolves when every entry has either
32612
+ * restored, been marked permanently failed, or the scheduler was
32613
+ * cancelled. Never rejects.
32614
+ */
32615
+ async run(initialFailures) {
32616
+ let pending = initialFailures.map((failure) => ({
32617
+ saved: failure.saved,
32618
+ lastError: failure.error,
32619
+ attempts: 1
32620
+ }));
32621
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32622
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32623
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32624
+ if (this.#abort.signal.aborted) break;
32625
+ pending = await this.#runRound(pending, round);
32626
+ }
32627
+ if (this.#abort.signal.aborted) return [];
32628
+ const terminal = pending.map((entry) => ({
32629
+ deviceId: entry.saved.id,
32630
+ stableId: entry.saved.stableId,
32631
+ type: String(entry.saved.type),
32632
+ attempts: entry.attempts,
32633
+ lastError: entry.lastError,
32634
+ failedAt: this.#now()
32635
+ }));
32636
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32637
+ return terminal;
32638
+ }
32639
+ /** One retry round: parents first (phase 0), then hub-adopted
32640
+ * children (phase 1) — a child's attempt depends on its parent
32641
+ * having landed, exactly like the initial two-pass restore. */
32642
+ async #runRound(pending, round) {
32643
+ const next = [];
32644
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32645
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32646
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32647
+ if (this.#abort.signal.aborted) {
32648
+ next.push(entry);
32649
+ return;
32650
+ }
32651
+ const attemptNo = entry.attempts + 1;
32652
+ try {
32653
+ await this.#attempt(entry.saved);
32654
+ this.#logger.info("Device restored on retry", {
32655
+ tags: {
32656
+ deviceId: entry.saved.id,
32657
+ stableId: entry.saved.stableId
32658
+ },
32659
+ meta: { attempt: attemptNo }
32660
+ });
32661
+ } catch (err) {
32662
+ const lastError = err instanceof Error ? err.message : String(err);
32663
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32664
+ this.#logger.warn("Device restore retry failed", {
32665
+ tags: {
32666
+ deviceId: entry.saved.id,
32667
+ stableId: entry.saved.stableId
32668
+ },
32669
+ meta: {
32670
+ attempt: attemptNo,
32671
+ remainingRetries,
32672
+ error: lastError
32673
+ }
32674
+ });
32675
+ next.push({
32676
+ saved: entry.saved,
32677
+ lastError,
32678
+ attempts: attemptNo
32679
+ });
32680
+ }
32681
+ });
32682
+ return next;
32683
+ }
32684
+ };
32685
+ /**
32421
32686
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32422
32687
  * device-provider cap router. Shared across all providers.
32423
32688
  */
@@ -32466,6 +32731,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32466
32731
  }];
32467
32732
  }
32468
32733
  async onShutdown() {
32734
+ this.cancelRestoreRetries();
32469
32735
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32470
32736
  for (const device of devices) try {
32471
32737
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32483,9 +32749,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32483
32749
  async start() {}
32484
32750
  async stop() {}
32485
32751
  async getStatus() {
32752
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32753
+ const summary = this.restoreFailureSummary();
32754
+ if (summary === null) return {
32755
+ connected: true,
32756
+ deviceCount: all.length
32757
+ };
32486
32758
  return {
32487
32759
  connected: true,
32488
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32760
+ deviceCount: all.length,
32761
+ error: summary
32489
32762
  };
32490
32763
  }
32491
32764
  async getDevices() {
@@ -32575,8 +32848,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32575
32848
  };
32576
32849
  }
32577
32850
  async restoreDevices(savedDevices) {
32578
- await this.onRestoreDevices(savedDevices);
32579
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32851
+ const report = await this.onRestoreDevices(savedDevices);
32852
+ if (savedDevices.length === 0) return;
32853
+ if (report && report.failedCount > 0) {
32854
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32855
+ return;
32856
+ }
32857
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32858
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32859
+ }
32860
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32861
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32862
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32863
+ * never re-stampede full-width while the initial pass does (D167). */
32864
+ restoreRetryConcurrency = 4;
32865
+ _restoreRetryScheduler = null;
32866
+ _restoreRetryCompletion = null;
32867
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32868
+ /** Settles when the background retry rounds finish (or `null` when
32869
+ * nothing failed). Exposed for tests and subclass diagnostics —
32870
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32871
+ * with the devices that restored, and a late success is announced
32872
+ * through the `native-cap-change` → `updateCaps` path. */
32873
+ get restoreRetryCompletion() {
32874
+ return this._restoreRetryCompletion;
32875
+ }
32876
+ /** Devices that exhausted the retry bound this process lifetime. */
32877
+ get permanentRestoreFailures() {
32878
+ return [...this._permanentRestoreFailures.values()];
32879
+ }
32880
+ /** One-line operator-facing summary for `getStatus().error`, or
32881
+ * `null` when every device restored. */
32882
+ restoreFailureSummary() {
32883
+ if (this._permanentRestoreFailures.size === 0) return null;
32884
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32885
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32886
+ }
32887
+ cancelRestoreRetries() {
32888
+ this._restoreRetryScheduler?.cancel();
32889
+ this._restoreRetryScheduler = null;
32890
+ }
32891
+ recordPermanentRestoreFailure(failure) {
32892
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32893
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32894
+ tags: {
32895
+ deviceId: failure.deviceId,
32896
+ stableId: failure.stableId
32897
+ },
32898
+ meta: {
32899
+ type: failure.type,
32900
+ attempts: failure.attempts,
32901
+ error: failure.lastError
32902
+ }
32903
+ });
32904
+ }
32905
+ scheduleRestoreRetries(failures, attempt) {
32906
+ const scheduler = new DeviceRestoreRetryScheduler({
32907
+ logger: this.ctx.logger,
32908
+ delaysMs: this.restoreRetryDelaysMs,
32909
+ concurrency: this.restoreRetryConcurrency,
32910
+ attempt,
32911
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32912
+ });
32913
+ this._restoreRetryScheduler = scheduler;
32914
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32915
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32916
+ });
32917
+ }
32918
+ /**
32919
+ * Tear down and reconstruct ONE device from its persisted rows — the
32920
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32921
+ * and no other device this provider owns is disturbed.
32922
+ *
32923
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32924
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32925
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32926
+ * whatever number the row carries NOW. The teardown is `decommission` —
32927
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32928
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32929
+ * the boot restore's own `create()` path, including its pass 2: first-class
32930
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32931
+ * parent by the cascade and must be re-created explicitly, because only
32932
+ * accessory children come back through `getAccessoryChildren()`.
32933
+ *
32934
+ * Reloading an accessory child directly is refused (no device class) —
32935
+ * reload its parent instead.
32936
+ */
32937
+ async reloadDevice(input) {
32938
+ const { stableId } = input;
32939
+ const devices = this.ctx.kernel.devices;
32940
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32941
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32942
+ if (live) await devices.decommission(live.id);
32943
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32944
+ addonId: this.addonId,
32945
+ stableId
32946
+ });
32947
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32948
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32949
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32950
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32951
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32952
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32953
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32954
+ for (const row of rows) {
32955
+ if (row.parentDeviceId !== id) continue;
32956
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32957
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32958
+ if (!ChildClass) continue;
32959
+ try {
32960
+ await devices.create(row.stableId, ChildClass, {}, id);
32961
+ } catch (err) {
32962
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32963
+ tags: {
32964
+ deviceId: row.id,
32965
+ stableId: row.stableId
32966
+ },
32967
+ meta: {
32968
+ parentDeviceId: id,
32969
+ error: err instanceof Error ? err.message : String(err)
32970
+ }
32971
+ });
32972
+ }
32973
+ }
32974
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32975
+ tags: { deviceId: id },
32976
+ meta: {
32977
+ stableId,
32978
+ type: meta.type
32979
+ }
32980
+ });
32981
+ return { deviceId: id };
32580
32982
  }
32581
32983
  /**
32582
32984
  * Restore devices from persisted state. Two-pass:
@@ -32602,55 +33004,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32602
33004
  * accessory-spawn flow handles via the parent's
32603
33005
  * `getAccessoryChildren()`. Override only when the default doesn't
32604
33006
  * fit.
33007
+ *
33008
+ * A row that fails either pass is NOT terminal (D347): it is handed
33009
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33010
+ * Only after the bound is exhausted is the device marked permanently
33011
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33012
+ * `getStatus().error`.
32605
33013
  */
32606
33014
  async onRestoreDevices(savedDevices) {
32607
33015
  const restored = /* @__PURE__ */ new Set();
33016
+ const failures = [];
33017
+ const attemptRestore = async (saved) => {
33018
+ if (restored.has(saved.id)) return;
33019
+ const Class = this.deviceClasses[saved.type];
33020
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33021
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33022
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33023
+ restored.add(saved.id);
33024
+ };
32608
33025
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32609
33026
  const restoreOne = async (saved) => {
32610
- const Class = this.deviceClasses[saved.type];
32611
- if (!Class) {
33027
+ if (!this.deviceClasses[saved.type]) {
32612
33028
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32613
- tags: { stableId: saved.stableId },
33029
+ tags: {
33030
+ deviceId: saved.id,
33031
+ stableId: saved.stableId
33032
+ },
32614
33033
  meta: { type: saved.type }
32615
33034
  });
32616
33035
  return;
32617
33036
  }
32618
33037
  try {
32619
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32620
- restored.add(saved.id);
33038
+ await attemptRestore(saved);
32621
33039
  } catch (err) {
32622
- this.ctx.logger.warn("Failed to restore device", {
32623
- tags: { stableId: saved.stableId },
33040
+ const error = err instanceof Error ? err.message : String(err);
33041
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33042
+ tags: {
33043
+ deviceId: saved.id,
33044
+ stableId: saved.stableId
33045
+ },
32624
33046
  meta: {
32625
33047
  type: saved.type,
32626
- error: err instanceof Error ? err.message : String(err)
33048
+ attempt: 1,
33049
+ error
32627
33050
  }
32628
33051
  });
33052
+ failures.push({
33053
+ saved,
33054
+ error
33055
+ });
32629
33056
  }
32630
33057
  };
32631
33058
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33059
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32632
33060
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32633
33061
  for (const saved of childRows) {
32634
- const Class = this.deviceClasses[saved.type];
32635
- if (!Class) continue;
33062
+ if (!this.deviceClasses[saved.type]) continue;
32636
33063
  if (saved.parentDeviceId === null) continue;
32637
- if (!restored.has(saved.parentDeviceId)) continue;
32638
- try {
32639
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32640
- restored.add(saved.id);
32641
- } catch (err) {
32642
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33064
+ if (restored.has(saved.parentDeviceId)) {
33065
+ try {
33066
+ await attemptRestore(saved);
33067
+ } catch (err) {
33068
+ const error = err instanceof Error ? err.message : String(err);
33069
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33070
+ tags: {
33071
+ deviceId: saved.id,
33072
+ stableId: saved.stableId,
33073
+ parentDeviceId: saved.parentDeviceId
33074
+ },
33075
+ meta: {
33076
+ type: saved.type,
33077
+ attempt: 1,
33078
+ error
33079
+ }
33080
+ });
33081
+ failures.push({
33082
+ saved,
33083
+ error
33084
+ });
33085
+ }
33086
+ continue;
33087
+ }
33088
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33089
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32643
33090
  tags: {
33091
+ deviceId: saved.id,
32644
33092
  stableId: saved.stableId,
32645
33093
  parentDeviceId: saved.parentDeviceId
32646
33094
  },
32647
- meta: {
32648
- type: saved.type,
32649
- error: err instanceof Error ? err.message : String(err)
32650
- }
33095
+ meta: { type: saved.type }
32651
33096
  });
33097
+ failures.push({
33098
+ saved,
33099
+ error: `parent device ${saved.parentDeviceId} not restored`
33100
+ });
33101
+ continue;
32652
33102
  }
32653
33103
  }
33104
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33105
+ return {
33106
+ restoredCount: restored.size,
33107
+ failedCount: failures.length
33108
+ };
32654
33109
  }
32655
33110
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32656
33111
  toSummary(device) {
@@ -33159,6 +33614,12 @@ Object.freeze({
33159
33614
  addonId: null,
33160
33615
  access: "create"
33161
33616
  },
33617
+ "backup.cancel": {
33618
+ capName: "backup",
33619
+ capScope: "system",
33620
+ addonId: null,
33621
+ access: "create"
33622
+ },
33162
33623
  "backup.delete": {
33163
33624
  capName: "backup",
33164
33625
  capScope: "system",
@@ -33201,6 +33662,12 @@ Object.freeze({
33201
33662
  addonId: null,
33202
33663
  access: "view"
33203
33664
  },
33665
+ "backup.listRuns": {
33666
+ capName: "backup",
33667
+ capScope: "system",
33668
+ addonId: null,
33669
+ access: "view"
33670
+ },
33204
33671
  "backup.listSchedules": {
33205
33672
  capName: "backup",
33206
33673
  capScope: "system",
@@ -34401,6 +34868,12 @@ Object.freeze({
34401
34868
  addonId: null,
34402
34869
  access: "view"
34403
34870
  },
34871
+ "deviceProvider.reloadDevice": {
34872
+ capName: "device-provider",
34873
+ capScope: "system",
34874
+ addonId: null,
34875
+ access: "create"
34876
+ },
34404
34877
  "deviceProvider.start": {
34405
34878
  capName: "device-provider",
34406
34879
  capScope: "system",
@@ -37767,6 +38240,12 @@ Object.freeze({
37767
38240
  addonId: null,
37768
38241
  access: "create"
37769
38242
  },
38243
+ "streamBroker.forgetDeviceHardware": {
38244
+ capName: "stream-broker",
38245
+ capScope: "system",
38246
+ addonId: null,
38247
+ access: "delete"
38248
+ },
37770
38249
  "streamBroker.getAllRtspEntries": {
37771
38250
  capName: "stream-broker",
37772
38251
  capScope: "system",
@@ -40225,6 +40704,11 @@ Object.freeze({
40225
40704
  form: "single",
40226
40705
  optional: false
40227
40706
  }],
40707
+ "streamBroker.forgetDeviceHardware": [{
40708
+ name: "deviceId",
40709
+ form: "single",
40710
+ optional: false
40711
+ }],
40228
40712
  "streamBroker.getDeviceAudioMute": [{
40229
40713
  name: "deviceId",
40230
40714
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10526,6 +10526,89 @@ var LocationStatSchema = object({
10526
10526
  fileCount: number(),
10527
10527
  present: boolean()
10528
10528
  });
10529
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10530
+ var BackupRunStateSchema = _enum([
10531
+ "queued",
10532
+ "running",
10533
+ "succeeded",
10534
+ "failed",
10535
+ "cancelled"
10536
+ ]);
10537
+ /**
10538
+ * Where a running backup currently is. `queued` before it starts,
10539
+ * `building` while the tar.gz is being staged, `uploading` during the
10540
+ * per-destination fan-out, `done` once terminal.
10541
+ */
10542
+ var BackupRunPhaseSchema = _enum([
10543
+ "queued",
10544
+ "building",
10545
+ "uploading",
10546
+ "done"
10547
+ ]);
10548
+ /**
10549
+ * Observable state of one backup run — readable WHILE it runs via
10550
+ * `backup.listRuns`. This is what makes the execution queue and
10551
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10552
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10553
+ * diagnosable with `du` because nothing reported that runs existed or
10554
+ * how large the staged archive had grown.
10555
+ */
10556
+ var BackupRunSchema = object({
10557
+ /** Stable run id — the handle `backup.cancel` takes. */
10558
+ id: string(),
10559
+ state: BackupRunStateSchema,
10560
+ phase: BackupRunPhaseSchema,
10561
+ /**
10562
+ * Resolved destination location ids. Empty while queued (targets are
10563
+ * resolved when the run starts, against the then-current policies).
10564
+ */
10565
+ destinationIds: array(string()).readonly(),
10566
+ label: string().optional(),
10567
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10568
+ requestedAt: number(),
10569
+ /** ms-epoch when the run left the queue and started building. */
10570
+ startedAt: number().optional(),
10571
+ /** ms-epoch when the run reached a terminal state. */
10572
+ finishedAt: number().optional(),
10573
+ /** Compressed bytes of the staging archive written so far. */
10574
+ stagedBytes: number(),
10575
+ /** Final staged archive size, once the build phase completes. */
10576
+ archiveSizeBytes: number().optional(),
10577
+ /** Bytes pushed to the destination currently uploading. */
10578
+ uploadedBytes: number(),
10579
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10580
+ completedDestinationIds: array(string()).readonly(),
10581
+ /** Destinations that failed during the fan-out. */
10582
+ failedDestinationIds: array(string()).readonly(),
10583
+ /** Failure message when `state === 'failed'`. */
10584
+ error: string().optional(),
10585
+ /**
10586
+ * 1-based place in the execution queue — 1 = runs next. Present only
10587
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10588
+ * queue's OWN pending order, never derived from timestamps, so the
10589
+ * UI cannot show an order the executor will not honour.
10590
+ */
10591
+ queuePosition: number().int().min(1).optional()
10592
+ });
10593
+ /**
10594
+ * Result of `backup.trigger`. The call still resolves when the run
10595
+ * terminates (compat with schedule-driven runs and the admin UI), but
10596
+ * it now names the run and says whether it had to WAIT: a trigger that
10597
+ * arrives while another run is in flight is enqueued (or joined onto
10598
+ * an identical already-queued run), never started concurrently.
10599
+ */
10600
+ var BackupTriggerResultSchema = object({
10601
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10602
+ runId: string(),
10603
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10604
+ queued: boolean(),
10605
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10606
+ joined: boolean(),
10607
+ /** True when the run was cancelled before completing every destination. */
10608
+ cancelled: boolean(),
10609
+ /** One entry per destination the archive landed at (partial on cancel). */
10610
+ entries: array(BackupEntrySchema).readonly()
10611
+ });
10529
10612
  /**
10530
10613
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10531
10614
  * SET of destination locations. Supersedes the per-location cron on
@@ -10573,7 +10656,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10573
10656
  * retention (manual runs).
10574
10657
  */
10575
10658
  retentionCount: number().int().min(1).max(1e3).optional()
10576
- }).optional(), array(BackupEntrySchema).readonly(), {
10659
+ }).optional(), BackupTriggerResultSchema, {
10660
+ kind: "mutation",
10661
+ auth: "admin"
10662
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10577
10663
  kind: "mutation",
10578
10664
  auth: "admin"
10579
10665
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11290,6 +11376,14 @@ method(object({
11290
11376
  }), object({ success: literal(true) }), {
11291
11377
  kind: "mutation",
11292
11378
  auth: "admin"
11379
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11380
+ derivedStreamsDeleted: array(string()).readonly(),
11381
+ assignmentsPurged: boolean(),
11382
+ probeSnapshotsDropped: number().int().nonnegative(),
11383
+ rtspTokenRowsDeleted: number().int().nonnegative()
11384
+ }), {
11385
+ kind: "mutation",
11386
+ auth: "admin"
11293
11387
  }), method(object({
11294
11388
  deviceId: number(),
11295
11389
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12717,6 +12811,35 @@ var deviceProviderCapability = {
12717
12811
  name: string(),
12718
12812
  type: string()
12719
12813
  }))),
12814
+ /**
12815
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12816
+ * touching no other device this provider owns.
12817
+ *
12818
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12819
+ * migrated numbers: after `swapIds` the runner's live instance still
12820
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12821
+ * registrations and its log tags), and a live object cannot be renumbered.
12822
+ * Before this method the only flush was restarting the whole owning addon
12823
+ * — which took every camera the provider owns down with it (28 devices
12824
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12825
+ * same day ~27 devices' native caps did not come back on their own).
12826
+ *
12827
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12828
+ * that changes. The reply carries the id the device answers on NOW.
12829
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12830
+ * instance (if any), then re-create from the persisted row: the same
12831
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12832
+ * An RPC, never an event: a dropped event would leave the runner writing
12833
+ * against the wrong camera (D8).
12834
+ *
12835
+ * Construction can dial hardware, and the migrated source is
12836
+ * characteristically dead — the timeout covers a full activate window
12837
+ * rather than the 60 s default.
12838
+ */
12839
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12840
+ kind: "mutation",
12841
+ timeoutMs: 3 * 6e4
12842
+ }),
12720
12843
  supportsDiscovery: method(object({}), boolean()),
12721
12844
  /**
12722
12845
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13044,7 +13167,8 @@ method(object({
13044
13167
  targetId: number()
13045
13168
  }), MigrateDeviceResultSchema, {
13046
13169
  kind: "mutation",
13047
- auth: "admin"
13170
+ auth: "admin",
13171
+ timeoutMs: 12 * 6e4
13048
13172
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13049
13173
  deviceId: number(),
13050
13174
  name: string()
@@ -32417,6 +32541,147 @@ var BaseDevice = class {
32417
32541
  }
32418
32542
  };
32419
32543
  /**
32544
+ * Delays before retry rounds 1..N — the round count IS the bound.
32545
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32546
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32547
+ * per attempt) covers a device-manager lock held for minutes — the
32548
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32549
+ */
32550
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32551
+ 1e4,
32552
+ 3e4,
32553
+ 9e4
32554
+ ];
32555
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32556
+ function sleep$1(ms, signal) {
32557
+ return new Promise((resolve) => {
32558
+ if (signal.aborted) {
32559
+ resolve();
32560
+ return;
32561
+ }
32562
+ const onAbort = () => {
32563
+ clearTimeout(timer);
32564
+ resolve();
32565
+ };
32566
+ const timer = setTimeout(() => {
32567
+ signal.removeEventListener("abort", onAbort);
32568
+ resolve();
32569
+ }, ms);
32570
+ timer.unref?.();
32571
+ signal.addEventListener("abort", onAbort, { once: true });
32572
+ });
32573
+ }
32574
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32575
+ * not reject (callers wrap their own try/catch). */
32576
+ async function runWithConcurrency(items, width, fn) {
32577
+ const queue = [...items];
32578
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32579
+ const lane = async () => {
32580
+ for (;;) {
32581
+ const item = queue.shift();
32582
+ if (item === void 0) return;
32583
+ await fn(item);
32584
+ }
32585
+ };
32586
+ await Promise.all(Array.from({ length: laneCount }, lane));
32587
+ }
32588
+ var DeviceRestoreRetryScheduler = class {
32589
+ #logger;
32590
+ #attempt;
32591
+ #onPermanentFailure;
32592
+ #delaysMs;
32593
+ #concurrency;
32594
+ #now;
32595
+ #abort = new AbortController();
32596
+ constructor(options) {
32597
+ this.#logger = options.logger;
32598
+ this.#attempt = options.attempt;
32599
+ this.#onPermanentFailure = options.onPermanentFailure;
32600
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32601
+ this.#concurrency = options.concurrency ?? 4;
32602
+ this.#now = options.now ?? Date.now;
32603
+ }
32604
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32605
+ * permanently failed — the next boot restores them from disk. */
32606
+ cancel() {
32607
+ this.#abort.abort();
32608
+ }
32609
+ /**
32610
+ * Run the bounded retry rounds. Resolves when every entry has either
32611
+ * restored, been marked permanently failed, or the scheduler was
32612
+ * cancelled. Never rejects.
32613
+ */
32614
+ async run(initialFailures) {
32615
+ let pending = initialFailures.map((failure) => ({
32616
+ saved: failure.saved,
32617
+ lastError: failure.error,
32618
+ attempts: 1
32619
+ }));
32620
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32621
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32622
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32623
+ if (this.#abort.signal.aborted) break;
32624
+ pending = await this.#runRound(pending, round);
32625
+ }
32626
+ if (this.#abort.signal.aborted) return [];
32627
+ const terminal = pending.map((entry) => ({
32628
+ deviceId: entry.saved.id,
32629
+ stableId: entry.saved.stableId,
32630
+ type: String(entry.saved.type),
32631
+ attempts: entry.attempts,
32632
+ lastError: entry.lastError,
32633
+ failedAt: this.#now()
32634
+ }));
32635
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32636
+ return terminal;
32637
+ }
32638
+ /** One retry round: parents first (phase 0), then hub-adopted
32639
+ * children (phase 1) — a child's attempt depends on its parent
32640
+ * having landed, exactly like the initial two-pass restore. */
32641
+ async #runRound(pending, round) {
32642
+ const next = [];
32643
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32644
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32645
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32646
+ if (this.#abort.signal.aborted) {
32647
+ next.push(entry);
32648
+ return;
32649
+ }
32650
+ const attemptNo = entry.attempts + 1;
32651
+ try {
32652
+ await this.#attempt(entry.saved);
32653
+ this.#logger.info("Device restored on retry", {
32654
+ tags: {
32655
+ deviceId: entry.saved.id,
32656
+ stableId: entry.saved.stableId
32657
+ },
32658
+ meta: { attempt: attemptNo }
32659
+ });
32660
+ } catch (err) {
32661
+ const lastError = err instanceof Error ? err.message : String(err);
32662
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32663
+ this.#logger.warn("Device restore retry failed", {
32664
+ tags: {
32665
+ deviceId: entry.saved.id,
32666
+ stableId: entry.saved.stableId
32667
+ },
32668
+ meta: {
32669
+ attempt: attemptNo,
32670
+ remainingRetries,
32671
+ error: lastError
32672
+ }
32673
+ });
32674
+ next.push({
32675
+ saved: entry.saved,
32676
+ lastError,
32677
+ attempts: attemptNo
32678
+ });
32679
+ }
32680
+ });
32681
+ return next;
32682
+ }
32683
+ };
32684
+ /**
32420
32685
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32421
32686
  * device-provider cap router. Shared across all providers.
32422
32687
  */
@@ -32465,6 +32730,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32465
32730
  }];
32466
32731
  }
32467
32732
  async onShutdown() {
32733
+ this.cancelRestoreRetries();
32468
32734
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32469
32735
  for (const device of devices) try {
32470
32736
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32482,9 +32748,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32482
32748
  async start() {}
32483
32749
  async stop() {}
32484
32750
  async getStatus() {
32751
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32752
+ const summary = this.restoreFailureSummary();
32753
+ if (summary === null) return {
32754
+ connected: true,
32755
+ deviceCount: all.length
32756
+ };
32485
32757
  return {
32486
32758
  connected: true,
32487
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32759
+ deviceCount: all.length,
32760
+ error: summary
32488
32761
  };
32489
32762
  }
32490
32763
  async getDevices() {
@@ -32574,8 +32847,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32574
32847
  };
32575
32848
  }
32576
32849
  async restoreDevices(savedDevices) {
32577
- await this.onRestoreDevices(savedDevices);
32578
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32850
+ const report = await this.onRestoreDevices(savedDevices);
32851
+ if (savedDevices.length === 0) return;
32852
+ if (report && report.failedCount > 0) {
32853
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32854
+ return;
32855
+ }
32856
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32857
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32858
+ }
32859
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32860
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32861
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32862
+ * never re-stampede full-width while the initial pass does (D167). */
32863
+ restoreRetryConcurrency = 4;
32864
+ _restoreRetryScheduler = null;
32865
+ _restoreRetryCompletion = null;
32866
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32867
+ /** Settles when the background retry rounds finish (or `null` when
32868
+ * nothing failed). Exposed for tests and subclass diagnostics —
32869
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32870
+ * with the devices that restored, and a late success is announced
32871
+ * through the `native-cap-change` → `updateCaps` path. */
32872
+ get restoreRetryCompletion() {
32873
+ return this._restoreRetryCompletion;
32874
+ }
32875
+ /** Devices that exhausted the retry bound this process lifetime. */
32876
+ get permanentRestoreFailures() {
32877
+ return [...this._permanentRestoreFailures.values()];
32878
+ }
32879
+ /** One-line operator-facing summary for `getStatus().error`, or
32880
+ * `null` when every device restored. */
32881
+ restoreFailureSummary() {
32882
+ if (this._permanentRestoreFailures.size === 0) return null;
32883
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32884
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32885
+ }
32886
+ cancelRestoreRetries() {
32887
+ this._restoreRetryScheduler?.cancel();
32888
+ this._restoreRetryScheduler = null;
32889
+ }
32890
+ recordPermanentRestoreFailure(failure) {
32891
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32892
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32893
+ tags: {
32894
+ deviceId: failure.deviceId,
32895
+ stableId: failure.stableId
32896
+ },
32897
+ meta: {
32898
+ type: failure.type,
32899
+ attempts: failure.attempts,
32900
+ error: failure.lastError
32901
+ }
32902
+ });
32903
+ }
32904
+ scheduleRestoreRetries(failures, attempt) {
32905
+ const scheduler = new DeviceRestoreRetryScheduler({
32906
+ logger: this.ctx.logger,
32907
+ delaysMs: this.restoreRetryDelaysMs,
32908
+ concurrency: this.restoreRetryConcurrency,
32909
+ attempt,
32910
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32911
+ });
32912
+ this._restoreRetryScheduler = scheduler;
32913
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32914
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32915
+ });
32916
+ }
32917
+ /**
32918
+ * Tear down and reconstruct ONE device from its persisted rows — the
32919
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32920
+ * and no other device this provider owns is disturbed.
32921
+ *
32922
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32923
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32924
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32925
+ * whatever number the row carries NOW. The teardown is `decommission` —
32926
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32927
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32928
+ * the boot restore's own `create()` path, including its pass 2: first-class
32929
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32930
+ * parent by the cascade and must be re-created explicitly, because only
32931
+ * accessory children come back through `getAccessoryChildren()`.
32932
+ *
32933
+ * Reloading an accessory child directly is refused (no device class) —
32934
+ * reload its parent instead.
32935
+ */
32936
+ async reloadDevice(input) {
32937
+ const { stableId } = input;
32938
+ const devices = this.ctx.kernel.devices;
32939
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32940
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32941
+ if (live) await devices.decommission(live.id);
32942
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32943
+ addonId: this.addonId,
32944
+ stableId
32945
+ });
32946
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32947
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32948
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32949
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32950
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32951
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32952
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32953
+ for (const row of rows) {
32954
+ if (row.parentDeviceId !== id) continue;
32955
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32956
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32957
+ if (!ChildClass) continue;
32958
+ try {
32959
+ await devices.create(row.stableId, ChildClass, {}, id);
32960
+ } catch (err) {
32961
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32962
+ tags: {
32963
+ deviceId: row.id,
32964
+ stableId: row.stableId
32965
+ },
32966
+ meta: {
32967
+ parentDeviceId: id,
32968
+ error: err instanceof Error ? err.message : String(err)
32969
+ }
32970
+ });
32971
+ }
32972
+ }
32973
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32974
+ tags: { deviceId: id },
32975
+ meta: {
32976
+ stableId,
32977
+ type: meta.type
32978
+ }
32979
+ });
32980
+ return { deviceId: id };
32579
32981
  }
32580
32982
  /**
32581
32983
  * Restore devices from persisted state. Two-pass:
@@ -32601,55 +33003,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32601
33003
  * accessory-spawn flow handles via the parent's
32602
33004
  * `getAccessoryChildren()`. Override only when the default doesn't
32603
33005
  * fit.
33006
+ *
33007
+ * A row that fails either pass is NOT terminal (D347): it is handed
33008
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33009
+ * Only after the bound is exhausted is the device marked permanently
33010
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33011
+ * `getStatus().error`.
32604
33012
  */
32605
33013
  async onRestoreDevices(savedDevices) {
32606
33014
  const restored = /* @__PURE__ */ new Set();
33015
+ const failures = [];
33016
+ const attemptRestore = async (saved) => {
33017
+ if (restored.has(saved.id)) return;
33018
+ const Class = this.deviceClasses[saved.type];
33019
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33020
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33021
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33022
+ restored.add(saved.id);
33023
+ };
32607
33024
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32608
33025
  const restoreOne = async (saved) => {
32609
- const Class = this.deviceClasses[saved.type];
32610
- if (!Class) {
33026
+ if (!this.deviceClasses[saved.type]) {
32611
33027
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32612
- tags: { stableId: saved.stableId },
33028
+ tags: {
33029
+ deviceId: saved.id,
33030
+ stableId: saved.stableId
33031
+ },
32613
33032
  meta: { type: saved.type }
32614
33033
  });
32615
33034
  return;
32616
33035
  }
32617
33036
  try {
32618
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32619
- restored.add(saved.id);
33037
+ await attemptRestore(saved);
32620
33038
  } catch (err) {
32621
- this.ctx.logger.warn("Failed to restore device", {
32622
- tags: { stableId: saved.stableId },
33039
+ const error = err instanceof Error ? err.message : String(err);
33040
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33041
+ tags: {
33042
+ deviceId: saved.id,
33043
+ stableId: saved.stableId
33044
+ },
32623
33045
  meta: {
32624
33046
  type: saved.type,
32625
- error: err instanceof Error ? err.message : String(err)
33047
+ attempt: 1,
33048
+ error
32626
33049
  }
32627
33050
  });
33051
+ failures.push({
33052
+ saved,
33053
+ error
33054
+ });
32628
33055
  }
32629
33056
  };
32630
33057
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33058
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32631
33059
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32632
33060
  for (const saved of childRows) {
32633
- const Class = this.deviceClasses[saved.type];
32634
- if (!Class) continue;
33061
+ if (!this.deviceClasses[saved.type]) continue;
32635
33062
  if (saved.parentDeviceId === null) continue;
32636
- if (!restored.has(saved.parentDeviceId)) continue;
32637
- try {
32638
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32639
- restored.add(saved.id);
32640
- } catch (err) {
32641
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33063
+ if (restored.has(saved.parentDeviceId)) {
33064
+ try {
33065
+ await attemptRestore(saved);
33066
+ } catch (err) {
33067
+ const error = err instanceof Error ? err.message : String(err);
33068
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33069
+ tags: {
33070
+ deviceId: saved.id,
33071
+ stableId: saved.stableId,
33072
+ parentDeviceId: saved.parentDeviceId
33073
+ },
33074
+ meta: {
33075
+ type: saved.type,
33076
+ attempt: 1,
33077
+ error
33078
+ }
33079
+ });
33080
+ failures.push({
33081
+ saved,
33082
+ error
33083
+ });
33084
+ }
33085
+ continue;
33086
+ }
33087
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33088
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32642
33089
  tags: {
33090
+ deviceId: saved.id,
32643
33091
  stableId: saved.stableId,
32644
33092
  parentDeviceId: saved.parentDeviceId
32645
33093
  },
32646
- meta: {
32647
- type: saved.type,
32648
- error: err instanceof Error ? err.message : String(err)
32649
- }
33094
+ meta: { type: saved.type }
32650
33095
  });
33096
+ failures.push({
33097
+ saved,
33098
+ error: `parent device ${saved.parentDeviceId} not restored`
33099
+ });
33100
+ continue;
32651
33101
  }
32652
33102
  }
33103
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33104
+ return {
33105
+ restoredCount: restored.size,
33106
+ failedCount: failures.length
33107
+ };
32653
33108
  }
32654
33109
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32655
33110
  toSummary(device) {
@@ -33158,6 +33613,12 @@ Object.freeze({
33158
33613
  addonId: null,
33159
33614
  access: "create"
33160
33615
  },
33616
+ "backup.cancel": {
33617
+ capName: "backup",
33618
+ capScope: "system",
33619
+ addonId: null,
33620
+ access: "create"
33621
+ },
33161
33622
  "backup.delete": {
33162
33623
  capName: "backup",
33163
33624
  capScope: "system",
@@ -33200,6 +33661,12 @@ Object.freeze({
33200
33661
  addonId: null,
33201
33662
  access: "view"
33202
33663
  },
33664
+ "backup.listRuns": {
33665
+ capName: "backup",
33666
+ capScope: "system",
33667
+ addonId: null,
33668
+ access: "view"
33669
+ },
33203
33670
  "backup.listSchedules": {
33204
33671
  capName: "backup",
33205
33672
  capScope: "system",
@@ -34400,6 +34867,12 @@ Object.freeze({
34400
34867
  addonId: null,
34401
34868
  access: "view"
34402
34869
  },
34870
+ "deviceProvider.reloadDevice": {
34871
+ capName: "device-provider",
34872
+ capScope: "system",
34873
+ addonId: null,
34874
+ access: "create"
34875
+ },
34403
34876
  "deviceProvider.start": {
34404
34877
  capName: "device-provider",
34405
34878
  capScope: "system",
@@ -37766,6 +38239,12 @@ Object.freeze({
37766
38239
  addonId: null,
37767
38240
  access: "create"
37768
38241
  },
38242
+ "streamBroker.forgetDeviceHardware": {
38243
+ capName: "stream-broker",
38244
+ capScope: "system",
38245
+ addonId: null,
38246
+ access: "delete"
38247
+ },
37769
38248
  "streamBroker.getAllRtspEntries": {
37770
38249
  capName: "stream-broker",
37771
38250
  capScope: "system",
@@ -40224,6 +40703,11 @@ Object.freeze({
40224
40703
  form: "single",
40225
40704
  optional: false
40226
40705
  }],
40706
+ "streamBroker.forgetDeviceHardware": [{
40707
+ name: "deviceId",
40708
+ form: "single",
40709
+ optional: false
40710
+ }],
40227
40711
  "streamBroker.getDeviceAudioMute": [{
40228
40712
  name: "deviceId",
40229
40713
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-unifi",
3
- "version": "0.2.58",
3
+ "version": "0.2.60",
4
4
  "description": "UniFi Network controller device-provider addon for CamStack — local-controller infra switches/APs (as containers) + network-client presence. NO cameras/Protect.",
5
5
  "keywords": [
6
6
  "camstack",