@camstack/addon-provider-homematic 1.2.59 → 1.2.61

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10536,6 +10536,89 @@ var LocationStatSchema = object({
10536
10536
  fileCount: number(),
10537
10537
  present: boolean()
10538
10538
  });
10539
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10540
+ var BackupRunStateSchema = _enum([
10541
+ "queued",
10542
+ "running",
10543
+ "succeeded",
10544
+ "failed",
10545
+ "cancelled"
10546
+ ]);
10547
+ /**
10548
+ * Where a running backup currently is. `queued` before it starts,
10549
+ * `building` while the tar.gz is being staged, `uploading` during the
10550
+ * per-destination fan-out, `done` once terminal.
10551
+ */
10552
+ var BackupRunPhaseSchema = _enum([
10553
+ "queued",
10554
+ "building",
10555
+ "uploading",
10556
+ "done"
10557
+ ]);
10558
+ /**
10559
+ * Observable state of one backup run — readable WHILE it runs via
10560
+ * `backup.listRuns`. This is what makes the execution queue and
10561
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10562
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10563
+ * diagnosable with `du` because nothing reported that runs existed or
10564
+ * how large the staged archive had grown.
10565
+ */
10566
+ var BackupRunSchema = object({
10567
+ /** Stable run id — the handle `backup.cancel` takes. */
10568
+ id: string(),
10569
+ state: BackupRunStateSchema,
10570
+ phase: BackupRunPhaseSchema,
10571
+ /**
10572
+ * Resolved destination location ids. Empty while queued (targets are
10573
+ * resolved when the run starts, against the then-current policies).
10574
+ */
10575
+ destinationIds: array(string()).readonly(),
10576
+ label: string().optional(),
10577
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10578
+ requestedAt: number(),
10579
+ /** ms-epoch when the run left the queue and started building. */
10580
+ startedAt: number().optional(),
10581
+ /** ms-epoch when the run reached a terminal state. */
10582
+ finishedAt: number().optional(),
10583
+ /** Compressed bytes of the staging archive written so far. */
10584
+ stagedBytes: number(),
10585
+ /** Final staged archive size, once the build phase completes. */
10586
+ archiveSizeBytes: number().optional(),
10587
+ /** Bytes pushed to the destination currently uploading. */
10588
+ uploadedBytes: number(),
10589
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10590
+ completedDestinationIds: array(string()).readonly(),
10591
+ /** Destinations that failed during the fan-out. */
10592
+ failedDestinationIds: array(string()).readonly(),
10593
+ /** Failure message when `state === 'failed'`. */
10594
+ error: string().optional(),
10595
+ /**
10596
+ * 1-based place in the execution queue — 1 = runs next. Present only
10597
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10598
+ * queue's OWN pending order, never derived from timestamps, so the
10599
+ * UI cannot show an order the executor will not honour.
10600
+ */
10601
+ queuePosition: number().int().min(1).optional()
10602
+ });
10603
+ /**
10604
+ * Result of `backup.trigger`. The call still resolves when the run
10605
+ * terminates (compat with schedule-driven runs and the admin UI), but
10606
+ * it now names the run and says whether it had to WAIT: a trigger that
10607
+ * arrives while another run is in flight is enqueued (or joined onto
10608
+ * an identical already-queued run), never started concurrently.
10609
+ */
10610
+ var BackupTriggerResultSchema = object({
10611
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10612
+ runId: string(),
10613
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10614
+ queued: boolean(),
10615
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10616
+ joined: boolean(),
10617
+ /** True when the run was cancelled before completing every destination. */
10618
+ cancelled: boolean(),
10619
+ /** One entry per destination the archive landed at (partial on cancel). */
10620
+ entries: array(BackupEntrySchema).readonly()
10621
+ });
10539
10622
  /**
10540
10623
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10541
10624
  * SET of destination locations. Supersedes the per-location cron on
@@ -10583,7 +10666,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10583
10666
  * retention (manual runs).
10584
10667
  */
10585
10668
  retentionCount: number().int().min(1).max(1e3).optional()
10586
- }).optional(), array(BackupEntrySchema).readonly(), {
10669
+ }).optional(), BackupTriggerResultSchema, {
10670
+ kind: "mutation",
10671
+ auth: "admin"
10672
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10587
10673
  kind: "mutation",
10588
10674
  auth: "admin"
10589
10675
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11349,6 +11435,14 @@ method(object({
11349
11435
  }), object({ success: literal(true) }), {
11350
11436
  kind: "mutation",
11351
11437
  auth: "admin"
11438
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11439
+ derivedStreamsDeleted: array(string()).readonly(),
11440
+ assignmentsPurged: boolean(),
11441
+ probeSnapshotsDropped: number().int().nonnegative(),
11442
+ rtspTokenRowsDeleted: number().int().nonnegative()
11443
+ }), {
11444
+ kind: "mutation",
11445
+ auth: "admin"
11352
11446
  }), method(object({
11353
11447
  deviceId: number(),
11354
11448
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12793,6 +12887,35 @@ var deviceProviderCapability = {
12793
12887
  name: string(),
12794
12888
  type: string()
12795
12889
  }))),
12890
+ /**
12891
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12892
+ * touching no other device this provider owns.
12893
+ *
12894
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12895
+ * migrated numbers: after `swapIds` the runner's live instance still
12896
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12897
+ * registrations and its log tags), and a live object cannot be renumbered.
12898
+ * Before this method the only flush was restarting the whole owning addon
12899
+ * — which took every camera the provider owns down with it (28 devices
12900
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12901
+ * same day ~27 devices' native caps did not come back on their own).
12902
+ *
12903
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12904
+ * that changes. The reply carries the id the device answers on NOW.
12905
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12906
+ * instance (if any), then re-create from the persisted row: the same
12907
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12908
+ * An RPC, never an event: a dropped event would leave the runner writing
12909
+ * against the wrong camera (D8).
12910
+ *
12911
+ * Construction can dial hardware, and the migrated source is
12912
+ * characteristically dead — the timeout covers a full activate window
12913
+ * rather than the 60 s default.
12914
+ */
12915
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12916
+ kind: "mutation",
12917
+ timeoutMs: 3 * 6e4
12918
+ }),
12796
12919
  supportsDiscovery: method(object({}), boolean()),
12797
12920
  /**
12798
12921
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13120,7 +13243,8 @@ method(object({
13120
13243
  targetId: number()
13121
13244
  }), MigrateDeviceResultSchema, {
13122
13245
  kind: "mutation",
13123
- auth: "admin"
13246
+ auth: "admin",
13247
+ timeoutMs: 12 * 6e4
13124
13248
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13125
13249
  deviceId: number(),
13126
13250
  name: string()
@@ -32476,6 +32600,147 @@ var BaseDevice = class {
32476
32600
  }
32477
32601
  };
32478
32602
  /**
32603
+ * Delays before retry rounds 1..N — the round count IS the bound.
32604
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32605
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32606
+ * per attempt) covers a device-manager lock held for minutes — the
32607
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32608
+ */
32609
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32610
+ 1e4,
32611
+ 3e4,
32612
+ 9e4
32613
+ ];
32614
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32615
+ function sleep$1(ms, signal) {
32616
+ return new Promise((resolve) => {
32617
+ if (signal.aborted) {
32618
+ resolve();
32619
+ return;
32620
+ }
32621
+ const onAbort = () => {
32622
+ clearTimeout(timer);
32623
+ resolve();
32624
+ };
32625
+ const timer = setTimeout(() => {
32626
+ signal.removeEventListener("abort", onAbort);
32627
+ resolve();
32628
+ }, ms);
32629
+ timer.unref?.();
32630
+ signal.addEventListener("abort", onAbort, { once: true });
32631
+ });
32632
+ }
32633
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32634
+ * not reject (callers wrap their own try/catch). */
32635
+ async function runWithConcurrency(items, width, fn) {
32636
+ const queue = [...items];
32637
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32638
+ const lane = async () => {
32639
+ for (;;) {
32640
+ const item = queue.shift();
32641
+ if (item === void 0) return;
32642
+ await fn(item);
32643
+ }
32644
+ };
32645
+ await Promise.all(Array.from({ length: laneCount }, lane));
32646
+ }
32647
+ var DeviceRestoreRetryScheduler = class {
32648
+ #logger;
32649
+ #attempt;
32650
+ #onPermanentFailure;
32651
+ #delaysMs;
32652
+ #concurrency;
32653
+ #now;
32654
+ #abort = new AbortController();
32655
+ constructor(options) {
32656
+ this.#logger = options.logger;
32657
+ this.#attempt = options.attempt;
32658
+ this.#onPermanentFailure = options.onPermanentFailure;
32659
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32660
+ this.#concurrency = options.concurrency ?? 4;
32661
+ this.#now = options.now ?? Date.now;
32662
+ }
32663
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32664
+ * permanently failed — the next boot restores them from disk. */
32665
+ cancel() {
32666
+ this.#abort.abort();
32667
+ }
32668
+ /**
32669
+ * Run the bounded retry rounds. Resolves when every entry has either
32670
+ * restored, been marked permanently failed, or the scheduler was
32671
+ * cancelled. Never rejects.
32672
+ */
32673
+ async run(initialFailures) {
32674
+ let pending = initialFailures.map((failure) => ({
32675
+ saved: failure.saved,
32676
+ lastError: failure.error,
32677
+ attempts: 1
32678
+ }));
32679
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32680
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32681
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32682
+ if (this.#abort.signal.aborted) break;
32683
+ pending = await this.#runRound(pending, round);
32684
+ }
32685
+ if (this.#abort.signal.aborted) return [];
32686
+ const terminal = pending.map((entry) => ({
32687
+ deviceId: entry.saved.id,
32688
+ stableId: entry.saved.stableId,
32689
+ type: String(entry.saved.type),
32690
+ attempts: entry.attempts,
32691
+ lastError: entry.lastError,
32692
+ failedAt: this.#now()
32693
+ }));
32694
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32695
+ return terminal;
32696
+ }
32697
+ /** One retry round: parents first (phase 0), then hub-adopted
32698
+ * children (phase 1) — a child's attempt depends on its parent
32699
+ * having landed, exactly like the initial two-pass restore. */
32700
+ async #runRound(pending, round) {
32701
+ const next = [];
32702
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32703
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32704
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32705
+ if (this.#abort.signal.aborted) {
32706
+ next.push(entry);
32707
+ return;
32708
+ }
32709
+ const attemptNo = entry.attempts + 1;
32710
+ try {
32711
+ await this.#attempt(entry.saved);
32712
+ this.#logger.info("Device restored on retry", {
32713
+ tags: {
32714
+ deviceId: entry.saved.id,
32715
+ stableId: entry.saved.stableId
32716
+ },
32717
+ meta: { attempt: attemptNo }
32718
+ });
32719
+ } catch (err) {
32720
+ const lastError = err instanceof Error ? err.message : String(err);
32721
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32722
+ this.#logger.warn("Device restore retry failed", {
32723
+ tags: {
32724
+ deviceId: entry.saved.id,
32725
+ stableId: entry.saved.stableId
32726
+ },
32727
+ meta: {
32728
+ attempt: attemptNo,
32729
+ remainingRetries,
32730
+ error: lastError
32731
+ }
32732
+ });
32733
+ next.push({
32734
+ saved: entry.saved,
32735
+ lastError,
32736
+ attempts: attemptNo
32737
+ });
32738
+ }
32739
+ });
32740
+ return next;
32741
+ }
32742
+ };
32743
+ /**
32479
32744
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32480
32745
  * device-provider cap router. Shared across all providers.
32481
32746
  */
@@ -32524,6 +32789,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32524
32789
  }];
32525
32790
  }
32526
32791
  async onShutdown() {
32792
+ this.cancelRestoreRetries();
32527
32793
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32528
32794
  for (const device of devices) try {
32529
32795
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32541,9 +32807,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32541
32807
  async start() {}
32542
32808
  async stop() {}
32543
32809
  async getStatus() {
32810
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32811
+ const summary = this.restoreFailureSummary();
32812
+ if (summary === null) return {
32813
+ connected: true,
32814
+ deviceCount: all.length
32815
+ };
32544
32816
  return {
32545
32817
  connected: true,
32546
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32818
+ deviceCount: all.length,
32819
+ error: summary
32547
32820
  };
32548
32821
  }
32549
32822
  async getDevices() {
@@ -32633,8 +32906,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32633
32906
  };
32634
32907
  }
32635
32908
  async restoreDevices(savedDevices) {
32636
- await this.onRestoreDevices(savedDevices);
32637
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32909
+ const report = await this.onRestoreDevices(savedDevices);
32910
+ if (savedDevices.length === 0) return;
32911
+ if (report && report.failedCount > 0) {
32912
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32913
+ return;
32914
+ }
32915
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32916
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32917
+ }
32918
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32919
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32920
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32921
+ * never re-stampede full-width while the initial pass does (D167). */
32922
+ restoreRetryConcurrency = 4;
32923
+ _restoreRetryScheduler = null;
32924
+ _restoreRetryCompletion = null;
32925
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32926
+ /** Settles when the background retry rounds finish (or `null` when
32927
+ * nothing failed). Exposed for tests and subclass diagnostics —
32928
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32929
+ * with the devices that restored, and a late success is announced
32930
+ * through the `native-cap-change` → `updateCaps` path. */
32931
+ get restoreRetryCompletion() {
32932
+ return this._restoreRetryCompletion;
32933
+ }
32934
+ /** Devices that exhausted the retry bound this process lifetime. */
32935
+ get permanentRestoreFailures() {
32936
+ return [...this._permanentRestoreFailures.values()];
32937
+ }
32938
+ /** One-line operator-facing summary for `getStatus().error`, or
32939
+ * `null` when every device restored. */
32940
+ restoreFailureSummary() {
32941
+ if (this._permanentRestoreFailures.size === 0) return null;
32942
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32943
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32944
+ }
32945
+ cancelRestoreRetries() {
32946
+ this._restoreRetryScheduler?.cancel();
32947
+ this._restoreRetryScheduler = null;
32948
+ }
32949
+ recordPermanentRestoreFailure(failure) {
32950
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32951
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32952
+ tags: {
32953
+ deviceId: failure.deviceId,
32954
+ stableId: failure.stableId
32955
+ },
32956
+ meta: {
32957
+ type: failure.type,
32958
+ attempts: failure.attempts,
32959
+ error: failure.lastError
32960
+ }
32961
+ });
32962
+ }
32963
+ scheduleRestoreRetries(failures, attempt) {
32964
+ const scheduler = new DeviceRestoreRetryScheduler({
32965
+ logger: this.ctx.logger,
32966
+ delaysMs: this.restoreRetryDelaysMs,
32967
+ concurrency: this.restoreRetryConcurrency,
32968
+ attempt,
32969
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32970
+ });
32971
+ this._restoreRetryScheduler = scheduler;
32972
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32973
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32974
+ });
32975
+ }
32976
+ /**
32977
+ * Tear down and reconstruct ONE device from its persisted rows — the
32978
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32979
+ * and no other device this provider owns is disturbed.
32980
+ *
32981
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32982
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32983
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32984
+ * whatever number the row carries NOW. The teardown is `decommission` —
32985
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32986
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32987
+ * the boot restore's own `create()` path, including its pass 2: first-class
32988
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32989
+ * parent by the cascade and must be re-created explicitly, because only
32990
+ * accessory children come back through `getAccessoryChildren()`.
32991
+ *
32992
+ * Reloading an accessory child directly is refused (no device class) —
32993
+ * reload its parent instead.
32994
+ */
32995
+ async reloadDevice(input) {
32996
+ const { stableId } = input;
32997
+ const devices = this.ctx.kernel.devices;
32998
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32999
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33000
+ if (live) await devices.decommission(live.id);
33001
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33002
+ addonId: this.addonId,
33003
+ stableId
33004
+ });
33005
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33006
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33007
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33008
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33009
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33010
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33011
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33012
+ for (const row of rows) {
33013
+ if (row.parentDeviceId !== id) continue;
33014
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33015
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33016
+ if (!ChildClass) continue;
33017
+ try {
33018
+ await devices.create(row.stableId, ChildClass, {}, id);
33019
+ } catch (err) {
33020
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33021
+ tags: {
33022
+ deviceId: row.id,
33023
+ stableId: row.stableId
33024
+ },
33025
+ meta: {
33026
+ parentDeviceId: id,
33027
+ error: err instanceof Error ? err.message : String(err)
33028
+ }
33029
+ });
33030
+ }
33031
+ }
33032
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33033
+ tags: { deviceId: id },
33034
+ meta: {
33035
+ stableId,
33036
+ type: meta.type
33037
+ }
33038
+ });
33039
+ return { deviceId: id };
32638
33040
  }
32639
33041
  /**
32640
33042
  * Restore devices from persisted state. Two-pass:
@@ -32660,55 +33062,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32660
33062
  * accessory-spawn flow handles via the parent's
32661
33063
  * `getAccessoryChildren()`. Override only when the default doesn't
32662
33064
  * fit.
33065
+ *
33066
+ * A row that fails either pass is NOT terminal (D347): it is handed
33067
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33068
+ * Only after the bound is exhausted is the device marked permanently
33069
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33070
+ * `getStatus().error`.
32663
33071
  */
32664
33072
  async onRestoreDevices(savedDevices) {
32665
33073
  const restored = /* @__PURE__ */ new Set();
33074
+ const failures = [];
33075
+ const attemptRestore = async (saved) => {
33076
+ if (restored.has(saved.id)) return;
33077
+ const Class = this.deviceClasses[saved.type];
33078
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33079
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33080
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33081
+ restored.add(saved.id);
33082
+ };
32666
33083
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32667
33084
  const restoreOne = async (saved) => {
32668
- const Class = this.deviceClasses[saved.type];
32669
- if (!Class) {
33085
+ if (!this.deviceClasses[saved.type]) {
32670
33086
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32671
- tags: { stableId: saved.stableId },
33087
+ tags: {
33088
+ deviceId: saved.id,
33089
+ stableId: saved.stableId
33090
+ },
32672
33091
  meta: { type: saved.type }
32673
33092
  });
32674
33093
  return;
32675
33094
  }
32676
33095
  try {
32677
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32678
- restored.add(saved.id);
33096
+ await attemptRestore(saved);
32679
33097
  } catch (err) {
32680
- this.ctx.logger.warn("Failed to restore device", {
32681
- tags: { stableId: saved.stableId },
33098
+ const error = err instanceof Error ? err.message : String(err);
33099
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33100
+ tags: {
33101
+ deviceId: saved.id,
33102
+ stableId: saved.stableId
33103
+ },
32682
33104
  meta: {
32683
33105
  type: saved.type,
32684
- error: err instanceof Error ? err.message : String(err)
33106
+ attempt: 1,
33107
+ error
32685
33108
  }
32686
33109
  });
33110
+ failures.push({
33111
+ saved,
33112
+ error
33113
+ });
32687
33114
  }
32688
33115
  };
32689
33116
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33117
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32690
33118
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32691
33119
  for (const saved of childRows) {
32692
- const Class = this.deviceClasses[saved.type];
32693
- if (!Class) continue;
33120
+ if (!this.deviceClasses[saved.type]) continue;
32694
33121
  if (saved.parentDeviceId === null) continue;
32695
- if (!restored.has(saved.parentDeviceId)) continue;
32696
- try {
32697
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32698
- restored.add(saved.id);
32699
- } catch (err) {
32700
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33122
+ if (restored.has(saved.parentDeviceId)) {
33123
+ try {
33124
+ await attemptRestore(saved);
33125
+ } catch (err) {
33126
+ const error = err instanceof Error ? err.message : String(err);
33127
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33128
+ tags: {
33129
+ deviceId: saved.id,
33130
+ stableId: saved.stableId,
33131
+ parentDeviceId: saved.parentDeviceId
33132
+ },
33133
+ meta: {
33134
+ type: saved.type,
33135
+ attempt: 1,
33136
+ error
33137
+ }
33138
+ });
33139
+ failures.push({
33140
+ saved,
33141
+ error
33142
+ });
33143
+ }
33144
+ continue;
33145
+ }
33146
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33147
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32701
33148
  tags: {
33149
+ deviceId: saved.id,
32702
33150
  stableId: saved.stableId,
32703
33151
  parentDeviceId: saved.parentDeviceId
32704
33152
  },
32705
- meta: {
32706
- type: saved.type,
32707
- error: err instanceof Error ? err.message : String(err)
32708
- }
33153
+ meta: { type: saved.type }
33154
+ });
33155
+ failures.push({
33156
+ saved,
33157
+ error: `parent device ${saved.parentDeviceId} not restored`
32709
33158
  });
33159
+ continue;
32710
33160
  }
32711
33161
  }
33162
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33163
+ return {
33164
+ restoredCount: restored.size,
33165
+ failedCount: failures.length
33166
+ };
32712
33167
  }
32713
33168
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32714
33169
  toSummary(device) {
@@ -33217,6 +33672,12 @@ Object.freeze({
33217
33672
  addonId: null,
33218
33673
  access: "create"
33219
33674
  },
33675
+ "backup.cancel": {
33676
+ capName: "backup",
33677
+ capScope: "system",
33678
+ addonId: null,
33679
+ access: "create"
33680
+ },
33220
33681
  "backup.delete": {
33221
33682
  capName: "backup",
33222
33683
  capScope: "system",
@@ -33259,6 +33720,12 @@ Object.freeze({
33259
33720
  addonId: null,
33260
33721
  access: "view"
33261
33722
  },
33723
+ "backup.listRuns": {
33724
+ capName: "backup",
33725
+ capScope: "system",
33726
+ addonId: null,
33727
+ access: "view"
33728
+ },
33262
33729
  "backup.listSchedules": {
33263
33730
  capName: "backup",
33264
33731
  capScope: "system",
@@ -34459,6 +34926,12 @@ Object.freeze({
34459
34926
  addonId: null,
34460
34927
  access: "view"
34461
34928
  },
34929
+ "deviceProvider.reloadDevice": {
34930
+ capName: "device-provider",
34931
+ capScope: "system",
34932
+ addonId: null,
34933
+ access: "create"
34934
+ },
34462
34935
  "deviceProvider.start": {
34463
34936
  capName: "device-provider",
34464
34937
  capScope: "system",
@@ -37825,6 +38298,12 @@ Object.freeze({
37825
38298
  addonId: null,
37826
38299
  access: "create"
37827
38300
  },
38301
+ "streamBroker.forgetDeviceHardware": {
38302
+ capName: "stream-broker",
38303
+ capScope: "system",
38304
+ addonId: null,
38305
+ access: "delete"
38306
+ },
37828
38307
  "streamBroker.getAllRtspEntries": {
37829
38308
  capName: "stream-broker",
37830
38309
  capScope: "system",
@@ -40283,6 +40762,11 @@ Object.freeze({
40283
40762
  form: "single",
40284
40763
  optional: false
40285
40764
  }],
40765
+ "streamBroker.forgetDeviceHardware": [{
40766
+ name: "deviceId",
40767
+ form: "single",
40768
+ optional: false
40769
+ }],
40286
40770
  "streamBroker.getDeviceAudioMute": [{
40287
40771
  name: "deviceId",
40288
40772
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10537,6 +10537,89 @@ var LocationStatSchema = object({
10537
10537
  fileCount: number(),
10538
10538
  present: boolean()
10539
10539
  });
10540
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10541
+ var BackupRunStateSchema = _enum([
10542
+ "queued",
10543
+ "running",
10544
+ "succeeded",
10545
+ "failed",
10546
+ "cancelled"
10547
+ ]);
10548
+ /**
10549
+ * Where a running backup currently is. `queued` before it starts,
10550
+ * `building` while the tar.gz is being staged, `uploading` during the
10551
+ * per-destination fan-out, `done` once terminal.
10552
+ */
10553
+ var BackupRunPhaseSchema = _enum([
10554
+ "queued",
10555
+ "building",
10556
+ "uploading",
10557
+ "done"
10558
+ ]);
10559
+ /**
10560
+ * Observable state of one backup run — readable WHILE it runs via
10561
+ * `backup.listRuns`. This is what makes the execution queue and
10562
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10563
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10564
+ * diagnosable with `du` because nothing reported that runs existed or
10565
+ * how large the staged archive had grown.
10566
+ */
10567
+ var BackupRunSchema = object({
10568
+ /** Stable run id — the handle `backup.cancel` takes. */
10569
+ id: string(),
10570
+ state: BackupRunStateSchema,
10571
+ phase: BackupRunPhaseSchema,
10572
+ /**
10573
+ * Resolved destination location ids. Empty while queued (targets are
10574
+ * resolved when the run starts, against the then-current policies).
10575
+ */
10576
+ destinationIds: array(string()).readonly(),
10577
+ label: string().optional(),
10578
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10579
+ requestedAt: number(),
10580
+ /** ms-epoch when the run left the queue and started building. */
10581
+ startedAt: number().optional(),
10582
+ /** ms-epoch when the run reached a terminal state. */
10583
+ finishedAt: number().optional(),
10584
+ /** Compressed bytes of the staging archive written so far. */
10585
+ stagedBytes: number(),
10586
+ /** Final staged archive size, once the build phase completes. */
10587
+ archiveSizeBytes: number().optional(),
10588
+ /** Bytes pushed to the destination currently uploading. */
10589
+ uploadedBytes: number(),
10590
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10591
+ completedDestinationIds: array(string()).readonly(),
10592
+ /** Destinations that failed during the fan-out. */
10593
+ failedDestinationIds: array(string()).readonly(),
10594
+ /** Failure message when `state === 'failed'`. */
10595
+ error: string().optional(),
10596
+ /**
10597
+ * 1-based place in the execution queue — 1 = runs next. Present only
10598
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10599
+ * queue's OWN pending order, never derived from timestamps, so the
10600
+ * UI cannot show an order the executor will not honour.
10601
+ */
10602
+ queuePosition: number().int().min(1).optional()
10603
+ });
10604
+ /**
10605
+ * Result of `backup.trigger`. The call still resolves when the run
10606
+ * terminates (compat with schedule-driven runs and the admin UI), but
10607
+ * it now names the run and says whether it had to WAIT: a trigger that
10608
+ * arrives while another run is in flight is enqueued (or joined onto
10609
+ * an identical already-queued run), never started concurrently.
10610
+ */
10611
+ var BackupTriggerResultSchema = object({
10612
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10613
+ runId: string(),
10614
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10615
+ queued: boolean(),
10616
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10617
+ joined: boolean(),
10618
+ /** True when the run was cancelled before completing every destination. */
10619
+ cancelled: boolean(),
10620
+ /** One entry per destination the archive landed at (partial on cancel). */
10621
+ entries: array(BackupEntrySchema).readonly()
10622
+ });
10540
10623
  /**
10541
10624
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10542
10625
  * SET of destination locations. Supersedes the per-location cron on
@@ -10584,7 +10667,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10584
10667
  * retention (manual runs).
10585
10668
  */
10586
10669
  retentionCount: number().int().min(1).max(1e3).optional()
10587
- }).optional(), array(BackupEntrySchema).readonly(), {
10670
+ }).optional(), BackupTriggerResultSchema, {
10671
+ kind: "mutation",
10672
+ auth: "admin"
10673
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10588
10674
  kind: "mutation",
10589
10675
  auth: "admin"
10590
10676
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11350,6 +11436,14 @@ method(object({
11350
11436
  }), object({ success: literal(true) }), {
11351
11437
  kind: "mutation",
11352
11438
  auth: "admin"
11439
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11440
+ derivedStreamsDeleted: array(string()).readonly(),
11441
+ assignmentsPurged: boolean(),
11442
+ probeSnapshotsDropped: number().int().nonnegative(),
11443
+ rtspTokenRowsDeleted: number().int().nonnegative()
11444
+ }), {
11445
+ kind: "mutation",
11446
+ auth: "admin"
11353
11447
  }), method(object({
11354
11448
  deviceId: number(),
11355
11449
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12794,6 +12888,35 @@ var deviceProviderCapability = {
12794
12888
  name: string(),
12795
12889
  type: string()
12796
12890
  }))),
12891
+ /**
12892
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12893
+ * touching no other device this provider owns.
12894
+ *
12895
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12896
+ * migrated numbers: after `swapIds` the runner's live instance still
12897
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12898
+ * registrations and its log tags), and a live object cannot be renumbered.
12899
+ * Before this method the only flush was restarting the whole owning addon
12900
+ * — which took every camera the provider owns down with it (28 devices
12901
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12902
+ * same day ~27 devices' native caps did not come back on their own).
12903
+ *
12904
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12905
+ * that changes. The reply carries the id the device answers on NOW.
12906
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12907
+ * instance (if any), then re-create from the persisted row: the same
12908
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12909
+ * An RPC, never an event: a dropped event would leave the runner writing
12910
+ * against the wrong camera (D8).
12911
+ *
12912
+ * Construction can dial hardware, and the migrated source is
12913
+ * characteristically dead — the timeout covers a full activate window
12914
+ * rather than the 60 s default.
12915
+ */
12916
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12917
+ kind: "mutation",
12918
+ timeoutMs: 3 * 6e4
12919
+ }),
12797
12920
  supportsDiscovery: method(object({}), boolean()),
12798
12921
  /**
12799
12922
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13121,7 +13244,8 @@ method(object({
13121
13244
  targetId: number()
13122
13245
  }), MigrateDeviceResultSchema, {
13123
13246
  kind: "mutation",
13124
- auth: "admin"
13247
+ auth: "admin",
13248
+ timeoutMs: 12 * 6e4
13125
13249
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13126
13250
  deviceId: number(),
13127
13251
  name: string()
@@ -32477,6 +32601,147 @@ var BaseDevice = class {
32477
32601
  }
32478
32602
  };
32479
32603
  /**
32604
+ * Delays before retry rounds 1..N — the round count IS the bound.
32605
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32606
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32607
+ * per attempt) covers a device-manager lock held for minutes — the
32608
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32609
+ */
32610
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32611
+ 1e4,
32612
+ 3e4,
32613
+ 9e4
32614
+ ];
32615
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32616
+ function sleep$1(ms, signal) {
32617
+ return new Promise((resolve) => {
32618
+ if (signal.aborted) {
32619
+ resolve();
32620
+ return;
32621
+ }
32622
+ const onAbort = () => {
32623
+ clearTimeout(timer);
32624
+ resolve();
32625
+ };
32626
+ const timer = setTimeout(() => {
32627
+ signal.removeEventListener("abort", onAbort);
32628
+ resolve();
32629
+ }, ms);
32630
+ timer.unref?.();
32631
+ signal.addEventListener("abort", onAbort, { once: true });
32632
+ });
32633
+ }
32634
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32635
+ * not reject (callers wrap their own try/catch). */
32636
+ async function runWithConcurrency(items, width, fn) {
32637
+ const queue = [...items];
32638
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32639
+ const lane = async () => {
32640
+ for (;;) {
32641
+ const item = queue.shift();
32642
+ if (item === void 0) return;
32643
+ await fn(item);
32644
+ }
32645
+ };
32646
+ await Promise.all(Array.from({ length: laneCount }, lane));
32647
+ }
32648
+ var DeviceRestoreRetryScheduler = class {
32649
+ #logger;
32650
+ #attempt;
32651
+ #onPermanentFailure;
32652
+ #delaysMs;
32653
+ #concurrency;
32654
+ #now;
32655
+ #abort = new AbortController();
32656
+ constructor(options) {
32657
+ this.#logger = options.logger;
32658
+ this.#attempt = options.attempt;
32659
+ this.#onPermanentFailure = options.onPermanentFailure;
32660
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32661
+ this.#concurrency = options.concurrency ?? 4;
32662
+ this.#now = options.now ?? Date.now;
32663
+ }
32664
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32665
+ * permanently failed — the next boot restores them from disk. */
32666
+ cancel() {
32667
+ this.#abort.abort();
32668
+ }
32669
+ /**
32670
+ * Run the bounded retry rounds. Resolves when every entry has either
32671
+ * restored, been marked permanently failed, or the scheduler was
32672
+ * cancelled. Never rejects.
32673
+ */
32674
+ async run(initialFailures) {
32675
+ let pending = initialFailures.map((failure) => ({
32676
+ saved: failure.saved,
32677
+ lastError: failure.error,
32678
+ attempts: 1
32679
+ }));
32680
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32681
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32682
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32683
+ if (this.#abort.signal.aborted) break;
32684
+ pending = await this.#runRound(pending, round);
32685
+ }
32686
+ if (this.#abort.signal.aborted) return [];
32687
+ const terminal = pending.map((entry) => ({
32688
+ deviceId: entry.saved.id,
32689
+ stableId: entry.saved.stableId,
32690
+ type: String(entry.saved.type),
32691
+ attempts: entry.attempts,
32692
+ lastError: entry.lastError,
32693
+ failedAt: this.#now()
32694
+ }));
32695
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32696
+ return terminal;
32697
+ }
32698
+ /** One retry round: parents first (phase 0), then hub-adopted
32699
+ * children (phase 1) — a child's attempt depends on its parent
32700
+ * having landed, exactly like the initial two-pass restore. */
32701
+ async #runRound(pending, round) {
32702
+ const next = [];
32703
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32704
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32705
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32706
+ if (this.#abort.signal.aborted) {
32707
+ next.push(entry);
32708
+ return;
32709
+ }
32710
+ const attemptNo = entry.attempts + 1;
32711
+ try {
32712
+ await this.#attempt(entry.saved);
32713
+ this.#logger.info("Device restored on retry", {
32714
+ tags: {
32715
+ deviceId: entry.saved.id,
32716
+ stableId: entry.saved.stableId
32717
+ },
32718
+ meta: { attempt: attemptNo }
32719
+ });
32720
+ } catch (err) {
32721
+ const lastError = err instanceof Error ? err.message : String(err);
32722
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32723
+ this.#logger.warn("Device restore retry failed", {
32724
+ tags: {
32725
+ deviceId: entry.saved.id,
32726
+ stableId: entry.saved.stableId
32727
+ },
32728
+ meta: {
32729
+ attempt: attemptNo,
32730
+ remainingRetries,
32731
+ error: lastError
32732
+ }
32733
+ });
32734
+ next.push({
32735
+ saved: entry.saved,
32736
+ lastError,
32737
+ attempts: attemptNo
32738
+ });
32739
+ }
32740
+ });
32741
+ return next;
32742
+ }
32743
+ };
32744
+ /**
32480
32745
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32481
32746
  * device-provider cap router. Shared across all providers.
32482
32747
  */
@@ -32525,6 +32790,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32525
32790
  }];
32526
32791
  }
32527
32792
  async onShutdown() {
32793
+ this.cancelRestoreRetries();
32528
32794
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32529
32795
  for (const device of devices) try {
32530
32796
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32542,9 +32808,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32542
32808
  async start() {}
32543
32809
  async stop() {}
32544
32810
  async getStatus() {
32811
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32812
+ const summary = this.restoreFailureSummary();
32813
+ if (summary === null) return {
32814
+ connected: true,
32815
+ deviceCount: all.length
32816
+ };
32545
32817
  return {
32546
32818
  connected: true,
32547
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32819
+ deviceCount: all.length,
32820
+ error: summary
32548
32821
  };
32549
32822
  }
32550
32823
  async getDevices() {
@@ -32634,8 +32907,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32634
32907
  };
32635
32908
  }
32636
32909
  async restoreDevices(savedDevices) {
32637
- await this.onRestoreDevices(savedDevices);
32638
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32910
+ const report = await this.onRestoreDevices(savedDevices);
32911
+ if (savedDevices.length === 0) return;
32912
+ if (report && report.failedCount > 0) {
32913
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32914
+ return;
32915
+ }
32916
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32917
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32918
+ }
32919
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32920
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32921
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32922
+ * never re-stampede full-width while the initial pass does (D167). */
32923
+ restoreRetryConcurrency = 4;
32924
+ _restoreRetryScheduler = null;
32925
+ _restoreRetryCompletion = null;
32926
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32927
+ /** Settles when the background retry rounds finish (or `null` when
32928
+ * nothing failed). Exposed for tests and subclass diagnostics —
32929
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32930
+ * with the devices that restored, and a late success is announced
32931
+ * through the `native-cap-change` → `updateCaps` path. */
32932
+ get restoreRetryCompletion() {
32933
+ return this._restoreRetryCompletion;
32934
+ }
32935
+ /** Devices that exhausted the retry bound this process lifetime. */
32936
+ get permanentRestoreFailures() {
32937
+ return [...this._permanentRestoreFailures.values()];
32938
+ }
32939
+ /** One-line operator-facing summary for `getStatus().error`, or
32940
+ * `null` when every device restored. */
32941
+ restoreFailureSummary() {
32942
+ if (this._permanentRestoreFailures.size === 0) return null;
32943
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32944
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32945
+ }
32946
+ cancelRestoreRetries() {
32947
+ this._restoreRetryScheduler?.cancel();
32948
+ this._restoreRetryScheduler = null;
32949
+ }
32950
+ recordPermanentRestoreFailure(failure) {
32951
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32952
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32953
+ tags: {
32954
+ deviceId: failure.deviceId,
32955
+ stableId: failure.stableId
32956
+ },
32957
+ meta: {
32958
+ type: failure.type,
32959
+ attempts: failure.attempts,
32960
+ error: failure.lastError
32961
+ }
32962
+ });
32963
+ }
32964
+ scheduleRestoreRetries(failures, attempt) {
32965
+ const scheduler = new DeviceRestoreRetryScheduler({
32966
+ logger: this.ctx.logger,
32967
+ delaysMs: this.restoreRetryDelaysMs,
32968
+ concurrency: this.restoreRetryConcurrency,
32969
+ attempt,
32970
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32971
+ });
32972
+ this._restoreRetryScheduler = scheduler;
32973
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32974
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32975
+ });
32976
+ }
32977
+ /**
32978
+ * Tear down and reconstruct ONE device from its persisted rows — the
32979
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32980
+ * and no other device this provider owns is disturbed.
32981
+ *
32982
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32983
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32984
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32985
+ * whatever number the row carries NOW. The teardown is `decommission` —
32986
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32987
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32988
+ * the boot restore's own `create()` path, including its pass 2: first-class
32989
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32990
+ * parent by the cascade and must be re-created explicitly, because only
32991
+ * accessory children come back through `getAccessoryChildren()`.
32992
+ *
32993
+ * Reloading an accessory child directly is refused (no device class) —
32994
+ * reload its parent instead.
32995
+ */
32996
+ async reloadDevice(input) {
32997
+ const { stableId } = input;
32998
+ const devices = this.ctx.kernel.devices;
32999
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33000
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33001
+ if (live) await devices.decommission(live.id);
33002
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33003
+ addonId: this.addonId,
33004
+ stableId
33005
+ });
33006
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33007
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33008
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33009
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33010
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33011
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33012
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33013
+ for (const row of rows) {
33014
+ if (row.parentDeviceId !== id) continue;
33015
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33016
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33017
+ if (!ChildClass) continue;
33018
+ try {
33019
+ await devices.create(row.stableId, ChildClass, {}, id);
33020
+ } catch (err) {
33021
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33022
+ tags: {
33023
+ deviceId: row.id,
33024
+ stableId: row.stableId
33025
+ },
33026
+ meta: {
33027
+ parentDeviceId: id,
33028
+ error: err instanceof Error ? err.message : String(err)
33029
+ }
33030
+ });
33031
+ }
33032
+ }
33033
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33034
+ tags: { deviceId: id },
33035
+ meta: {
33036
+ stableId,
33037
+ type: meta.type
33038
+ }
33039
+ });
33040
+ return { deviceId: id };
32639
33041
  }
32640
33042
  /**
32641
33043
  * Restore devices from persisted state. Two-pass:
@@ -32661,55 +33063,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32661
33063
  * accessory-spawn flow handles via the parent's
32662
33064
  * `getAccessoryChildren()`. Override only when the default doesn't
32663
33065
  * fit.
33066
+ *
33067
+ * A row that fails either pass is NOT terminal (D347): it is handed
33068
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33069
+ * Only after the bound is exhausted is the device marked permanently
33070
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33071
+ * `getStatus().error`.
32664
33072
  */
32665
33073
  async onRestoreDevices(savedDevices) {
32666
33074
  const restored = /* @__PURE__ */ new Set();
33075
+ const failures = [];
33076
+ const attemptRestore = async (saved) => {
33077
+ if (restored.has(saved.id)) return;
33078
+ const Class = this.deviceClasses[saved.type];
33079
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33080
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33081
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33082
+ restored.add(saved.id);
33083
+ };
32667
33084
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32668
33085
  const restoreOne = async (saved) => {
32669
- const Class = this.deviceClasses[saved.type];
32670
- if (!Class) {
33086
+ if (!this.deviceClasses[saved.type]) {
32671
33087
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32672
- tags: { stableId: saved.stableId },
33088
+ tags: {
33089
+ deviceId: saved.id,
33090
+ stableId: saved.stableId
33091
+ },
32673
33092
  meta: { type: saved.type }
32674
33093
  });
32675
33094
  return;
32676
33095
  }
32677
33096
  try {
32678
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32679
- restored.add(saved.id);
33097
+ await attemptRestore(saved);
32680
33098
  } catch (err) {
32681
- this.ctx.logger.warn("Failed to restore device", {
32682
- tags: { stableId: saved.stableId },
33099
+ const error = err instanceof Error ? err.message : String(err);
33100
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33101
+ tags: {
33102
+ deviceId: saved.id,
33103
+ stableId: saved.stableId
33104
+ },
32683
33105
  meta: {
32684
33106
  type: saved.type,
32685
- error: err instanceof Error ? err.message : String(err)
33107
+ attempt: 1,
33108
+ error
32686
33109
  }
32687
33110
  });
33111
+ failures.push({
33112
+ saved,
33113
+ error
33114
+ });
32688
33115
  }
32689
33116
  };
32690
33117
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33118
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32691
33119
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32692
33120
  for (const saved of childRows) {
32693
- const Class = this.deviceClasses[saved.type];
32694
- if (!Class) continue;
33121
+ if (!this.deviceClasses[saved.type]) continue;
32695
33122
  if (saved.parentDeviceId === null) continue;
32696
- if (!restored.has(saved.parentDeviceId)) continue;
32697
- try {
32698
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32699
- restored.add(saved.id);
32700
- } catch (err) {
32701
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33123
+ if (restored.has(saved.parentDeviceId)) {
33124
+ try {
33125
+ await attemptRestore(saved);
33126
+ } catch (err) {
33127
+ const error = err instanceof Error ? err.message : String(err);
33128
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33129
+ tags: {
33130
+ deviceId: saved.id,
33131
+ stableId: saved.stableId,
33132
+ parentDeviceId: saved.parentDeviceId
33133
+ },
33134
+ meta: {
33135
+ type: saved.type,
33136
+ attempt: 1,
33137
+ error
33138
+ }
33139
+ });
33140
+ failures.push({
33141
+ saved,
33142
+ error
33143
+ });
33144
+ }
33145
+ continue;
33146
+ }
33147
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33148
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32702
33149
  tags: {
33150
+ deviceId: saved.id,
32703
33151
  stableId: saved.stableId,
32704
33152
  parentDeviceId: saved.parentDeviceId
32705
33153
  },
32706
- meta: {
32707
- type: saved.type,
32708
- error: err instanceof Error ? err.message : String(err)
32709
- }
33154
+ meta: { type: saved.type }
33155
+ });
33156
+ failures.push({
33157
+ saved,
33158
+ error: `parent device ${saved.parentDeviceId} not restored`
32710
33159
  });
33160
+ continue;
32711
33161
  }
32712
33162
  }
33163
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33164
+ return {
33165
+ restoredCount: restored.size,
33166
+ failedCount: failures.length
33167
+ };
32713
33168
  }
32714
33169
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32715
33170
  toSummary(device) {
@@ -33218,6 +33673,12 @@ Object.freeze({
33218
33673
  addonId: null,
33219
33674
  access: "create"
33220
33675
  },
33676
+ "backup.cancel": {
33677
+ capName: "backup",
33678
+ capScope: "system",
33679
+ addonId: null,
33680
+ access: "create"
33681
+ },
33221
33682
  "backup.delete": {
33222
33683
  capName: "backup",
33223
33684
  capScope: "system",
@@ -33260,6 +33721,12 @@ Object.freeze({
33260
33721
  addonId: null,
33261
33722
  access: "view"
33262
33723
  },
33724
+ "backup.listRuns": {
33725
+ capName: "backup",
33726
+ capScope: "system",
33727
+ addonId: null,
33728
+ access: "view"
33729
+ },
33263
33730
  "backup.listSchedules": {
33264
33731
  capName: "backup",
33265
33732
  capScope: "system",
@@ -34460,6 +34927,12 @@ Object.freeze({
34460
34927
  addonId: null,
34461
34928
  access: "view"
34462
34929
  },
34930
+ "deviceProvider.reloadDevice": {
34931
+ capName: "device-provider",
34932
+ capScope: "system",
34933
+ addonId: null,
34934
+ access: "create"
34935
+ },
34463
34936
  "deviceProvider.start": {
34464
34937
  capName: "device-provider",
34465
34938
  capScope: "system",
@@ -37826,6 +38299,12 @@ Object.freeze({
37826
38299
  addonId: null,
37827
38300
  access: "create"
37828
38301
  },
38302
+ "streamBroker.forgetDeviceHardware": {
38303
+ capName: "stream-broker",
38304
+ capScope: "system",
38305
+ addonId: null,
38306
+ access: "delete"
38307
+ },
37829
38308
  "streamBroker.getAllRtspEntries": {
37830
38309
  capName: "stream-broker",
37831
38310
  capScope: "system",
@@ -40284,6 +40763,11 @@ Object.freeze({
40284
40763
  form: "single",
40285
40764
  optional: false
40286
40765
  }],
40766
+ "streamBroker.forgetDeviceHardware": [{
40767
+ name: "deviceId",
40768
+ form: "single",
40769
+ optional: false
40770
+ }],
40287
40771
  "streamBroker.getDeviceAudioMute": [{
40288
40772
  name: "deviceId",
40289
40773
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-homematic",
3
- "version": "1.2.59",
3
+ "version": "1.2.61",
4
4
  "description": "Homematic / HomematicIP (CCU3 / RaspberryMatic) device-provider addon for CamStack — wraps the nodehomematic library",
5
5
  "keywords": [
6
6
  "camstack",