@camstack/addon-provider-rademacher 0.2.57 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -11512,6 +11512,89 @@ var LocationStatSchema = object({
11512
11512
  fileCount: number(),
11513
11513
  present: boolean()
11514
11514
  });
11515
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
11516
+ var BackupRunStateSchema = _enum([
11517
+ "queued",
11518
+ "running",
11519
+ "succeeded",
11520
+ "failed",
11521
+ "cancelled"
11522
+ ]);
11523
+ /**
11524
+ * Where a running backup currently is. `queued` before it starts,
11525
+ * `building` while the tar.gz is being staged, `uploading` during the
11526
+ * per-destination fan-out, `done` once terminal.
11527
+ */
11528
+ var BackupRunPhaseSchema = _enum([
11529
+ "queued",
11530
+ "building",
11531
+ "uploading",
11532
+ "done"
11533
+ ]);
11534
+ /**
11535
+ * Observable state of one backup run — readable WHILE it runs via
11536
+ * `backup.listRuns`. This is what makes the execution queue and
11537
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
11538
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
11539
+ * diagnosable with `du` because nothing reported that runs existed or
11540
+ * how large the staged archive had grown.
11541
+ */
11542
+ var BackupRunSchema = object({
11543
+ /** Stable run id — the handle `backup.cancel` takes. */
11544
+ id: string(),
11545
+ state: BackupRunStateSchema,
11546
+ phase: BackupRunPhaseSchema,
11547
+ /**
11548
+ * Resolved destination location ids. Empty while queued (targets are
11549
+ * resolved when the run starts, against the then-current policies).
11550
+ */
11551
+ destinationIds: array(string()).readonly(),
11552
+ label: string().optional(),
11553
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
11554
+ requestedAt: number(),
11555
+ /** ms-epoch when the run left the queue and started building. */
11556
+ startedAt: number().optional(),
11557
+ /** ms-epoch when the run reached a terminal state. */
11558
+ finishedAt: number().optional(),
11559
+ /** Compressed bytes of the staging archive written so far. */
11560
+ stagedBytes: number(),
11561
+ /** Final staged archive size, once the build phase completes. */
11562
+ archiveSizeBytes: number().optional(),
11563
+ /** Bytes pushed to the destination currently uploading. */
11564
+ uploadedBytes: number(),
11565
+ /** Destinations where the archive fully landed (uploaded + indexed). */
11566
+ completedDestinationIds: array(string()).readonly(),
11567
+ /** Destinations that failed during the fan-out. */
11568
+ failedDestinationIds: array(string()).readonly(),
11569
+ /** Failure message when `state === 'failed'`. */
11570
+ error: string().optional(),
11571
+ /**
11572
+ * 1-based place in the execution queue — 1 = runs next. Present only
11573
+ * while `state === 'queued'`. Stamped by the orchestrator from the
11574
+ * queue's OWN pending order, never derived from timestamps, so the
11575
+ * UI cannot show an order the executor will not honour.
11576
+ */
11577
+ queuePosition: number().int().min(1).optional()
11578
+ });
11579
+ /**
11580
+ * Result of `backup.trigger`. The call still resolves when the run
11581
+ * terminates (compat with schedule-driven runs and the admin UI), but
11582
+ * it now names the run and says whether it had to WAIT: a trigger that
11583
+ * arrives while another run is in flight is enqueued (or joined onto
11584
+ * an identical already-queued run), never started concurrently.
11585
+ */
11586
+ var BackupTriggerResultSchema = object({
11587
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
11588
+ runId: string(),
11589
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
11590
+ queued: boolean(),
11591
+ /** True when this trigger was coalesced onto an identical already-queued run. */
11592
+ joined: boolean(),
11593
+ /** True when the run was cancelled before completing every destination. */
11594
+ cancelled: boolean(),
11595
+ /** One entry per destination the archive landed at (partial on cancel). */
11596
+ entries: array(BackupEntrySchema).readonly()
11597
+ });
11515
11598
  /**
11516
11599
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
11517
11600
  * SET of destination locations. Supersedes the per-location cron on
@@ -11559,7 +11642,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
11559
11642
  * retention (manual runs).
11560
11643
  */
11561
11644
  retentionCount: number().int().min(1).max(1e3).optional()
11562
- }).optional(), array(BackupEntrySchema).readonly(), {
11645
+ }).optional(), BackupTriggerResultSchema, {
11646
+ kind: "mutation",
11647
+ auth: "admin"
11648
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
11563
11649
  kind: "mutation",
11564
11650
  auth: "admin"
11565
11651
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -12276,6 +12362,14 @@ method(object({
12276
12362
  }), object({ success: literal(true) }), {
12277
12363
  kind: "mutation",
12278
12364
  auth: "admin"
12365
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
12366
+ derivedStreamsDeleted: array(string()).readonly(),
12367
+ assignmentsPurged: boolean(),
12368
+ probeSnapshotsDropped: number().int().nonnegative(),
12369
+ rtspTokenRowsDeleted: number().int().nonnegative()
12370
+ }), {
12371
+ kind: "mutation",
12372
+ auth: "admin"
12279
12373
  }), method(object({
12280
12374
  deviceId: number(),
12281
12375
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13703,6 +13797,35 @@ var deviceProviderCapability = {
13703
13797
  name: string(),
13704
13798
  type: string()
13705
13799
  }))),
13800
+ /**
13801
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13802
+ * touching no other device this provider owns.
13803
+ *
13804
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13805
+ * migrated numbers: after `swapIds` the runner's live instance still
13806
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13807
+ * registrations and its log tags), and a live object cannot be renumbered.
13808
+ * Before this method the only flush was restarting the whole owning addon
13809
+ * — which took every camera the provider owns down with it (28 devices
13810
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13811
+ * same day ~27 devices' native caps did not come back on their own).
13812
+ *
13813
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13814
+ * that changes. The reply carries the id the device answers on NOW.
13815
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13816
+ * instance (if any), then re-create from the persisted row: the same
13817
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13818
+ * An RPC, never an event: a dropped event would leave the runner writing
13819
+ * against the wrong camera (D8).
13820
+ *
13821
+ * Construction can dial hardware, and the migrated source is
13822
+ * characteristically dead — the timeout covers a full activate window
13823
+ * rather than the 60 s default.
13824
+ */
13825
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13826
+ kind: "mutation",
13827
+ timeoutMs: 3 * 6e4
13828
+ }),
13706
13829
  supportsDiscovery: method(object({}), boolean()),
13707
13830
  /**
13708
13831
  * Run a network scan. `params` carries optional provider-specific scan
@@ -14030,7 +14153,8 @@ method(object({
14030
14153
  targetId: number()
14031
14154
  }), MigrateDeviceResultSchema, {
14032
14155
  kind: "mutation",
14033
- auth: "admin"
14156
+ auth: "admin",
14157
+ timeoutMs: 12 * 6e4
14034
14158
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
14035
14159
  deviceId: number(),
14036
14160
  name: string()
@@ -33386,6 +33510,147 @@ var BaseDevice = class {
33386
33510
  }
33387
33511
  };
33388
33512
  /**
33513
+ * Delays before retry rounds 1..N — the round count IS the bound.
33514
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33515
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33516
+ * per attempt) covers a device-manager lock held for minutes — the
33517
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33518
+ */
33519
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33520
+ 1e4,
33521
+ 3e4,
33522
+ 9e4
33523
+ ];
33524
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33525
+ function sleep$1(ms, signal) {
33526
+ return new Promise((resolve) => {
33527
+ if (signal.aborted) {
33528
+ resolve();
33529
+ return;
33530
+ }
33531
+ const onAbort = () => {
33532
+ clearTimeout(timer);
33533
+ resolve();
33534
+ };
33535
+ const timer = setTimeout(() => {
33536
+ signal.removeEventListener("abort", onAbort);
33537
+ resolve();
33538
+ }, ms);
33539
+ timer.unref?.();
33540
+ signal.addEventListener("abort", onAbort, { once: true });
33541
+ });
33542
+ }
33543
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33544
+ * not reject (callers wrap their own try/catch). */
33545
+ async function runWithConcurrency(items, width, fn) {
33546
+ const queue = [...items];
33547
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33548
+ const lane = async () => {
33549
+ for (;;) {
33550
+ const item = queue.shift();
33551
+ if (item === void 0) return;
33552
+ await fn(item);
33553
+ }
33554
+ };
33555
+ await Promise.all(Array.from({ length: laneCount }, lane));
33556
+ }
33557
+ var DeviceRestoreRetryScheduler = class {
33558
+ #logger;
33559
+ #attempt;
33560
+ #onPermanentFailure;
33561
+ #delaysMs;
33562
+ #concurrency;
33563
+ #now;
33564
+ #abort = new AbortController();
33565
+ constructor(options) {
33566
+ this.#logger = options.logger;
33567
+ this.#attempt = options.attempt;
33568
+ this.#onPermanentFailure = options.onPermanentFailure;
33569
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33570
+ this.#concurrency = options.concurrency ?? 4;
33571
+ this.#now = options.now ?? Date.now;
33572
+ }
33573
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33574
+ * permanently failed — the next boot restores them from disk. */
33575
+ cancel() {
33576
+ this.#abort.abort();
33577
+ }
33578
+ /**
33579
+ * Run the bounded retry rounds. Resolves when every entry has either
33580
+ * restored, been marked permanently failed, or the scheduler was
33581
+ * cancelled. Never rejects.
33582
+ */
33583
+ async run(initialFailures) {
33584
+ let pending = initialFailures.map((failure) => ({
33585
+ saved: failure.saved,
33586
+ lastError: failure.error,
33587
+ attempts: 1
33588
+ }));
33589
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33590
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33591
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33592
+ if (this.#abort.signal.aborted) break;
33593
+ pending = await this.#runRound(pending, round);
33594
+ }
33595
+ if (this.#abort.signal.aborted) return [];
33596
+ const terminal = pending.map((entry) => ({
33597
+ deviceId: entry.saved.id,
33598
+ stableId: entry.saved.stableId,
33599
+ type: String(entry.saved.type),
33600
+ attempts: entry.attempts,
33601
+ lastError: entry.lastError,
33602
+ failedAt: this.#now()
33603
+ }));
33604
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33605
+ return terminal;
33606
+ }
33607
+ /** One retry round: parents first (phase 0), then hub-adopted
33608
+ * children (phase 1) — a child's attempt depends on its parent
33609
+ * having landed, exactly like the initial two-pass restore. */
33610
+ async #runRound(pending, round) {
33611
+ const next = [];
33612
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33613
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33614
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33615
+ if (this.#abort.signal.aborted) {
33616
+ next.push(entry);
33617
+ return;
33618
+ }
33619
+ const attemptNo = entry.attempts + 1;
33620
+ try {
33621
+ await this.#attempt(entry.saved);
33622
+ this.#logger.info("Device restored on retry", {
33623
+ tags: {
33624
+ deviceId: entry.saved.id,
33625
+ stableId: entry.saved.stableId
33626
+ },
33627
+ meta: { attempt: attemptNo }
33628
+ });
33629
+ } catch (err) {
33630
+ const lastError = err instanceof Error ? err.message : String(err);
33631
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33632
+ this.#logger.warn("Device restore retry failed", {
33633
+ tags: {
33634
+ deviceId: entry.saved.id,
33635
+ stableId: entry.saved.stableId
33636
+ },
33637
+ meta: {
33638
+ attempt: attemptNo,
33639
+ remainingRetries,
33640
+ error: lastError
33641
+ }
33642
+ });
33643
+ next.push({
33644
+ saved: entry.saved,
33645
+ lastError,
33646
+ attempts: attemptNo
33647
+ });
33648
+ }
33649
+ });
33650
+ return next;
33651
+ }
33652
+ };
33653
+ /**
33389
33654
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33390
33655
  * device-provider cap router. Shared across all providers.
33391
33656
  */
@@ -33434,6 +33699,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33434
33699
  }];
33435
33700
  }
33436
33701
  async onShutdown() {
33702
+ this.cancelRestoreRetries();
33437
33703
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33438
33704
  for (const device of devices) try {
33439
33705
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33451,9 +33717,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33451
33717
  async start() {}
33452
33718
  async stop() {}
33453
33719
  async getStatus() {
33720
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33721
+ const summary = this.restoreFailureSummary();
33722
+ if (summary === null) return {
33723
+ connected: true,
33724
+ deviceCount: all.length
33725
+ };
33454
33726
  return {
33455
33727
  connected: true,
33456
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33728
+ deviceCount: all.length,
33729
+ error: summary
33457
33730
  };
33458
33731
  }
33459
33732
  async getDevices() {
@@ -33543,8 +33816,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33543
33816
  };
33544
33817
  }
33545
33818
  async restoreDevices(savedDevices) {
33546
- await this.onRestoreDevices(savedDevices);
33547
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33819
+ const report = await this.onRestoreDevices(savedDevices);
33820
+ if (savedDevices.length === 0) return;
33821
+ if (report && report.failedCount > 0) {
33822
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33823
+ return;
33824
+ }
33825
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33826
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33827
+ }
33828
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33829
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33830
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33831
+ * never re-stampede full-width while the initial pass does (D167). */
33832
+ restoreRetryConcurrency = 4;
33833
+ _restoreRetryScheduler = null;
33834
+ _restoreRetryCompletion = null;
33835
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33836
+ /** Settles when the background retry rounds finish (or `null` when
33837
+ * nothing failed). Exposed for tests and subclass diagnostics —
33838
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33839
+ * with the devices that restored, and a late success is announced
33840
+ * through the `native-cap-change` → `updateCaps` path. */
33841
+ get restoreRetryCompletion() {
33842
+ return this._restoreRetryCompletion;
33843
+ }
33844
+ /** Devices that exhausted the retry bound this process lifetime. */
33845
+ get permanentRestoreFailures() {
33846
+ return [...this._permanentRestoreFailures.values()];
33847
+ }
33848
+ /** One-line operator-facing summary for `getStatus().error`, or
33849
+ * `null` when every device restored. */
33850
+ restoreFailureSummary() {
33851
+ if (this._permanentRestoreFailures.size === 0) return null;
33852
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33853
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33854
+ }
33855
+ cancelRestoreRetries() {
33856
+ this._restoreRetryScheduler?.cancel();
33857
+ this._restoreRetryScheduler = null;
33858
+ }
33859
+ recordPermanentRestoreFailure(failure) {
33860
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33861
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33862
+ tags: {
33863
+ deviceId: failure.deviceId,
33864
+ stableId: failure.stableId
33865
+ },
33866
+ meta: {
33867
+ type: failure.type,
33868
+ attempts: failure.attempts,
33869
+ error: failure.lastError
33870
+ }
33871
+ });
33872
+ }
33873
+ scheduleRestoreRetries(failures, attempt) {
33874
+ const scheduler = new DeviceRestoreRetryScheduler({
33875
+ logger: this.ctx.logger,
33876
+ delaysMs: this.restoreRetryDelaysMs,
33877
+ concurrency: this.restoreRetryConcurrency,
33878
+ attempt,
33879
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33880
+ });
33881
+ this._restoreRetryScheduler = scheduler;
33882
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33883
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33884
+ });
33885
+ }
33886
+ /**
33887
+ * Tear down and reconstruct ONE device from its persisted rows — the
33888
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33889
+ * and no other device this provider owns is disturbed.
33890
+ *
33891
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33892
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33893
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33894
+ * whatever number the row carries NOW. The teardown is `decommission` —
33895
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33896
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33897
+ * the boot restore's own `create()` path, including its pass 2: first-class
33898
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33899
+ * parent by the cascade and must be re-created explicitly, because only
33900
+ * accessory children come back through `getAccessoryChildren()`.
33901
+ *
33902
+ * Reloading an accessory child directly is refused (no device class) —
33903
+ * reload its parent instead.
33904
+ */
33905
+ async reloadDevice(input) {
33906
+ const { stableId } = input;
33907
+ const devices = this.ctx.kernel.devices;
33908
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33909
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33910
+ if (live) await devices.decommission(live.id);
33911
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33912
+ addonId: this.addonId,
33913
+ stableId
33914
+ });
33915
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33916
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33917
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33918
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33919
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33920
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33921
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33922
+ for (const row of rows) {
33923
+ if (row.parentDeviceId !== id) continue;
33924
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33925
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33926
+ if (!ChildClass) continue;
33927
+ try {
33928
+ await devices.create(row.stableId, ChildClass, {}, id);
33929
+ } catch (err) {
33930
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33931
+ tags: {
33932
+ deviceId: row.id,
33933
+ stableId: row.stableId
33934
+ },
33935
+ meta: {
33936
+ parentDeviceId: id,
33937
+ error: err instanceof Error ? err.message : String(err)
33938
+ }
33939
+ });
33940
+ }
33941
+ }
33942
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33943
+ tags: { deviceId: id },
33944
+ meta: {
33945
+ stableId,
33946
+ type: meta.type
33947
+ }
33948
+ });
33949
+ return { deviceId: id };
33548
33950
  }
33549
33951
  /**
33550
33952
  * Restore devices from persisted state. Two-pass:
@@ -33570,55 +33972,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33570
33972
  * accessory-spawn flow handles via the parent's
33571
33973
  * `getAccessoryChildren()`. Override only when the default doesn't
33572
33974
  * fit.
33975
+ *
33976
+ * A row that fails either pass is NOT terminal (D347): it is handed
33977
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33978
+ * Only after the bound is exhausted is the device marked permanently
33979
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33980
+ * `getStatus().error`.
33573
33981
  */
33574
33982
  async onRestoreDevices(savedDevices) {
33575
33983
  const restored = /* @__PURE__ */ new Set();
33984
+ const failures = [];
33985
+ const attemptRestore = async (saved) => {
33986
+ if (restored.has(saved.id)) return;
33987
+ const Class = this.deviceClasses[saved.type];
33988
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33989
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33990
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33991
+ restored.add(saved.id);
33992
+ };
33576
33993
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33577
33994
  const restoreOne = async (saved) => {
33578
- const Class = this.deviceClasses[saved.type];
33579
- if (!Class) {
33995
+ if (!this.deviceClasses[saved.type]) {
33580
33996
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33581
- tags: { stableId: saved.stableId },
33997
+ tags: {
33998
+ deviceId: saved.id,
33999
+ stableId: saved.stableId
34000
+ },
33582
34001
  meta: { type: saved.type }
33583
34002
  });
33584
34003
  return;
33585
34004
  }
33586
34005
  try {
33587
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33588
- restored.add(saved.id);
34006
+ await attemptRestore(saved);
33589
34007
  } catch (err) {
33590
- this.ctx.logger.warn("Failed to restore device", {
33591
- tags: { stableId: saved.stableId },
34008
+ const error = err instanceof Error ? err.message : String(err);
34009
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
34010
+ tags: {
34011
+ deviceId: saved.id,
34012
+ stableId: saved.stableId
34013
+ },
33592
34014
  meta: {
33593
34015
  type: saved.type,
33594
- error: err instanceof Error ? err.message : String(err)
34016
+ attempt: 1,
34017
+ error
33595
34018
  }
33596
34019
  });
34020
+ failures.push({
34021
+ saved,
34022
+ error
34023
+ });
33597
34024
  }
33598
34025
  };
33599
34026
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
34027
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33600
34028
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33601
34029
  for (const saved of childRows) {
33602
- const Class = this.deviceClasses[saved.type];
33603
- if (!Class) continue;
34030
+ if (!this.deviceClasses[saved.type]) continue;
33604
34031
  if (saved.parentDeviceId === null) continue;
33605
- if (!restored.has(saved.parentDeviceId)) continue;
33606
- try {
33607
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33608
- restored.add(saved.id);
33609
- } catch (err) {
33610
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
34032
+ if (restored.has(saved.parentDeviceId)) {
34033
+ try {
34034
+ await attemptRestore(saved);
34035
+ } catch (err) {
34036
+ const error = err instanceof Error ? err.message : String(err);
34037
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
34038
+ tags: {
34039
+ deviceId: saved.id,
34040
+ stableId: saved.stableId,
34041
+ parentDeviceId: saved.parentDeviceId
34042
+ },
34043
+ meta: {
34044
+ type: saved.type,
34045
+ attempt: 1,
34046
+ error
34047
+ }
34048
+ });
34049
+ failures.push({
34050
+ saved,
34051
+ error
34052
+ });
34053
+ }
34054
+ continue;
34055
+ }
34056
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
34057
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33611
34058
  tags: {
34059
+ deviceId: saved.id,
33612
34060
  stableId: saved.stableId,
33613
34061
  parentDeviceId: saved.parentDeviceId
33614
34062
  },
33615
- meta: {
33616
- type: saved.type,
33617
- error: err instanceof Error ? err.message : String(err)
33618
- }
34063
+ meta: { type: saved.type }
33619
34064
  });
34065
+ failures.push({
34066
+ saved,
34067
+ error: `parent device ${saved.parentDeviceId} not restored`
34068
+ });
34069
+ continue;
33620
34070
  }
33621
34071
  }
34072
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
34073
+ return {
34074
+ restoredCount: restored.size,
34075
+ failedCount: failures.length
34076
+ };
33622
34077
  }
33623
34078
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33624
34079
  toSummary(device) {
@@ -34127,6 +34582,12 @@ Object.freeze({
34127
34582
  addonId: null,
34128
34583
  access: "create"
34129
34584
  },
34585
+ "backup.cancel": {
34586
+ capName: "backup",
34587
+ capScope: "system",
34588
+ addonId: null,
34589
+ access: "create"
34590
+ },
34130
34591
  "backup.delete": {
34131
34592
  capName: "backup",
34132
34593
  capScope: "system",
@@ -34169,6 +34630,12 @@ Object.freeze({
34169
34630
  addonId: null,
34170
34631
  access: "view"
34171
34632
  },
34633
+ "backup.listRuns": {
34634
+ capName: "backup",
34635
+ capScope: "system",
34636
+ addonId: null,
34637
+ access: "view"
34638
+ },
34172
34639
  "backup.listSchedules": {
34173
34640
  capName: "backup",
34174
34641
  capScope: "system",
@@ -35369,6 +35836,12 @@ Object.freeze({
35369
35836
  addonId: null,
35370
35837
  access: "view"
35371
35838
  },
35839
+ "deviceProvider.reloadDevice": {
35840
+ capName: "device-provider",
35841
+ capScope: "system",
35842
+ addonId: null,
35843
+ access: "create"
35844
+ },
35372
35845
  "deviceProvider.start": {
35373
35846
  capName: "device-provider",
35374
35847
  capScope: "system",
@@ -38735,6 +39208,12 @@ Object.freeze({
38735
39208
  addonId: null,
38736
39209
  access: "create"
38737
39210
  },
39211
+ "streamBroker.forgetDeviceHardware": {
39212
+ capName: "stream-broker",
39213
+ capScope: "system",
39214
+ addonId: null,
39215
+ access: "delete"
39216
+ },
38738
39217
  "streamBroker.getAllRtspEntries": {
38739
39218
  capName: "stream-broker",
38740
39219
  capScope: "system",
@@ -41193,6 +41672,11 @@ Object.freeze({
41193
41672
  form: "single",
41194
41673
  optional: false
41195
41674
  }],
41675
+ "streamBroker.forgetDeviceHardware": [{
41676
+ name: "deviceId",
41677
+ form: "single",
41678
+ optional: false
41679
+ }],
41196
41680
  "streamBroker.getDeviceAudioMute": [{
41197
41681
  name: "deviceId",
41198
41682
  form: "single",
package/dist/addon.mjs CHANGED
@@ -11511,6 +11511,89 @@ var LocationStatSchema = object({
11511
11511
  fileCount: number(),
11512
11512
  present: boolean()
11513
11513
  });
11514
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
11515
+ var BackupRunStateSchema = _enum([
11516
+ "queued",
11517
+ "running",
11518
+ "succeeded",
11519
+ "failed",
11520
+ "cancelled"
11521
+ ]);
11522
+ /**
11523
+ * Where a running backup currently is. `queued` before it starts,
11524
+ * `building` while the tar.gz is being staged, `uploading` during the
11525
+ * per-destination fan-out, `done` once terminal.
11526
+ */
11527
+ var BackupRunPhaseSchema = _enum([
11528
+ "queued",
11529
+ "building",
11530
+ "uploading",
11531
+ "done"
11532
+ ]);
11533
+ /**
11534
+ * Observable state of one backup run — readable WHILE it runs via
11535
+ * `backup.listRuns`. This is what makes the execution queue and
11536
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
11537
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
11538
+ * diagnosable with `du` because nothing reported that runs existed or
11539
+ * how large the staged archive had grown.
11540
+ */
11541
+ var BackupRunSchema = object({
11542
+ /** Stable run id — the handle `backup.cancel` takes. */
11543
+ id: string(),
11544
+ state: BackupRunStateSchema,
11545
+ phase: BackupRunPhaseSchema,
11546
+ /**
11547
+ * Resolved destination location ids. Empty while queued (targets are
11548
+ * resolved when the run starts, against the then-current policies).
11549
+ */
11550
+ destinationIds: array(string()).readonly(),
11551
+ label: string().optional(),
11552
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
11553
+ requestedAt: number(),
11554
+ /** ms-epoch when the run left the queue and started building. */
11555
+ startedAt: number().optional(),
11556
+ /** ms-epoch when the run reached a terminal state. */
11557
+ finishedAt: number().optional(),
11558
+ /** Compressed bytes of the staging archive written so far. */
11559
+ stagedBytes: number(),
11560
+ /** Final staged archive size, once the build phase completes. */
11561
+ archiveSizeBytes: number().optional(),
11562
+ /** Bytes pushed to the destination currently uploading. */
11563
+ uploadedBytes: number(),
11564
+ /** Destinations where the archive fully landed (uploaded + indexed). */
11565
+ completedDestinationIds: array(string()).readonly(),
11566
+ /** Destinations that failed during the fan-out. */
11567
+ failedDestinationIds: array(string()).readonly(),
11568
+ /** Failure message when `state === 'failed'`. */
11569
+ error: string().optional(),
11570
+ /**
11571
+ * 1-based place in the execution queue — 1 = runs next. Present only
11572
+ * while `state === 'queued'`. Stamped by the orchestrator from the
11573
+ * queue's OWN pending order, never derived from timestamps, so the
11574
+ * UI cannot show an order the executor will not honour.
11575
+ */
11576
+ queuePosition: number().int().min(1).optional()
11577
+ });
11578
+ /**
11579
+ * Result of `backup.trigger`. The call still resolves when the run
11580
+ * terminates (compat with schedule-driven runs and the admin UI), but
11581
+ * it now names the run and says whether it had to WAIT: a trigger that
11582
+ * arrives while another run is in flight is enqueued (or joined onto
11583
+ * an identical already-queued run), never started concurrently.
11584
+ */
11585
+ var BackupTriggerResultSchema = object({
11586
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
11587
+ runId: string(),
11588
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
11589
+ queued: boolean(),
11590
+ /** True when this trigger was coalesced onto an identical already-queued run. */
11591
+ joined: boolean(),
11592
+ /** True when the run was cancelled before completing every destination. */
11593
+ cancelled: boolean(),
11594
+ /** One entry per destination the archive landed at (partial on cancel). */
11595
+ entries: array(BackupEntrySchema).readonly()
11596
+ });
11514
11597
  /**
11515
11598
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
11516
11599
  * SET of destination locations. Supersedes the per-location cron on
@@ -11558,7 +11641,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
11558
11641
  * retention (manual runs).
11559
11642
  */
11560
11643
  retentionCount: number().int().min(1).max(1e3).optional()
11561
- }).optional(), array(BackupEntrySchema).readonly(), {
11644
+ }).optional(), BackupTriggerResultSchema, {
11645
+ kind: "mutation",
11646
+ auth: "admin"
11647
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
11562
11648
  kind: "mutation",
11563
11649
  auth: "admin"
11564
11650
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -12275,6 +12361,14 @@ method(object({
12275
12361
  }), object({ success: literal(true) }), {
12276
12362
  kind: "mutation",
12277
12363
  auth: "admin"
12364
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
12365
+ derivedStreamsDeleted: array(string()).readonly(),
12366
+ assignmentsPurged: boolean(),
12367
+ probeSnapshotsDropped: number().int().nonnegative(),
12368
+ rtspTokenRowsDeleted: number().int().nonnegative()
12369
+ }), {
12370
+ kind: "mutation",
12371
+ auth: "admin"
12278
12372
  }), method(object({
12279
12373
  deviceId: number(),
12280
12374
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13702,6 +13796,35 @@ var deviceProviderCapability = {
13702
13796
  name: string(),
13703
13797
  type: string()
13704
13798
  }))),
13799
+ /**
13800
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13801
+ * touching no other device this provider owns.
13802
+ *
13803
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13804
+ * migrated numbers: after `swapIds` the runner's live instance still
13805
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13806
+ * registrations and its log tags), and a live object cannot be renumbered.
13807
+ * Before this method the only flush was restarting the whole owning addon
13808
+ * — which took every camera the provider owns down with it (28 devices
13809
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13810
+ * same day ~27 devices' native caps did not come back on their own).
13811
+ *
13812
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13813
+ * that changes. The reply carries the id the device answers on NOW.
13814
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13815
+ * instance (if any), then re-create from the persisted row: the same
13816
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13817
+ * An RPC, never an event: a dropped event would leave the runner writing
13818
+ * against the wrong camera (D8).
13819
+ *
13820
+ * Construction can dial hardware, and the migrated source is
13821
+ * characteristically dead — the timeout covers a full activate window
13822
+ * rather than the 60 s default.
13823
+ */
13824
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13825
+ kind: "mutation",
13826
+ timeoutMs: 3 * 6e4
13827
+ }),
13705
13828
  supportsDiscovery: method(object({}), boolean()),
13706
13829
  /**
13707
13830
  * Run a network scan. `params` carries optional provider-specific scan
@@ -14029,7 +14152,8 @@ method(object({
14029
14152
  targetId: number()
14030
14153
  }), MigrateDeviceResultSchema, {
14031
14154
  kind: "mutation",
14032
- auth: "admin"
14155
+ auth: "admin",
14156
+ timeoutMs: 12 * 6e4
14033
14157
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
14034
14158
  deviceId: number(),
14035
14159
  name: string()
@@ -33385,6 +33509,147 @@ var BaseDevice = class {
33385
33509
  }
33386
33510
  };
33387
33511
  /**
33512
+ * Delays before retry rounds 1..N — the round count IS the bound.
33513
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33514
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33515
+ * per attempt) covers a device-manager lock held for minutes — the
33516
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33517
+ */
33518
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33519
+ 1e4,
33520
+ 3e4,
33521
+ 9e4
33522
+ ];
33523
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33524
+ function sleep$1(ms, signal) {
33525
+ return new Promise((resolve) => {
33526
+ if (signal.aborted) {
33527
+ resolve();
33528
+ return;
33529
+ }
33530
+ const onAbort = () => {
33531
+ clearTimeout(timer);
33532
+ resolve();
33533
+ };
33534
+ const timer = setTimeout(() => {
33535
+ signal.removeEventListener("abort", onAbort);
33536
+ resolve();
33537
+ }, ms);
33538
+ timer.unref?.();
33539
+ signal.addEventListener("abort", onAbort, { once: true });
33540
+ });
33541
+ }
33542
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33543
+ * not reject (callers wrap their own try/catch). */
33544
+ async function runWithConcurrency(items, width, fn) {
33545
+ const queue = [...items];
33546
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33547
+ const lane = async () => {
33548
+ for (;;) {
33549
+ const item = queue.shift();
33550
+ if (item === void 0) return;
33551
+ await fn(item);
33552
+ }
33553
+ };
33554
+ await Promise.all(Array.from({ length: laneCount }, lane));
33555
+ }
33556
+ var DeviceRestoreRetryScheduler = class {
33557
+ #logger;
33558
+ #attempt;
33559
+ #onPermanentFailure;
33560
+ #delaysMs;
33561
+ #concurrency;
33562
+ #now;
33563
+ #abort = new AbortController();
33564
+ constructor(options) {
33565
+ this.#logger = options.logger;
33566
+ this.#attempt = options.attempt;
33567
+ this.#onPermanentFailure = options.onPermanentFailure;
33568
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33569
+ this.#concurrency = options.concurrency ?? 4;
33570
+ this.#now = options.now ?? Date.now;
33571
+ }
33572
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33573
+ * permanently failed — the next boot restores them from disk. */
33574
+ cancel() {
33575
+ this.#abort.abort();
33576
+ }
33577
+ /**
33578
+ * Run the bounded retry rounds. Resolves when every entry has either
33579
+ * restored, been marked permanently failed, or the scheduler was
33580
+ * cancelled. Never rejects.
33581
+ */
33582
+ async run(initialFailures) {
33583
+ let pending = initialFailures.map((failure) => ({
33584
+ saved: failure.saved,
33585
+ lastError: failure.error,
33586
+ attempts: 1
33587
+ }));
33588
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33589
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33590
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33591
+ if (this.#abort.signal.aborted) break;
33592
+ pending = await this.#runRound(pending, round);
33593
+ }
33594
+ if (this.#abort.signal.aborted) return [];
33595
+ const terminal = pending.map((entry) => ({
33596
+ deviceId: entry.saved.id,
33597
+ stableId: entry.saved.stableId,
33598
+ type: String(entry.saved.type),
33599
+ attempts: entry.attempts,
33600
+ lastError: entry.lastError,
33601
+ failedAt: this.#now()
33602
+ }));
33603
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33604
+ return terminal;
33605
+ }
33606
+ /** One retry round: parents first (phase 0), then hub-adopted
33607
+ * children (phase 1) — a child's attempt depends on its parent
33608
+ * having landed, exactly like the initial two-pass restore. */
33609
+ async #runRound(pending, round) {
33610
+ const next = [];
33611
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33612
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33613
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33614
+ if (this.#abort.signal.aborted) {
33615
+ next.push(entry);
33616
+ return;
33617
+ }
33618
+ const attemptNo = entry.attempts + 1;
33619
+ try {
33620
+ await this.#attempt(entry.saved);
33621
+ this.#logger.info("Device restored on retry", {
33622
+ tags: {
33623
+ deviceId: entry.saved.id,
33624
+ stableId: entry.saved.stableId
33625
+ },
33626
+ meta: { attempt: attemptNo }
33627
+ });
33628
+ } catch (err) {
33629
+ const lastError = err instanceof Error ? err.message : String(err);
33630
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33631
+ this.#logger.warn("Device restore retry failed", {
33632
+ tags: {
33633
+ deviceId: entry.saved.id,
33634
+ stableId: entry.saved.stableId
33635
+ },
33636
+ meta: {
33637
+ attempt: attemptNo,
33638
+ remainingRetries,
33639
+ error: lastError
33640
+ }
33641
+ });
33642
+ next.push({
33643
+ saved: entry.saved,
33644
+ lastError,
33645
+ attempts: attemptNo
33646
+ });
33647
+ }
33648
+ });
33649
+ return next;
33650
+ }
33651
+ };
33652
+ /**
33388
33653
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33389
33654
  * device-provider cap router. Shared across all providers.
33390
33655
  */
@@ -33433,6 +33698,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33433
33698
  }];
33434
33699
  }
33435
33700
  async onShutdown() {
33701
+ this.cancelRestoreRetries();
33436
33702
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33437
33703
  for (const device of devices) try {
33438
33704
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33450,9 +33716,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33450
33716
  async start() {}
33451
33717
  async stop() {}
33452
33718
  async getStatus() {
33719
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33720
+ const summary = this.restoreFailureSummary();
33721
+ if (summary === null) return {
33722
+ connected: true,
33723
+ deviceCount: all.length
33724
+ };
33453
33725
  return {
33454
33726
  connected: true,
33455
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33727
+ deviceCount: all.length,
33728
+ error: summary
33456
33729
  };
33457
33730
  }
33458
33731
  async getDevices() {
@@ -33542,8 +33815,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33542
33815
  };
33543
33816
  }
33544
33817
  async restoreDevices(savedDevices) {
33545
- await this.onRestoreDevices(savedDevices);
33546
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33818
+ const report = await this.onRestoreDevices(savedDevices);
33819
+ if (savedDevices.length === 0) return;
33820
+ if (report && report.failedCount > 0) {
33821
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33822
+ return;
33823
+ }
33824
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33825
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33826
+ }
33827
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33828
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33829
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33830
+ * never re-stampede full-width while the initial pass does (D167). */
33831
+ restoreRetryConcurrency = 4;
33832
+ _restoreRetryScheduler = null;
33833
+ _restoreRetryCompletion = null;
33834
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33835
+ /** Settles when the background retry rounds finish (or `null` when
33836
+ * nothing failed). Exposed for tests and subclass diagnostics —
33837
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33838
+ * with the devices that restored, and a late success is announced
33839
+ * through the `native-cap-change` → `updateCaps` path. */
33840
+ get restoreRetryCompletion() {
33841
+ return this._restoreRetryCompletion;
33842
+ }
33843
+ /** Devices that exhausted the retry bound this process lifetime. */
33844
+ get permanentRestoreFailures() {
33845
+ return [...this._permanentRestoreFailures.values()];
33846
+ }
33847
+ /** One-line operator-facing summary for `getStatus().error`, or
33848
+ * `null` when every device restored. */
33849
+ restoreFailureSummary() {
33850
+ if (this._permanentRestoreFailures.size === 0) return null;
33851
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33852
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33853
+ }
33854
+ cancelRestoreRetries() {
33855
+ this._restoreRetryScheduler?.cancel();
33856
+ this._restoreRetryScheduler = null;
33857
+ }
33858
+ recordPermanentRestoreFailure(failure) {
33859
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33860
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33861
+ tags: {
33862
+ deviceId: failure.deviceId,
33863
+ stableId: failure.stableId
33864
+ },
33865
+ meta: {
33866
+ type: failure.type,
33867
+ attempts: failure.attempts,
33868
+ error: failure.lastError
33869
+ }
33870
+ });
33871
+ }
33872
+ scheduleRestoreRetries(failures, attempt) {
33873
+ const scheduler = new DeviceRestoreRetryScheduler({
33874
+ logger: this.ctx.logger,
33875
+ delaysMs: this.restoreRetryDelaysMs,
33876
+ concurrency: this.restoreRetryConcurrency,
33877
+ attempt,
33878
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33879
+ });
33880
+ this._restoreRetryScheduler = scheduler;
33881
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33882
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33883
+ });
33884
+ }
33885
+ /**
33886
+ * Tear down and reconstruct ONE device from its persisted rows — the
33887
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33888
+ * and no other device this provider owns is disturbed.
33889
+ *
33890
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33891
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33892
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33893
+ * whatever number the row carries NOW. The teardown is `decommission` —
33894
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33895
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33896
+ * the boot restore's own `create()` path, including its pass 2: first-class
33897
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33898
+ * parent by the cascade and must be re-created explicitly, because only
33899
+ * accessory children come back through `getAccessoryChildren()`.
33900
+ *
33901
+ * Reloading an accessory child directly is refused (no device class) —
33902
+ * reload its parent instead.
33903
+ */
33904
+ async reloadDevice(input) {
33905
+ const { stableId } = input;
33906
+ const devices = this.ctx.kernel.devices;
33907
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33908
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33909
+ if (live) await devices.decommission(live.id);
33910
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33911
+ addonId: this.addonId,
33912
+ stableId
33913
+ });
33914
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33915
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33916
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33917
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33918
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33919
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33920
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33921
+ for (const row of rows) {
33922
+ if (row.parentDeviceId !== id) continue;
33923
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33924
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33925
+ if (!ChildClass) continue;
33926
+ try {
33927
+ await devices.create(row.stableId, ChildClass, {}, id);
33928
+ } catch (err) {
33929
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33930
+ tags: {
33931
+ deviceId: row.id,
33932
+ stableId: row.stableId
33933
+ },
33934
+ meta: {
33935
+ parentDeviceId: id,
33936
+ error: err instanceof Error ? err.message : String(err)
33937
+ }
33938
+ });
33939
+ }
33940
+ }
33941
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33942
+ tags: { deviceId: id },
33943
+ meta: {
33944
+ stableId,
33945
+ type: meta.type
33946
+ }
33947
+ });
33948
+ return { deviceId: id };
33547
33949
  }
33548
33950
  /**
33549
33951
  * Restore devices from persisted state. Two-pass:
@@ -33569,55 +33971,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33569
33971
  * accessory-spawn flow handles via the parent's
33570
33972
  * `getAccessoryChildren()`. Override only when the default doesn't
33571
33973
  * fit.
33974
+ *
33975
+ * A row that fails either pass is NOT terminal (D347): it is handed
33976
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33977
+ * Only after the bound is exhausted is the device marked permanently
33978
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33979
+ * `getStatus().error`.
33572
33980
  */
33573
33981
  async onRestoreDevices(savedDevices) {
33574
33982
  const restored = /* @__PURE__ */ new Set();
33983
+ const failures = [];
33984
+ const attemptRestore = async (saved) => {
33985
+ if (restored.has(saved.id)) return;
33986
+ const Class = this.deviceClasses[saved.type];
33987
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33988
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33989
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33990
+ restored.add(saved.id);
33991
+ };
33575
33992
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33576
33993
  const restoreOne = async (saved) => {
33577
- const Class = this.deviceClasses[saved.type];
33578
- if (!Class) {
33994
+ if (!this.deviceClasses[saved.type]) {
33579
33995
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33580
- tags: { stableId: saved.stableId },
33996
+ tags: {
33997
+ deviceId: saved.id,
33998
+ stableId: saved.stableId
33999
+ },
33581
34000
  meta: { type: saved.type }
33582
34001
  });
33583
34002
  return;
33584
34003
  }
33585
34004
  try {
33586
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33587
- restored.add(saved.id);
34005
+ await attemptRestore(saved);
33588
34006
  } catch (err) {
33589
- this.ctx.logger.warn("Failed to restore device", {
33590
- tags: { stableId: saved.stableId },
34007
+ const error = err instanceof Error ? err.message : String(err);
34008
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
34009
+ tags: {
34010
+ deviceId: saved.id,
34011
+ stableId: saved.stableId
34012
+ },
33591
34013
  meta: {
33592
34014
  type: saved.type,
33593
- error: err instanceof Error ? err.message : String(err)
34015
+ attempt: 1,
34016
+ error
33594
34017
  }
33595
34018
  });
34019
+ failures.push({
34020
+ saved,
34021
+ error
34022
+ });
33596
34023
  }
33597
34024
  };
33598
34025
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
34026
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33599
34027
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33600
34028
  for (const saved of childRows) {
33601
- const Class = this.deviceClasses[saved.type];
33602
- if (!Class) continue;
34029
+ if (!this.deviceClasses[saved.type]) continue;
33603
34030
  if (saved.parentDeviceId === null) continue;
33604
- if (!restored.has(saved.parentDeviceId)) continue;
33605
- try {
33606
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33607
- restored.add(saved.id);
33608
- } catch (err) {
33609
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
34031
+ if (restored.has(saved.parentDeviceId)) {
34032
+ try {
34033
+ await attemptRestore(saved);
34034
+ } catch (err) {
34035
+ const error = err instanceof Error ? err.message : String(err);
34036
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
34037
+ tags: {
34038
+ deviceId: saved.id,
34039
+ stableId: saved.stableId,
34040
+ parentDeviceId: saved.parentDeviceId
34041
+ },
34042
+ meta: {
34043
+ type: saved.type,
34044
+ attempt: 1,
34045
+ error
34046
+ }
34047
+ });
34048
+ failures.push({
34049
+ saved,
34050
+ error
34051
+ });
34052
+ }
34053
+ continue;
34054
+ }
34055
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
34056
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33610
34057
  tags: {
34058
+ deviceId: saved.id,
33611
34059
  stableId: saved.stableId,
33612
34060
  parentDeviceId: saved.parentDeviceId
33613
34061
  },
33614
- meta: {
33615
- type: saved.type,
33616
- error: err instanceof Error ? err.message : String(err)
33617
- }
34062
+ meta: { type: saved.type }
33618
34063
  });
34064
+ failures.push({
34065
+ saved,
34066
+ error: `parent device ${saved.parentDeviceId} not restored`
34067
+ });
34068
+ continue;
33619
34069
  }
33620
34070
  }
34071
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
34072
+ return {
34073
+ restoredCount: restored.size,
34074
+ failedCount: failures.length
34075
+ };
33621
34076
  }
33622
34077
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33623
34078
  toSummary(device) {
@@ -34126,6 +34581,12 @@ Object.freeze({
34126
34581
  addonId: null,
34127
34582
  access: "create"
34128
34583
  },
34584
+ "backup.cancel": {
34585
+ capName: "backup",
34586
+ capScope: "system",
34587
+ addonId: null,
34588
+ access: "create"
34589
+ },
34129
34590
  "backup.delete": {
34130
34591
  capName: "backup",
34131
34592
  capScope: "system",
@@ -34168,6 +34629,12 @@ Object.freeze({
34168
34629
  addonId: null,
34169
34630
  access: "view"
34170
34631
  },
34632
+ "backup.listRuns": {
34633
+ capName: "backup",
34634
+ capScope: "system",
34635
+ addonId: null,
34636
+ access: "view"
34637
+ },
34171
34638
  "backup.listSchedules": {
34172
34639
  capName: "backup",
34173
34640
  capScope: "system",
@@ -35368,6 +35835,12 @@ Object.freeze({
35368
35835
  addonId: null,
35369
35836
  access: "view"
35370
35837
  },
35838
+ "deviceProvider.reloadDevice": {
35839
+ capName: "device-provider",
35840
+ capScope: "system",
35841
+ addonId: null,
35842
+ access: "create"
35843
+ },
35371
35844
  "deviceProvider.start": {
35372
35845
  capName: "device-provider",
35373
35846
  capScope: "system",
@@ -38734,6 +39207,12 @@ Object.freeze({
38734
39207
  addonId: null,
38735
39208
  access: "create"
38736
39209
  },
39210
+ "streamBroker.forgetDeviceHardware": {
39211
+ capName: "stream-broker",
39212
+ capScope: "system",
39213
+ addonId: null,
39214
+ access: "delete"
39215
+ },
38737
39216
  "streamBroker.getAllRtspEntries": {
38738
39217
  capName: "stream-broker",
38739
39218
  capScope: "system",
@@ -41192,6 +41671,11 @@ Object.freeze({
41192
41671
  form: "single",
41193
41672
  optional: false
41194
41673
  }],
41674
+ "streamBroker.forgetDeviceHardware": [{
41675
+ name: "deviceId",
41676
+ form: "single",
41677
+ optional: false
41678
+ }],
41195
41679
  "streamBroker.getDeviceAudioMute": [{
41196
41680
  name: "deviceId",
41197
41681
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-rademacher",
3
- "version": "0.2.57",
3
+ "version": "0.2.59",
4
4
  "description": "Rademacher HomePilot device-provider addon for CamStack — wraps the @apocaliss92/noderademacher local-hub client (roller shutters over the cover cap)",
5
5
  "keywords": [
6
6
  "camstack",