@camstack/addon-provider-amcrest 0.2.59 → 0.2.61

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10520,6 +10520,89 @@ var LocationStatSchema = object({
10520
10520
  fileCount: number(),
10521
10521
  present: boolean()
10522
10522
  });
10523
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10524
+ var BackupRunStateSchema = _enum([
10525
+ "queued",
10526
+ "running",
10527
+ "succeeded",
10528
+ "failed",
10529
+ "cancelled"
10530
+ ]);
10531
+ /**
10532
+ * Where a running backup currently is. `queued` before it starts,
10533
+ * `building` while the tar.gz is being staged, `uploading` during the
10534
+ * per-destination fan-out, `done` once terminal.
10535
+ */
10536
+ var BackupRunPhaseSchema = _enum([
10537
+ "queued",
10538
+ "building",
10539
+ "uploading",
10540
+ "done"
10541
+ ]);
10542
+ /**
10543
+ * Observable state of one backup run — readable WHILE it runs via
10544
+ * `backup.listRuns`. This is what makes the execution queue and
10545
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10546
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10547
+ * diagnosable with `du` because nothing reported that runs existed or
10548
+ * how large the staged archive had grown.
10549
+ */
10550
+ var BackupRunSchema = object({
10551
+ /** Stable run id — the handle `backup.cancel` takes. */
10552
+ id: string(),
10553
+ state: BackupRunStateSchema,
10554
+ phase: BackupRunPhaseSchema,
10555
+ /**
10556
+ * Resolved destination location ids. Empty while queued (targets are
10557
+ * resolved when the run starts, against the then-current policies).
10558
+ */
10559
+ destinationIds: array(string()).readonly(),
10560
+ label: string().optional(),
10561
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10562
+ requestedAt: number(),
10563
+ /** ms-epoch when the run left the queue and started building. */
10564
+ startedAt: number().optional(),
10565
+ /** ms-epoch when the run reached a terminal state. */
10566
+ finishedAt: number().optional(),
10567
+ /** Compressed bytes of the staging archive written so far. */
10568
+ stagedBytes: number(),
10569
+ /** Final staged archive size, once the build phase completes. */
10570
+ archiveSizeBytes: number().optional(),
10571
+ /** Bytes pushed to the destination currently uploading. */
10572
+ uploadedBytes: number(),
10573
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10574
+ completedDestinationIds: array(string()).readonly(),
10575
+ /** Destinations that failed during the fan-out. */
10576
+ failedDestinationIds: array(string()).readonly(),
10577
+ /** Failure message when `state === 'failed'`. */
10578
+ error: string().optional(),
10579
+ /**
10580
+ * 1-based place in the execution queue — 1 = runs next. Present only
10581
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10582
+ * queue's OWN pending order, never derived from timestamps, so the
10583
+ * UI cannot show an order the executor will not honour.
10584
+ */
10585
+ queuePosition: number().int().min(1).optional()
10586
+ });
10587
+ /**
10588
+ * Result of `backup.trigger`. The call still resolves when the run
10589
+ * terminates (compat with schedule-driven runs and the admin UI), but
10590
+ * it now names the run and says whether it had to WAIT: a trigger that
10591
+ * arrives while another run is in flight is enqueued (or joined onto
10592
+ * an identical already-queued run), never started concurrently.
10593
+ */
10594
+ var BackupTriggerResultSchema = object({
10595
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10596
+ runId: string(),
10597
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10598
+ queued: boolean(),
10599
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10600
+ joined: boolean(),
10601
+ /** True when the run was cancelled before completing every destination. */
10602
+ cancelled: boolean(),
10603
+ /** One entry per destination the archive landed at (partial on cancel). */
10604
+ entries: array(BackupEntrySchema).readonly()
10605
+ });
10523
10606
  /**
10524
10607
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10525
10608
  * SET of destination locations. Supersedes the per-location cron on
@@ -10567,7 +10650,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10567
10650
  * retention (manual runs).
10568
10651
  */
10569
10652
  retentionCount: number().int().min(1).max(1e3).optional()
10570
- }).optional(), array(BackupEntrySchema).readonly(), {
10653
+ }).optional(), BackupTriggerResultSchema, {
10654
+ kind: "mutation",
10655
+ auth: "admin"
10656
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10571
10657
  kind: "mutation",
10572
10658
  auth: "admin"
10573
10659
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11284,6 +11370,14 @@ method(object({
11284
11370
  }), object({ success: literal(true) }), {
11285
11371
  kind: "mutation",
11286
11372
  auth: "admin"
11373
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11374
+ derivedStreamsDeleted: array(string()).readonly(),
11375
+ assignmentsPurged: boolean(),
11376
+ probeSnapshotsDropped: number().int().nonnegative(),
11377
+ rtspTokenRowsDeleted: number().int().nonnegative()
11378
+ }), {
11379
+ kind: "mutation",
11380
+ auth: "admin"
11287
11381
  }), method(object({
11288
11382
  deviceId: number(),
11289
11383
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12711,6 +12805,35 @@ var deviceProviderCapability = {
12711
12805
  name: string(),
12712
12806
  type: string()
12713
12807
  }))),
12808
+ /**
12809
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12810
+ * touching no other device this provider owns.
12811
+ *
12812
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12813
+ * migrated numbers: after `swapIds` the runner's live instance still
12814
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12815
+ * registrations and its log tags), and a live object cannot be renumbered.
12816
+ * Before this method the only flush was restarting the whole owning addon
12817
+ * — which took every camera the provider owns down with it (28 devices
12818
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12819
+ * same day ~27 devices' native caps did not come back on their own).
12820
+ *
12821
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12822
+ * that changes. The reply carries the id the device answers on NOW.
12823
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12824
+ * instance (if any), then re-create from the persisted row: the same
12825
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12826
+ * An RPC, never an event: a dropped event would leave the runner writing
12827
+ * against the wrong camera (D8).
12828
+ *
12829
+ * Construction can dial hardware, and the migrated source is
12830
+ * characteristically dead — the timeout covers a full activate window
12831
+ * rather than the 60 s default.
12832
+ */
12833
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12834
+ kind: "mutation",
12835
+ timeoutMs: 3 * 6e4
12836
+ }),
12714
12837
  supportsDiscovery: method(object({}), boolean()),
12715
12838
  /**
12716
12839
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13038,7 +13161,8 @@ method(object({
13038
13161
  targetId: number()
13039
13162
  }), MigrateDeviceResultSchema, {
13040
13163
  kind: "mutation",
13041
- auth: "admin"
13164
+ auth: "admin",
13165
+ timeoutMs: 12 * 6e4
13042
13166
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13043
13167
  deviceId: number(),
13044
13168
  name: string()
@@ -32784,6 +32908,147 @@ var BaseDevice = class {
32784
32908
  }
32785
32909
  };
32786
32910
  /**
32911
+ * Delays before retry rounds 1..N — the round count IS the bound.
32912
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32913
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32914
+ * per attempt) covers a device-manager lock held for minutes — the
32915
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32916
+ */
32917
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32918
+ 1e4,
32919
+ 3e4,
32920
+ 9e4
32921
+ ];
32922
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32923
+ function sleep$1(ms, signal) {
32924
+ return new Promise((resolve) => {
32925
+ if (signal.aborted) {
32926
+ resolve();
32927
+ return;
32928
+ }
32929
+ const onAbort = () => {
32930
+ clearTimeout(timer);
32931
+ resolve();
32932
+ };
32933
+ const timer = setTimeout(() => {
32934
+ signal.removeEventListener("abort", onAbort);
32935
+ resolve();
32936
+ }, ms);
32937
+ timer.unref?.();
32938
+ signal.addEventListener("abort", onAbort, { once: true });
32939
+ });
32940
+ }
32941
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32942
+ * not reject (callers wrap their own try/catch). */
32943
+ async function runWithConcurrency(items, width, fn) {
32944
+ const queue = [...items];
32945
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32946
+ const lane = async () => {
32947
+ for (;;) {
32948
+ const item = queue.shift();
32949
+ if (item === void 0) return;
32950
+ await fn(item);
32951
+ }
32952
+ };
32953
+ await Promise.all(Array.from({ length: laneCount }, lane));
32954
+ }
32955
+ var DeviceRestoreRetryScheduler = class {
32956
+ #logger;
32957
+ #attempt;
32958
+ #onPermanentFailure;
32959
+ #delaysMs;
32960
+ #concurrency;
32961
+ #now;
32962
+ #abort = new AbortController();
32963
+ constructor(options) {
32964
+ this.#logger = options.logger;
32965
+ this.#attempt = options.attempt;
32966
+ this.#onPermanentFailure = options.onPermanentFailure;
32967
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32968
+ this.#concurrency = options.concurrency ?? 4;
32969
+ this.#now = options.now ?? Date.now;
32970
+ }
32971
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32972
+ * permanently failed — the next boot restores them from disk. */
32973
+ cancel() {
32974
+ this.#abort.abort();
32975
+ }
32976
+ /**
32977
+ * Run the bounded retry rounds. Resolves when every entry has either
32978
+ * restored, been marked permanently failed, or the scheduler was
32979
+ * cancelled. Never rejects.
32980
+ */
32981
+ async run(initialFailures) {
32982
+ let pending = initialFailures.map((failure) => ({
32983
+ saved: failure.saved,
32984
+ lastError: failure.error,
32985
+ attempts: 1
32986
+ }));
32987
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32988
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32989
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32990
+ if (this.#abort.signal.aborted) break;
32991
+ pending = await this.#runRound(pending, round);
32992
+ }
32993
+ if (this.#abort.signal.aborted) return [];
32994
+ const terminal = pending.map((entry) => ({
32995
+ deviceId: entry.saved.id,
32996
+ stableId: entry.saved.stableId,
32997
+ type: String(entry.saved.type),
32998
+ attempts: entry.attempts,
32999
+ lastError: entry.lastError,
33000
+ failedAt: this.#now()
33001
+ }));
33002
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33003
+ return terminal;
33004
+ }
33005
+ /** One retry round: parents first (phase 0), then hub-adopted
33006
+ * children (phase 1) — a child's attempt depends on its parent
33007
+ * having landed, exactly like the initial two-pass restore. */
33008
+ async #runRound(pending, round) {
33009
+ const next = [];
33010
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33011
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33012
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33013
+ if (this.#abort.signal.aborted) {
33014
+ next.push(entry);
33015
+ return;
33016
+ }
33017
+ const attemptNo = entry.attempts + 1;
33018
+ try {
33019
+ await this.#attempt(entry.saved);
33020
+ this.#logger.info("Device restored on retry", {
33021
+ tags: {
33022
+ deviceId: entry.saved.id,
33023
+ stableId: entry.saved.stableId
33024
+ },
33025
+ meta: { attempt: attemptNo }
33026
+ });
33027
+ } catch (err) {
33028
+ const lastError = err instanceof Error ? err.message : String(err);
33029
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33030
+ this.#logger.warn("Device restore retry failed", {
33031
+ tags: {
33032
+ deviceId: entry.saved.id,
33033
+ stableId: entry.saved.stableId
33034
+ },
33035
+ meta: {
33036
+ attempt: attemptNo,
33037
+ remainingRetries,
33038
+ error: lastError
33039
+ }
33040
+ });
33041
+ next.push({
33042
+ saved: entry.saved,
33043
+ lastError,
33044
+ attempts: attemptNo
33045
+ });
33046
+ }
33047
+ });
33048
+ return next;
33049
+ }
33050
+ };
33051
+ /**
32787
33052
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32788
33053
  * device-provider cap router. Shared across all providers.
32789
33054
  */
@@ -32832,6 +33097,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32832
33097
  }];
32833
33098
  }
32834
33099
  async onShutdown() {
33100
+ this.cancelRestoreRetries();
32835
33101
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32836
33102
  for (const device of devices) try {
32837
33103
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32849,9 +33115,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32849
33115
  async start() {}
32850
33116
  async stop() {}
32851
33117
  async getStatus() {
33118
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33119
+ const summary = this.restoreFailureSummary();
33120
+ if (summary === null) return {
33121
+ connected: true,
33122
+ deviceCount: all.length
33123
+ };
32852
33124
  return {
32853
33125
  connected: true,
32854
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33126
+ deviceCount: all.length,
33127
+ error: summary
32855
33128
  };
32856
33129
  }
32857
33130
  async getDevices() {
@@ -32941,8 +33214,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32941
33214
  };
32942
33215
  }
32943
33216
  async restoreDevices(savedDevices) {
32944
- await this.onRestoreDevices(savedDevices);
32945
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33217
+ const report = await this.onRestoreDevices(savedDevices);
33218
+ if (savedDevices.length === 0) return;
33219
+ if (report && report.failedCount > 0) {
33220
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33221
+ return;
33222
+ }
33223
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33224
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33225
+ }
33226
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33227
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33228
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33229
+ * never re-stampede full-width while the initial pass does (D167). */
33230
+ restoreRetryConcurrency = 4;
33231
+ _restoreRetryScheduler = null;
33232
+ _restoreRetryCompletion = null;
33233
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33234
+ /** Settles when the background retry rounds finish (or `null` when
33235
+ * nothing failed). Exposed for tests and subclass diagnostics —
33236
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33237
+ * with the devices that restored, and a late success is announced
33238
+ * through the `native-cap-change` → `updateCaps` path. */
33239
+ get restoreRetryCompletion() {
33240
+ return this._restoreRetryCompletion;
33241
+ }
33242
+ /** Devices that exhausted the retry bound this process lifetime. */
33243
+ get permanentRestoreFailures() {
33244
+ return [...this._permanentRestoreFailures.values()];
33245
+ }
33246
+ /** One-line operator-facing summary for `getStatus().error`, or
33247
+ * `null` when every device restored. */
33248
+ restoreFailureSummary() {
33249
+ if (this._permanentRestoreFailures.size === 0) return null;
33250
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33251
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33252
+ }
33253
+ cancelRestoreRetries() {
33254
+ this._restoreRetryScheduler?.cancel();
33255
+ this._restoreRetryScheduler = null;
33256
+ }
33257
+ recordPermanentRestoreFailure(failure) {
33258
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33259
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33260
+ tags: {
33261
+ deviceId: failure.deviceId,
33262
+ stableId: failure.stableId
33263
+ },
33264
+ meta: {
33265
+ type: failure.type,
33266
+ attempts: failure.attempts,
33267
+ error: failure.lastError
33268
+ }
33269
+ });
33270
+ }
33271
+ scheduleRestoreRetries(failures, attempt) {
33272
+ const scheduler = new DeviceRestoreRetryScheduler({
33273
+ logger: this.ctx.logger,
33274
+ delaysMs: this.restoreRetryDelaysMs,
33275
+ concurrency: this.restoreRetryConcurrency,
33276
+ attempt,
33277
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33278
+ });
33279
+ this._restoreRetryScheduler = scheduler;
33280
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33281
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33282
+ });
33283
+ }
33284
+ /**
33285
+ * Tear down and reconstruct ONE device from its persisted rows — the
33286
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33287
+ * and no other device this provider owns is disturbed.
33288
+ *
33289
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33290
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33291
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33292
+ * whatever number the row carries NOW. The teardown is `decommission` —
33293
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33294
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33295
+ * the boot restore's own `create()` path, including its pass 2: first-class
33296
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33297
+ * parent by the cascade and must be re-created explicitly, because only
33298
+ * accessory children come back through `getAccessoryChildren()`.
33299
+ *
33300
+ * Reloading an accessory child directly is refused (no device class) —
33301
+ * reload its parent instead.
33302
+ */
33303
+ async reloadDevice(input) {
33304
+ const { stableId } = input;
33305
+ const devices = this.ctx.kernel.devices;
33306
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33307
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33308
+ if (live) await devices.decommission(live.id);
33309
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33310
+ addonId: this.addonId,
33311
+ stableId
33312
+ });
33313
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33314
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33315
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33316
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33317
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33318
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33319
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33320
+ for (const row of rows) {
33321
+ if (row.parentDeviceId !== id) continue;
33322
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33323
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33324
+ if (!ChildClass) continue;
33325
+ try {
33326
+ await devices.create(row.stableId, ChildClass, {}, id);
33327
+ } catch (err) {
33328
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33329
+ tags: {
33330
+ deviceId: row.id,
33331
+ stableId: row.stableId
33332
+ },
33333
+ meta: {
33334
+ parentDeviceId: id,
33335
+ error: err instanceof Error ? err.message : String(err)
33336
+ }
33337
+ });
33338
+ }
33339
+ }
33340
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33341
+ tags: { deviceId: id },
33342
+ meta: {
33343
+ stableId,
33344
+ type: meta.type
33345
+ }
33346
+ });
33347
+ return { deviceId: id };
32946
33348
  }
32947
33349
  /**
32948
33350
  * Restore devices from persisted state. Two-pass:
@@ -32968,55 +33370,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32968
33370
  * accessory-spawn flow handles via the parent's
32969
33371
  * `getAccessoryChildren()`. Override only when the default doesn't
32970
33372
  * fit.
33373
+ *
33374
+ * A row that fails either pass is NOT terminal (D347): it is handed
33375
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33376
+ * Only after the bound is exhausted is the device marked permanently
33377
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33378
+ * `getStatus().error`.
32971
33379
  */
32972
33380
  async onRestoreDevices(savedDevices) {
32973
33381
  const restored = /* @__PURE__ */ new Set();
33382
+ const failures = [];
33383
+ const attemptRestore = async (saved) => {
33384
+ if (restored.has(saved.id)) return;
33385
+ const Class = this.deviceClasses[saved.type];
33386
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33387
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33388
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33389
+ restored.add(saved.id);
33390
+ };
32974
33391
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32975
33392
  const restoreOne = async (saved) => {
32976
- const Class = this.deviceClasses[saved.type];
32977
- if (!Class) {
33393
+ if (!this.deviceClasses[saved.type]) {
32978
33394
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32979
- tags: { stableId: saved.stableId },
33395
+ tags: {
33396
+ deviceId: saved.id,
33397
+ stableId: saved.stableId
33398
+ },
32980
33399
  meta: { type: saved.type }
32981
33400
  });
32982
33401
  return;
32983
33402
  }
32984
33403
  try {
32985
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32986
- restored.add(saved.id);
33404
+ await attemptRestore(saved);
32987
33405
  } catch (err) {
32988
- this.ctx.logger.warn("Failed to restore device", {
32989
- tags: { stableId: saved.stableId },
33406
+ const error = err instanceof Error ? err.message : String(err);
33407
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33408
+ tags: {
33409
+ deviceId: saved.id,
33410
+ stableId: saved.stableId
33411
+ },
32990
33412
  meta: {
32991
33413
  type: saved.type,
32992
- error: err instanceof Error ? err.message : String(err)
33414
+ attempt: 1,
33415
+ error
32993
33416
  }
32994
33417
  });
33418
+ failures.push({
33419
+ saved,
33420
+ error
33421
+ });
32995
33422
  }
32996
33423
  };
32997
33424
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33425
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32998
33426
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32999
33427
  for (const saved of childRows) {
33000
- const Class = this.deviceClasses[saved.type];
33001
- if (!Class) continue;
33428
+ if (!this.deviceClasses[saved.type]) continue;
33002
33429
  if (saved.parentDeviceId === null) continue;
33003
- if (!restored.has(saved.parentDeviceId)) continue;
33004
- try {
33005
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33006
- restored.add(saved.id);
33007
- } catch (err) {
33008
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33430
+ if (restored.has(saved.parentDeviceId)) {
33431
+ try {
33432
+ await attemptRestore(saved);
33433
+ } catch (err) {
33434
+ const error = err instanceof Error ? err.message : String(err);
33435
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33436
+ tags: {
33437
+ deviceId: saved.id,
33438
+ stableId: saved.stableId,
33439
+ parentDeviceId: saved.parentDeviceId
33440
+ },
33441
+ meta: {
33442
+ type: saved.type,
33443
+ attempt: 1,
33444
+ error
33445
+ }
33446
+ });
33447
+ failures.push({
33448
+ saved,
33449
+ error
33450
+ });
33451
+ }
33452
+ continue;
33453
+ }
33454
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33455
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33009
33456
  tags: {
33457
+ deviceId: saved.id,
33010
33458
  stableId: saved.stableId,
33011
33459
  parentDeviceId: saved.parentDeviceId
33012
33460
  },
33013
- meta: {
33014
- type: saved.type,
33015
- error: err instanceof Error ? err.message : String(err)
33016
- }
33461
+ meta: { type: saved.type }
33462
+ });
33463
+ failures.push({
33464
+ saved,
33465
+ error: `parent device ${saved.parentDeviceId} not restored`
33017
33466
  });
33467
+ continue;
33018
33468
  }
33019
33469
  }
33470
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33471
+ return {
33472
+ restoredCount: restored.size,
33473
+ failedCount: failures.length
33474
+ };
33020
33475
  }
33021
33476
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33022
33477
  toSummary(device) {
@@ -33731,6 +34186,12 @@ Object.freeze({
33731
34186
  addonId: null,
33732
34187
  access: "create"
33733
34188
  },
34189
+ "backup.cancel": {
34190
+ capName: "backup",
34191
+ capScope: "system",
34192
+ addonId: null,
34193
+ access: "create"
34194
+ },
33734
34195
  "backup.delete": {
33735
34196
  capName: "backup",
33736
34197
  capScope: "system",
@@ -33773,6 +34234,12 @@ Object.freeze({
33773
34234
  addonId: null,
33774
34235
  access: "view"
33775
34236
  },
34237
+ "backup.listRuns": {
34238
+ capName: "backup",
34239
+ capScope: "system",
34240
+ addonId: null,
34241
+ access: "view"
34242
+ },
33776
34243
  "backup.listSchedules": {
33777
34244
  capName: "backup",
33778
34245
  capScope: "system",
@@ -34973,6 +35440,12 @@ Object.freeze({
34973
35440
  addonId: null,
34974
35441
  access: "view"
34975
35442
  },
35443
+ "deviceProvider.reloadDevice": {
35444
+ capName: "device-provider",
35445
+ capScope: "system",
35446
+ addonId: null,
35447
+ access: "create"
35448
+ },
34976
35449
  "deviceProvider.start": {
34977
35450
  capName: "device-provider",
34978
35451
  capScope: "system",
@@ -38339,6 +38812,12 @@ Object.freeze({
38339
38812
  addonId: null,
38340
38813
  access: "create"
38341
38814
  },
38815
+ "streamBroker.forgetDeviceHardware": {
38816
+ capName: "stream-broker",
38817
+ capScope: "system",
38818
+ addonId: null,
38819
+ access: "delete"
38820
+ },
38342
38821
  "streamBroker.getAllRtspEntries": {
38343
38822
  capName: "stream-broker",
38344
38823
  capScope: "system",
@@ -40797,6 +41276,11 @@ Object.freeze({
40797
41276
  form: "single",
40798
41277
  optional: false
40799
41278
  }],
41279
+ "streamBroker.forgetDeviceHardware": [{
41280
+ name: "deviceId",
41281
+ form: "single",
41282
+ optional: false
41283
+ }],
40800
41284
  "streamBroker.getDeviceAudioMute": [{
40801
41285
  name: "deviceId",
40802
41286
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10521,6 +10521,89 @@ var LocationStatSchema = object({
10521
10521
  fileCount: number(),
10522
10522
  present: boolean()
10523
10523
  });
10524
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10525
+ var BackupRunStateSchema = _enum([
10526
+ "queued",
10527
+ "running",
10528
+ "succeeded",
10529
+ "failed",
10530
+ "cancelled"
10531
+ ]);
10532
+ /**
10533
+ * Where a running backup currently is. `queued` before it starts,
10534
+ * `building` while the tar.gz is being staged, `uploading` during the
10535
+ * per-destination fan-out, `done` once terminal.
10536
+ */
10537
+ var BackupRunPhaseSchema = _enum([
10538
+ "queued",
10539
+ "building",
10540
+ "uploading",
10541
+ "done"
10542
+ ]);
10543
+ /**
10544
+ * Observable state of one backup run — readable WHILE it runs via
10545
+ * `backup.listRuns`. This is what makes the execution queue and
10546
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10547
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10548
+ * diagnosable with `du` because nothing reported that runs existed or
10549
+ * how large the staged archive had grown.
10550
+ */
10551
+ var BackupRunSchema = object({
10552
+ /** Stable run id — the handle `backup.cancel` takes. */
10553
+ id: string(),
10554
+ state: BackupRunStateSchema,
10555
+ phase: BackupRunPhaseSchema,
10556
+ /**
10557
+ * Resolved destination location ids. Empty while queued (targets are
10558
+ * resolved when the run starts, against the then-current policies).
10559
+ */
10560
+ destinationIds: array(string()).readonly(),
10561
+ label: string().optional(),
10562
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10563
+ requestedAt: number(),
10564
+ /** ms-epoch when the run left the queue and started building. */
10565
+ startedAt: number().optional(),
10566
+ /** ms-epoch when the run reached a terminal state. */
10567
+ finishedAt: number().optional(),
10568
+ /** Compressed bytes of the staging archive written so far. */
10569
+ stagedBytes: number(),
10570
+ /** Final staged archive size, once the build phase completes. */
10571
+ archiveSizeBytes: number().optional(),
10572
+ /** Bytes pushed to the destination currently uploading. */
10573
+ uploadedBytes: number(),
10574
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10575
+ completedDestinationIds: array(string()).readonly(),
10576
+ /** Destinations that failed during the fan-out. */
10577
+ failedDestinationIds: array(string()).readonly(),
10578
+ /** Failure message when `state === 'failed'`. */
10579
+ error: string().optional(),
10580
+ /**
10581
+ * 1-based place in the execution queue — 1 = runs next. Present only
10582
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10583
+ * queue's OWN pending order, never derived from timestamps, so the
10584
+ * UI cannot show an order the executor will not honour.
10585
+ */
10586
+ queuePosition: number().int().min(1).optional()
10587
+ });
10588
+ /**
10589
+ * Result of `backup.trigger`. The call still resolves when the run
10590
+ * terminates (compat with schedule-driven runs and the admin UI), but
10591
+ * it now names the run and says whether it had to WAIT: a trigger that
10592
+ * arrives while another run is in flight is enqueued (or joined onto
10593
+ * an identical already-queued run), never started concurrently.
10594
+ */
10595
+ var BackupTriggerResultSchema = object({
10596
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10597
+ runId: string(),
10598
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10599
+ queued: boolean(),
10600
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10601
+ joined: boolean(),
10602
+ /** True when the run was cancelled before completing every destination. */
10603
+ cancelled: boolean(),
10604
+ /** One entry per destination the archive landed at (partial on cancel). */
10605
+ entries: array(BackupEntrySchema).readonly()
10606
+ });
10524
10607
  /**
10525
10608
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10526
10609
  * SET of destination locations. Supersedes the per-location cron on
@@ -10568,7 +10651,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10568
10651
  * retention (manual runs).
10569
10652
  */
10570
10653
  retentionCount: number().int().min(1).max(1e3).optional()
10571
- }).optional(), array(BackupEntrySchema).readonly(), {
10654
+ }).optional(), BackupTriggerResultSchema, {
10655
+ kind: "mutation",
10656
+ auth: "admin"
10657
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10572
10658
  kind: "mutation",
10573
10659
  auth: "admin"
10574
10660
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11285,6 +11371,14 @@ method(object({
11285
11371
  }), object({ success: literal(true) }), {
11286
11372
  kind: "mutation",
11287
11373
  auth: "admin"
11374
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11375
+ derivedStreamsDeleted: array(string()).readonly(),
11376
+ assignmentsPurged: boolean(),
11377
+ probeSnapshotsDropped: number().int().nonnegative(),
11378
+ rtspTokenRowsDeleted: number().int().nonnegative()
11379
+ }), {
11380
+ kind: "mutation",
11381
+ auth: "admin"
11288
11382
  }), method(object({
11289
11383
  deviceId: number(),
11290
11384
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12712,6 +12806,35 @@ var deviceProviderCapability = {
12712
12806
  name: string(),
12713
12807
  type: string()
12714
12808
  }))),
12809
+ /**
12810
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12811
+ * touching no other device this provider owns.
12812
+ *
12813
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12814
+ * migrated numbers: after `swapIds` the runner's live instance still
12815
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12816
+ * registrations and its log tags), and a live object cannot be renumbered.
12817
+ * Before this method the only flush was restarting the whole owning addon
12818
+ * — which took every camera the provider owns down with it (28 devices
12819
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12820
+ * same day ~27 devices' native caps did not come back on their own).
12821
+ *
12822
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12823
+ * that changes. The reply carries the id the device answers on NOW.
12824
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12825
+ * instance (if any), then re-create from the persisted row: the same
12826
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12827
+ * An RPC, never an event: a dropped event would leave the runner writing
12828
+ * against the wrong camera (D8).
12829
+ *
12830
+ * Construction can dial hardware, and the migrated source is
12831
+ * characteristically dead — the timeout covers a full activate window
12832
+ * rather than the 60 s default.
12833
+ */
12834
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12835
+ kind: "mutation",
12836
+ timeoutMs: 3 * 6e4
12837
+ }),
12715
12838
  supportsDiscovery: method(object({}), boolean()),
12716
12839
  /**
12717
12840
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13039,7 +13162,8 @@ method(object({
13039
13162
  targetId: number()
13040
13163
  }), MigrateDeviceResultSchema, {
13041
13164
  kind: "mutation",
13042
- auth: "admin"
13165
+ auth: "admin",
13166
+ timeoutMs: 12 * 6e4
13043
13167
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13044
13168
  deviceId: number(),
13045
13169
  name: string()
@@ -32785,6 +32909,147 @@ var BaseDevice = class {
32785
32909
  }
32786
32910
  };
32787
32911
  /**
32912
+ * Delays before retry rounds 1..N — the round count IS the bound.
32913
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32914
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32915
+ * per attempt) covers a device-manager lock held for minutes — the
32916
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32917
+ */
32918
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32919
+ 1e4,
32920
+ 3e4,
32921
+ 9e4
32922
+ ];
32923
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32924
+ function sleep$1(ms, signal) {
32925
+ return new Promise((resolve) => {
32926
+ if (signal.aborted) {
32927
+ resolve();
32928
+ return;
32929
+ }
32930
+ const onAbort = () => {
32931
+ clearTimeout(timer);
32932
+ resolve();
32933
+ };
32934
+ const timer = setTimeout(() => {
32935
+ signal.removeEventListener("abort", onAbort);
32936
+ resolve();
32937
+ }, ms);
32938
+ timer.unref?.();
32939
+ signal.addEventListener("abort", onAbort, { once: true });
32940
+ });
32941
+ }
32942
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32943
+ * not reject (callers wrap their own try/catch). */
32944
+ async function runWithConcurrency(items, width, fn) {
32945
+ const queue = [...items];
32946
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32947
+ const lane = async () => {
32948
+ for (;;) {
32949
+ const item = queue.shift();
32950
+ if (item === void 0) return;
32951
+ await fn(item);
32952
+ }
32953
+ };
32954
+ await Promise.all(Array.from({ length: laneCount }, lane));
32955
+ }
32956
+ var DeviceRestoreRetryScheduler = class {
32957
+ #logger;
32958
+ #attempt;
32959
+ #onPermanentFailure;
32960
+ #delaysMs;
32961
+ #concurrency;
32962
+ #now;
32963
+ #abort = new AbortController();
32964
+ constructor(options) {
32965
+ this.#logger = options.logger;
32966
+ this.#attempt = options.attempt;
32967
+ this.#onPermanentFailure = options.onPermanentFailure;
32968
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32969
+ this.#concurrency = options.concurrency ?? 4;
32970
+ this.#now = options.now ?? Date.now;
32971
+ }
32972
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32973
+ * permanently failed — the next boot restores them from disk. */
32974
+ cancel() {
32975
+ this.#abort.abort();
32976
+ }
32977
+ /**
32978
+ * Run the bounded retry rounds. Resolves when every entry has either
32979
+ * restored, been marked permanently failed, or the scheduler was
32980
+ * cancelled. Never rejects.
32981
+ */
32982
+ async run(initialFailures) {
32983
+ let pending = initialFailures.map((failure) => ({
32984
+ saved: failure.saved,
32985
+ lastError: failure.error,
32986
+ attempts: 1
32987
+ }));
32988
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32989
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32990
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32991
+ if (this.#abort.signal.aborted) break;
32992
+ pending = await this.#runRound(pending, round);
32993
+ }
32994
+ if (this.#abort.signal.aborted) return [];
32995
+ const terminal = pending.map((entry) => ({
32996
+ deviceId: entry.saved.id,
32997
+ stableId: entry.saved.stableId,
32998
+ type: String(entry.saved.type),
32999
+ attempts: entry.attempts,
33000
+ lastError: entry.lastError,
33001
+ failedAt: this.#now()
33002
+ }));
33003
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33004
+ return terminal;
33005
+ }
33006
+ /** One retry round: parents first (phase 0), then hub-adopted
33007
+ * children (phase 1) — a child's attempt depends on its parent
33008
+ * having landed, exactly like the initial two-pass restore. */
33009
+ async #runRound(pending, round) {
33010
+ const next = [];
33011
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33012
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33013
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33014
+ if (this.#abort.signal.aborted) {
33015
+ next.push(entry);
33016
+ return;
33017
+ }
33018
+ const attemptNo = entry.attempts + 1;
33019
+ try {
33020
+ await this.#attempt(entry.saved);
33021
+ this.#logger.info("Device restored on retry", {
33022
+ tags: {
33023
+ deviceId: entry.saved.id,
33024
+ stableId: entry.saved.stableId
33025
+ },
33026
+ meta: { attempt: attemptNo }
33027
+ });
33028
+ } catch (err) {
33029
+ const lastError = err instanceof Error ? err.message : String(err);
33030
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33031
+ this.#logger.warn("Device restore retry failed", {
33032
+ tags: {
33033
+ deviceId: entry.saved.id,
33034
+ stableId: entry.saved.stableId
33035
+ },
33036
+ meta: {
33037
+ attempt: attemptNo,
33038
+ remainingRetries,
33039
+ error: lastError
33040
+ }
33041
+ });
33042
+ next.push({
33043
+ saved: entry.saved,
33044
+ lastError,
33045
+ attempts: attemptNo
33046
+ });
33047
+ }
33048
+ });
33049
+ return next;
33050
+ }
33051
+ };
33052
+ /**
32788
33053
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32789
33054
  * device-provider cap router. Shared across all providers.
32790
33055
  */
@@ -32833,6 +33098,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32833
33098
  }];
32834
33099
  }
32835
33100
  async onShutdown() {
33101
+ this.cancelRestoreRetries();
32836
33102
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32837
33103
  for (const device of devices) try {
32838
33104
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32850,9 +33116,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32850
33116
  async start() {}
32851
33117
  async stop() {}
32852
33118
  async getStatus() {
33119
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33120
+ const summary = this.restoreFailureSummary();
33121
+ if (summary === null) return {
33122
+ connected: true,
33123
+ deviceCount: all.length
33124
+ };
32853
33125
  return {
32854
33126
  connected: true,
32855
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33127
+ deviceCount: all.length,
33128
+ error: summary
32856
33129
  };
32857
33130
  }
32858
33131
  async getDevices() {
@@ -32942,8 +33215,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32942
33215
  };
32943
33216
  }
32944
33217
  async restoreDevices(savedDevices) {
32945
- await this.onRestoreDevices(savedDevices);
32946
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33218
+ const report = await this.onRestoreDevices(savedDevices);
33219
+ if (savedDevices.length === 0) return;
33220
+ if (report && report.failedCount > 0) {
33221
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33222
+ return;
33223
+ }
33224
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33225
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33226
+ }
33227
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33228
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33229
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33230
+ * never re-stampede full-width while the initial pass does (D167). */
33231
+ restoreRetryConcurrency = 4;
33232
+ _restoreRetryScheduler = null;
33233
+ _restoreRetryCompletion = null;
33234
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33235
+ /** Settles when the background retry rounds finish (or `null` when
33236
+ * nothing failed). Exposed for tests and subclass diagnostics —
33237
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33238
+ * with the devices that restored, and a late success is announced
33239
+ * through the `native-cap-change` → `updateCaps` path. */
33240
+ get restoreRetryCompletion() {
33241
+ return this._restoreRetryCompletion;
33242
+ }
33243
+ /** Devices that exhausted the retry bound this process lifetime. */
33244
+ get permanentRestoreFailures() {
33245
+ return [...this._permanentRestoreFailures.values()];
33246
+ }
33247
+ /** One-line operator-facing summary for `getStatus().error`, or
33248
+ * `null` when every device restored. */
33249
+ restoreFailureSummary() {
33250
+ if (this._permanentRestoreFailures.size === 0) return null;
33251
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33252
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33253
+ }
33254
+ cancelRestoreRetries() {
33255
+ this._restoreRetryScheduler?.cancel();
33256
+ this._restoreRetryScheduler = null;
33257
+ }
33258
+ recordPermanentRestoreFailure(failure) {
33259
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33260
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33261
+ tags: {
33262
+ deviceId: failure.deviceId,
33263
+ stableId: failure.stableId
33264
+ },
33265
+ meta: {
33266
+ type: failure.type,
33267
+ attempts: failure.attempts,
33268
+ error: failure.lastError
33269
+ }
33270
+ });
33271
+ }
33272
+ scheduleRestoreRetries(failures, attempt) {
33273
+ const scheduler = new DeviceRestoreRetryScheduler({
33274
+ logger: this.ctx.logger,
33275
+ delaysMs: this.restoreRetryDelaysMs,
33276
+ concurrency: this.restoreRetryConcurrency,
33277
+ attempt,
33278
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33279
+ });
33280
+ this._restoreRetryScheduler = scheduler;
33281
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33282
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33283
+ });
33284
+ }
33285
+ /**
33286
+ * Tear down and reconstruct ONE device from its persisted rows — the
33287
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33288
+ * and no other device this provider owns is disturbed.
33289
+ *
33290
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33291
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33292
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33293
+ * whatever number the row carries NOW. The teardown is `decommission` —
33294
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33295
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33296
+ * the boot restore's own `create()` path, including its pass 2: first-class
33297
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33298
+ * parent by the cascade and must be re-created explicitly, because only
33299
+ * accessory children come back through `getAccessoryChildren()`.
33300
+ *
33301
+ * Reloading an accessory child directly is refused (no device class) —
33302
+ * reload its parent instead.
33303
+ */
33304
+ async reloadDevice(input) {
33305
+ const { stableId } = input;
33306
+ const devices = this.ctx.kernel.devices;
33307
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33308
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33309
+ if (live) await devices.decommission(live.id);
33310
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33311
+ addonId: this.addonId,
33312
+ stableId
33313
+ });
33314
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33315
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33316
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33317
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33318
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33319
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33320
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33321
+ for (const row of rows) {
33322
+ if (row.parentDeviceId !== id) continue;
33323
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33324
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33325
+ if (!ChildClass) continue;
33326
+ try {
33327
+ await devices.create(row.stableId, ChildClass, {}, id);
33328
+ } catch (err) {
33329
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33330
+ tags: {
33331
+ deviceId: row.id,
33332
+ stableId: row.stableId
33333
+ },
33334
+ meta: {
33335
+ parentDeviceId: id,
33336
+ error: err instanceof Error ? err.message : String(err)
33337
+ }
33338
+ });
33339
+ }
33340
+ }
33341
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33342
+ tags: { deviceId: id },
33343
+ meta: {
33344
+ stableId,
33345
+ type: meta.type
33346
+ }
33347
+ });
33348
+ return { deviceId: id };
32947
33349
  }
32948
33350
  /**
32949
33351
  * Restore devices from persisted state. Two-pass:
@@ -32969,55 +33371,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32969
33371
  * accessory-spawn flow handles via the parent's
32970
33372
  * `getAccessoryChildren()`. Override only when the default doesn't
32971
33373
  * fit.
33374
+ *
33375
+ * A row that fails either pass is NOT terminal (D347): it is handed
33376
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33377
+ * Only after the bound is exhausted is the device marked permanently
33378
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33379
+ * `getStatus().error`.
32972
33380
  */
32973
33381
  async onRestoreDevices(savedDevices) {
32974
33382
  const restored = /* @__PURE__ */ new Set();
33383
+ const failures = [];
33384
+ const attemptRestore = async (saved) => {
33385
+ if (restored.has(saved.id)) return;
33386
+ const Class = this.deviceClasses[saved.type];
33387
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33388
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33389
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33390
+ restored.add(saved.id);
33391
+ };
32975
33392
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32976
33393
  const restoreOne = async (saved) => {
32977
- const Class = this.deviceClasses[saved.type];
32978
- if (!Class) {
33394
+ if (!this.deviceClasses[saved.type]) {
32979
33395
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32980
- tags: { stableId: saved.stableId },
33396
+ tags: {
33397
+ deviceId: saved.id,
33398
+ stableId: saved.stableId
33399
+ },
32981
33400
  meta: { type: saved.type }
32982
33401
  });
32983
33402
  return;
32984
33403
  }
32985
33404
  try {
32986
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32987
- restored.add(saved.id);
33405
+ await attemptRestore(saved);
32988
33406
  } catch (err) {
32989
- this.ctx.logger.warn("Failed to restore device", {
32990
- tags: { stableId: saved.stableId },
33407
+ const error = err instanceof Error ? err.message : String(err);
33408
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33409
+ tags: {
33410
+ deviceId: saved.id,
33411
+ stableId: saved.stableId
33412
+ },
32991
33413
  meta: {
32992
33414
  type: saved.type,
32993
- error: err instanceof Error ? err.message : String(err)
33415
+ attempt: 1,
33416
+ error
32994
33417
  }
32995
33418
  });
33419
+ failures.push({
33420
+ saved,
33421
+ error
33422
+ });
32996
33423
  }
32997
33424
  };
32998
33425
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33426
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32999
33427
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33000
33428
  for (const saved of childRows) {
33001
- const Class = this.deviceClasses[saved.type];
33002
- if (!Class) continue;
33429
+ if (!this.deviceClasses[saved.type]) continue;
33003
33430
  if (saved.parentDeviceId === null) continue;
33004
- if (!restored.has(saved.parentDeviceId)) continue;
33005
- try {
33006
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33007
- restored.add(saved.id);
33008
- } catch (err) {
33009
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33431
+ if (restored.has(saved.parentDeviceId)) {
33432
+ try {
33433
+ await attemptRestore(saved);
33434
+ } catch (err) {
33435
+ const error = err instanceof Error ? err.message : String(err);
33436
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33437
+ tags: {
33438
+ deviceId: saved.id,
33439
+ stableId: saved.stableId,
33440
+ parentDeviceId: saved.parentDeviceId
33441
+ },
33442
+ meta: {
33443
+ type: saved.type,
33444
+ attempt: 1,
33445
+ error
33446
+ }
33447
+ });
33448
+ failures.push({
33449
+ saved,
33450
+ error
33451
+ });
33452
+ }
33453
+ continue;
33454
+ }
33455
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33456
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33010
33457
  tags: {
33458
+ deviceId: saved.id,
33011
33459
  stableId: saved.stableId,
33012
33460
  parentDeviceId: saved.parentDeviceId
33013
33461
  },
33014
- meta: {
33015
- type: saved.type,
33016
- error: err instanceof Error ? err.message : String(err)
33017
- }
33462
+ meta: { type: saved.type }
33463
+ });
33464
+ failures.push({
33465
+ saved,
33466
+ error: `parent device ${saved.parentDeviceId} not restored`
33018
33467
  });
33468
+ continue;
33019
33469
  }
33020
33470
  }
33471
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33472
+ return {
33473
+ restoredCount: restored.size,
33474
+ failedCount: failures.length
33475
+ };
33021
33476
  }
33022
33477
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33023
33478
  toSummary(device) {
@@ -33732,6 +34187,12 @@ Object.freeze({
33732
34187
  addonId: null,
33733
34188
  access: "create"
33734
34189
  },
34190
+ "backup.cancel": {
34191
+ capName: "backup",
34192
+ capScope: "system",
34193
+ addonId: null,
34194
+ access: "create"
34195
+ },
33735
34196
  "backup.delete": {
33736
34197
  capName: "backup",
33737
34198
  capScope: "system",
@@ -33774,6 +34235,12 @@ Object.freeze({
33774
34235
  addonId: null,
33775
34236
  access: "view"
33776
34237
  },
34238
+ "backup.listRuns": {
34239
+ capName: "backup",
34240
+ capScope: "system",
34241
+ addonId: null,
34242
+ access: "view"
34243
+ },
33777
34244
  "backup.listSchedules": {
33778
34245
  capName: "backup",
33779
34246
  capScope: "system",
@@ -34974,6 +35441,12 @@ Object.freeze({
34974
35441
  addonId: null,
34975
35442
  access: "view"
34976
35443
  },
35444
+ "deviceProvider.reloadDevice": {
35445
+ capName: "device-provider",
35446
+ capScope: "system",
35447
+ addonId: null,
35448
+ access: "create"
35449
+ },
34977
35450
  "deviceProvider.start": {
34978
35451
  capName: "device-provider",
34979
35452
  capScope: "system",
@@ -38340,6 +38813,12 @@ Object.freeze({
38340
38813
  addonId: null,
38341
38814
  access: "create"
38342
38815
  },
38816
+ "streamBroker.forgetDeviceHardware": {
38817
+ capName: "stream-broker",
38818
+ capScope: "system",
38819
+ addonId: null,
38820
+ access: "delete"
38821
+ },
38343
38822
  "streamBroker.getAllRtspEntries": {
38344
38823
  capName: "stream-broker",
38345
38824
  capScope: "system",
@@ -40798,6 +41277,11 @@ Object.freeze({
40798
41277
  form: "single",
40799
41278
  optional: false
40800
41279
  }],
41280
+ "streamBroker.forgetDeviceHardware": [{
41281
+ name: "deviceId",
41282
+ form: "single",
41283
+ optional: false
41284
+ }],
40801
41285
  "streamBroker.getDeviceAudioMute": [{
40802
41286
  name: "deviceId",
40803
41287
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-amcrest",
3
- "version": "0.2.59",
3
+ "version": "0.2.61",
4
4
  "description": "Amcrest/Dahua camera device provider addon for CamStack — Dahua CGI over HTTP(S) with digest auth (snapshot, RTSP catalog, PTZ, image/day-night config)",
5
5
  "keywords": [
6
6
  "camstack",