@camstack/addon-provider-hikvision 1.2.66 → 1.2.68

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10803,6 +10803,89 @@ var LocationStatSchema = object({
10803
10803
  fileCount: number(),
10804
10804
  present: boolean()
10805
10805
  });
10806
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10807
+ var BackupRunStateSchema = _enum([
10808
+ "queued",
10809
+ "running",
10810
+ "succeeded",
10811
+ "failed",
10812
+ "cancelled"
10813
+ ]);
10814
+ /**
10815
+ * Where a running backup currently is. `queued` before it starts,
10816
+ * `building` while the tar.gz is being staged, `uploading` during the
10817
+ * per-destination fan-out, `done` once terminal.
10818
+ */
10819
+ var BackupRunPhaseSchema = _enum([
10820
+ "queued",
10821
+ "building",
10822
+ "uploading",
10823
+ "done"
10824
+ ]);
10825
+ /**
10826
+ * Observable state of one backup run — readable WHILE it runs via
10827
+ * `backup.listRuns`. This is what makes the execution queue and
10828
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10829
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10830
+ * diagnosable with `du` because nothing reported that runs existed or
10831
+ * how large the staged archive had grown.
10832
+ */
10833
+ var BackupRunSchema = object({
10834
+ /** Stable run id — the handle `backup.cancel` takes. */
10835
+ id: string(),
10836
+ state: BackupRunStateSchema,
10837
+ phase: BackupRunPhaseSchema,
10838
+ /**
10839
+ * Resolved destination location ids. Empty while queued (targets are
10840
+ * resolved when the run starts, against the then-current policies).
10841
+ */
10842
+ destinationIds: array(string()).readonly(),
10843
+ label: string().optional(),
10844
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10845
+ requestedAt: number(),
10846
+ /** ms-epoch when the run left the queue and started building. */
10847
+ startedAt: number().optional(),
10848
+ /** ms-epoch when the run reached a terminal state. */
10849
+ finishedAt: number().optional(),
10850
+ /** Compressed bytes of the staging archive written so far. */
10851
+ stagedBytes: number(),
10852
+ /** Final staged archive size, once the build phase completes. */
10853
+ archiveSizeBytes: number().optional(),
10854
+ /** Bytes pushed to the destination currently uploading. */
10855
+ uploadedBytes: number(),
10856
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10857
+ completedDestinationIds: array(string()).readonly(),
10858
+ /** Destinations that failed during the fan-out. */
10859
+ failedDestinationIds: array(string()).readonly(),
10860
+ /** Failure message when `state === 'failed'`. */
10861
+ error: string().optional(),
10862
+ /**
10863
+ * 1-based place in the execution queue — 1 = runs next. Present only
10864
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10865
+ * queue's OWN pending order, never derived from timestamps, so the
10866
+ * UI cannot show an order the executor will not honour.
10867
+ */
10868
+ queuePosition: number().int().min(1).optional()
10869
+ });
10870
+ /**
10871
+ * Result of `backup.trigger`. The call still resolves when the run
10872
+ * terminates (compat with schedule-driven runs and the admin UI), but
10873
+ * it now names the run and says whether it had to WAIT: a trigger that
10874
+ * arrives while another run is in flight is enqueued (or joined onto
10875
+ * an identical already-queued run), never started concurrently.
10876
+ */
10877
+ var BackupTriggerResultSchema = object({
10878
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10879
+ runId: string(),
10880
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10881
+ queued: boolean(),
10882
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10883
+ joined: boolean(),
10884
+ /** True when the run was cancelled before completing every destination. */
10885
+ cancelled: boolean(),
10886
+ /** One entry per destination the archive landed at (partial on cancel). */
10887
+ entries: array(BackupEntrySchema).readonly()
10888
+ });
10806
10889
  /**
10807
10890
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10808
10891
  * SET of destination locations. Supersedes the per-location cron on
@@ -10850,7 +10933,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10850
10933
  * retention (manual runs).
10851
10934
  */
10852
10935
  retentionCount: number().int().min(1).max(1e3).optional()
10853
- }).optional(), array(BackupEntrySchema).readonly(), {
10936
+ }).optional(), BackupTriggerResultSchema, {
10937
+ kind: "mutation",
10938
+ auth: "admin"
10939
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10854
10940
  kind: "mutation",
10855
10941
  auth: "admin"
10856
10942
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11567,6 +11653,14 @@ method(object({
11567
11653
  }), object({ success: literal(true) }), {
11568
11654
  kind: "mutation",
11569
11655
  auth: "admin"
11656
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11657
+ derivedStreamsDeleted: array(string()).readonly(),
11658
+ assignmentsPurged: boolean(),
11659
+ probeSnapshotsDropped: number().int().nonnegative(),
11660
+ rtspTokenRowsDeleted: number().int().nonnegative()
11661
+ }), {
11662
+ kind: "mutation",
11663
+ auth: "admin"
11570
11664
  }), method(object({
11571
11665
  deviceId: number(),
11572
11666
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13006,6 +13100,35 @@ var deviceProviderCapability = {
13006
13100
  name: string(),
13007
13101
  type: string()
13008
13102
  }))),
13103
+ /**
13104
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13105
+ * touching no other device this provider owns.
13106
+ *
13107
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13108
+ * migrated numbers: after `swapIds` the runner's live instance still
13109
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13110
+ * registrations and its log tags), and a live object cannot be renumbered.
13111
+ * Before this method the only flush was restarting the whole owning addon
13112
+ * — which took every camera the provider owns down with it (28 devices
13113
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13114
+ * same day ~27 devices' native caps did not come back on their own).
13115
+ *
13116
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13117
+ * that changes. The reply carries the id the device answers on NOW.
13118
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13119
+ * instance (if any), then re-create from the persisted row: the same
13120
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13121
+ * An RPC, never an event: a dropped event would leave the runner writing
13122
+ * against the wrong camera (D8).
13123
+ *
13124
+ * Construction can dial hardware, and the migrated source is
13125
+ * characteristically dead — the timeout covers a full activate window
13126
+ * rather than the 60 s default.
13127
+ */
13128
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13129
+ kind: "mutation",
13130
+ timeoutMs: 3 * 6e4
13131
+ }),
13009
13132
  supportsDiscovery: method(object({}), boolean()),
13010
13133
  /**
13011
13134
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13333,7 +13456,8 @@ method(object({
13333
13456
  targetId: number()
13334
13457
  }), MigrateDeviceResultSchema, {
13335
13458
  kind: "mutation",
13336
- auth: "admin"
13459
+ auth: "admin",
13460
+ timeoutMs: 12 * 6e4
13337
13461
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13338
13462
  deviceId: number(),
13339
13463
  name: string()
@@ -33158,6 +33282,147 @@ var BaseDevice = class {
33158
33282
  }
33159
33283
  };
33160
33284
  /**
33285
+ * Delays before retry rounds 1..N — the round count IS the bound.
33286
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33287
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33288
+ * per attempt) covers a device-manager lock held for minutes — the
33289
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33290
+ */
33291
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33292
+ 1e4,
33293
+ 3e4,
33294
+ 9e4
33295
+ ];
33296
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33297
+ function sleep$1(ms, signal) {
33298
+ return new Promise((resolve) => {
33299
+ if (signal.aborted) {
33300
+ resolve();
33301
+ return;
33302
+ }
33303
+ const onAbort = () => {
33304
+ clearTimeout(timer);
33305
+ resolve();
33306
+ };
33307
+ const timer = setTimeout(() => {
33308
+ signal.removeEventListener("abort", onAbort);
33309
+ resolve();
33310
+ }, ms);
33311
+ timer.unref?.();
33312
+ signal.addEventListener("abort", onAbort, { once: true });
33313
+ });
33314
+ }
33315
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33316
+ * not reject (callers wrap their own try/catch). */
33317
+ async function runWithConcurrency(items, width, fn) {
33318
+ const queue = [...items];
33319
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33320
+ const lane = async () => {
33321
+ for (;;) {
33322
+ const item = queue.shift();
33323
+ if (item === void 0) return;
33324
+ await fn(item);
33325
+ }
33326
+ };
33327
+ await Promise.all(Array.from({ length: laneCount }, lane));
33328
+ }
33329
+ var DeviceRestoreRetryScheduler = class {
33330
+ #logger;
33331
+ #attempt;
33332
+ #onPermanentFailure;
33333
+ #delaysMs;
33334
+ #concurrency;
33335
+ #now;
33336
+ #abort = new AbortController();
33337
+ constructor(options) {
33338
+ this.#logger = options.logger;
33339
+ this.#attempt = options.attempt;
33340
+ this.#onPermanentFailure = options.onPermanentFailure;
33341
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33342
+ this.#concurrency = options.concurrency ?? 4;
33343
+ this.#now = options.now ?? Date.now;
33344
+ }
33345
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33346
+ * permanently failed — the next boot restores them from disk. */
33347
+ cancel() {
33348
+ this.#abort.abort();
33349
+ }
33350
+ /**
33351
+ * Run the bounded retry rounds. Resolves when every entry has either
33352
+ * restored, been marked permanently failed, or the scheduler was
33353
+ * cancelled. Never rejects.
33354
+ */
33355
+ async run(initialFailures) {
33356
+ let pending = initialFailures.map((failure) => ({
33357
+ saved: failure.saved,
33358
+ lastError: failure.error,
33359
+ attempts: 1
33360
+ }));
33361
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33362
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33363
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33364
+ if (this.#abort.signal.aborted) break;
33365
+ pending = await this.#runRound(pending, round);
33366
+ }
33367
+ if (this.#abort.signal.aborted) return [];
33368
+ const terminal = pending.map((entry) => ({
33369
+ deviceId: entry.saved.id,
33370
+ stableId: entry.saved.stableId,
33371
+ type: String(entry.saved.type),
33372
+ attempts: entry.attempts,
33373
+ lastError: entry.lastError,
33374
+ failedAt: this.#now()
33375
+ }));
33376
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33377
+ return terminal;
33378
+ }
33379
+ /** One retry round: parents first (phase 0), then hub-adopted
33380
+ * children (phase 1) — a child's attempt depends on its parent
33381
+ * having landed, exactly like the initial two-pass restore. */
33382
+ async #runRound(pending, round) {
33383
+ const next = [];
33384
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33385
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33386
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33387
+ if (this.#abort.signal.aborted) {
33388
+ next.push(entry);
33389
+ return;
33390
+ }
33391
+ const attemptNo = entry.attempts + 1;
33392
+ try {
33393
+ await this.#attempt(entry.saved);
33394
+ this.#logger.info("Device restored on retry", {
33395
+ tags: {
33396
+ deviceId: entry.saved.id,
33397
+ stableId: entry.saved.stableId
33398
+ },
33399
+ meta: { attempt: attemptNo }
33400
+ });
33401
+ } catch (err) {
33402
+ const lastError = err instanceof Error ? err.message : String(err);
33403
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33404
+ this.#logger.warn("Device restore retry failed", {
33405
+ tags: {
33406
+ deviceId: entry.saved.id,
33407
+ stableId: entry.saved.stableId
33408
+ },
33409
+ meta: {
33410
+ attempt: attemptNo,
33411
+ remainingRetries,
33412
+ error: lastError
33413
+ }
33414
+ });
33415
+ next.push({
33416
+ saved: entry.saved,
33417
+ lastError,
33418
+ attempts: attemptNo
33419
+ });
33420
+ }
33421
+ });
33422
+ return next;
33423
+ }
33424
+ };
33425
+ /**
33161
33426
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33162
33427
  * device-provider cap router. Shared across all providers.
33163
33428
  */
@@ -33206,6 +33471,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33206
33471
  }];
33207
33472
  }
33208
33473
  async onShutdown() {
33474
+ this.cancelRestoreRetries();
33209
33475
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33210
33476
  for (const device of devices) try {
33211
33477
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33223,9 +33489,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33223
33489
  async start() {}
33224
33490
  async stop() {}
33225
33491
  async getStatus() {
33492
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33493
+ const summary = this.restoreFailureSummary();
33494
+ if (summary === null) return {
33495
+ connected: true,
33496
+ deviceCount: all.length
33497
+ };
33226
33498
  return {
33227
33499
  connected: true,
33228
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33500
+ deviceCount: all.length,
33501
+ error: summary
33229
33502
  };
33230
33503
  }
33231
33504
  async getDevices() {
@@ -33315,8 +33588,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33315
33588
  };
33316
33589
  }
33317
33590
  async restoreDevices(savedDevices) {
33318
- await this.onRestoreDevices(savedDevices);
33319
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33591
+ const report = await this.onRestoreDevices(savedDevices);
33592
+ if (savedDevices.length === 0) return;
33593
+ if (report && report.failedCount > 0) {
33594
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33595
+ return;
33596
+ }
33597
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33598
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33599
+ }
33600
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33601
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33602
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33603
+ * never re-stampede full-width while the initial pass does (D167). */
33604
+ restoreRetryConcurrency = 4;
33605
+ _restoreRetryScheduler = null;
33606
+ _restoreRetryCompletion = null;
33607
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33608
+ /** Settles when the background retry rounds finish (or `null` when
33609
+ * nothing failed). Exposed for tests and subclass diagnostics —
33610
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33611
+ * with the devices that restored, and a late success is announced
33612
+ * through the `native-cap-change` → `updateCaps` path. */
33613
+ get restoreRetryCompletion() {
33614
+ return this._restoreRetryCompletion;
33615
+ }
33616
+ /** Devices that exhausted the retry bound this process lifetime. */
33617
+ get permanentRestoreFailures() {
33618
+ return [...this._permanentRestoreFailures.values()];
33619
+ }
33620
+ /** One-line operator-facing summary for `getStatus().error`, or
33621
+ * `null` when every device restored. */
33622
+ restoreFailureSummary() {
33623
+ if (this._permanentRestoreFailures.size === 0) return null;
33624
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33625
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33626
+ }
33627
+ cancelRestoreRetries() {
33628
+ this._restoreRetryScheduler?.cancel();
33629
+ this._restoreRetryScheduler = null;
33630
+ }
33631
+ recordPermanentRestoreFailure(failure) {
33632
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33633
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33634
+ tags: {
33635
+ deviceId: failure.deviceId,
33636
+ stableId: failure.stableId
33637
+ },
33638
+ meta: {
33639
+ type: failure.type,
33640
+ attempts: failure.attempts,
33641
+ error: failure.lastError
33642
+ }
33643
+ });
33644
+ }
33645
+ scheduleRestoreRetries(failures, attempt) {
33646
+ const scheduler = new DeviceRestoreRetryScheduler({
33647
+ logger: this.ctx.logger,
33648
+ delaysMs: this.restoreRetryDelaysMs,
33649
+ concurrency: this.restoreRetryConcurrency,
33650
+ attempt,
33651
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33652
+ });
33653
+ this._restoreRetryScheduler = scheduler;
33654
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33655
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33656
+ });
33657
+ }
33658
+ /**
33659
+ * Tear down and reconstruct ONE device from its persisted rows — the
33660
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33661
+ * and no other device this provider owns is disturbed.
33662
+ *
33663
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33664
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33665
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33666
+ * whatever number the row carries NOW. The teardown is `decommission` —
33667
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33668
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33669
+ * the boot restore's own `create()` path, including its pass 2: first-class
33670
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33671
+ * parent by the cascade and must be re-created explicitly, because only
33672
+ * accessory children come back through `getAccessoryChildren()`.
33673
+ *
33674
+ * Reloading an accessory child directly is refused (no device class) —
33675
+ * reload its parent instead.
33676
+ */
33677
+ async reloadDevice(input) {
33678
+ const { stableId } = input;
33679
+ const devices = this.ctx.kernel.devices;
33680
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33681
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33682
+ if (live) await devices.decommission(live.id);
33683
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33684
+ addonId: this.addonId,
33685
+ stableId
33686
+ });
33687
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33688
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33689
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33690
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33691
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33692
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33693
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33694
+ for (const row of rows) {
33695
+ if (row.parentDeviceId !== id) continue;
33696
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33697
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33698
+ if (!ChildClass) continue;
33699
+ try {
33700
+ await devices.create(row.stableId, ChildClass, {}, id);
33701
+ } catch (err) {
33702
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33703
+ tags: {
33704
+ deviceId: row.id,
33705
+ stableId: row.stableId
33706
+ },
33707
+ meta: {
33708
+ parentDeviceId: id,
33709
+ error: err instanceof Error ? err.message : String(err)
33710
+ }
33711
+ });
33712
+ }
33713
+ }
33714
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33715
+ tags: { deviceId: id },
33716
+ meta: {
33717
+ stableId,
33718
+ type: meta.type
33719
+ }
33720
+ });
33721
+ return { deviceId: id };
33320
33722
  }
33321
33723
  /**
33322
33724
  * Restore devices from persisted state. Two-pass:
@@ -33342,55 +33744,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33342
33744
  * accessory-spawn flow handles via the parent's
33343
33745
  * `getAccessoryChildren()`. Override only when the default doesn't
33344
33746
  * fit.
33747
+ *
33748
+ * A row that fails either pass is NOT terminal (D347): it is handed
33749
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33750
+ * Only after the bound is exhausted is the device marked permanently
33751
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33752
+ * `getStatus().error`.
33345
33753
  */
33346
33754
  async onRestoreDevices(savedDevices) {
33347
33755
  const restored = /* @__PURE__ */ new Set();
33756
+ const failures = [];
33757
+ const attemptRestore = async (saved) => {
33758
+ if (restored.has(saved.id)) return;
33759
+ const Class = this.deviceClasses[saved.type];
33760
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33761
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33762
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33763
+ restored.add(saved.id);
33764
+ };
33348
33765
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33349
33766
  const restoreOne = async (saved) => {
33350
- const Class = this.deviceClasses[saved.type];
33351
- if (!Class) {
33767
+ if (!this.deviceClasses[saved.type]) {
33352
33768
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33353
- tags: { stableId: saved.stableId },
33769
+ tags: {
33770
+ deviceId: saved.id,
33771
+ stableId: saved.stableId
33772
+ },
33354
33773
  meta: { type: saved.type }
33355
33774
  });
33356
33775
  return;
33357
33776
  }
33358
33777
  try {
33359
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33360
- restored.add(saved.id);
33778
+ await attemptRestore(saved);
33361
33779
  } catch (err) {
33362
- this.ctx.logger.warn("Failed to restore device", {
33363
- tags: { stableId: saved.stableId },
33780
+ const error = err instanceof Error ? err.message : String(err);
33781
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33782
+ tags: {
33783
+ deviceId: saved.id,
33784
+ stableId: saved.stableId
33785
+ },
33364
33786
  meta: {
33365
33787
  type: saved.type,
33366
- error: err instanceof Error ? err.message : String(err)
33788
+ attempt: 1,
33789
+ error
33367
33790
  }
33368
33791
  });
33792
+ failures.push({
33793
+ saved,
33794
+ error
33795
+ });
33369
33796
  }
33370
33797
  };
33371
33798
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33799
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33372
33800
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33373
33801
  for (const saved of childRows) {
33374
- const Class = this.deviceClasses[saved.type];
33375
- if (!Class) continue;
33802
+ if (!this.deviceClasses[saved.type]) continue;
33376
33803
  if (saved.parentDeviceId === null) continue;
33377
- if (!restored.has(saved.parentDeviceId)) continue;
33378
- try {
33379
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33380
- restored.add(saved.id);
33381
- } catch (err) {
33382
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33804
+ if (restored.has(saved.parentDeviceId)) {
33805
+ try {
33806
+ await attemptRestore(saved);
33807
+ } catch (err) {
33808
+ const error = err instanceof Error ? err.message : String(err);
33809
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33810
+ tags: {
33811
+ deviceId: saved.id,
33812
+ stableId: saved.stableId,
33813
+ parentDeviceId: saved.parentDeviceId
33814
+ },
33815
+ meta: {
33816
+ type: saved.type,
33817
+ attempt: 1,
33818
+ error
33819
+ }
33820
+ });
33821
+ failures.push({
33822
+ saved,
33823
+ error
33824
+ });
33825
+ }
33826
+ continue;
33827
+ }
33828
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33829
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33383
33830
  tags: {
33831
+ deviceId: saved.id,
33384
33832
  stableId: saved.stableId,
33385
33833
  parentDeviceId: saved.parentDeviceId
33386
33834
  },
33387
- meta: {
33388
- type: saved.type,
33389
- error: err instanceof Error ? err.message : String(err)
33390
- }
33835
+ meta: { type: saved.type }
33391
33836
  });
33837
+ failures.push({
33838
+ saved,
33839
+ error: `parent device ${saved.parentDeviceId} not restored`
33840
+ });
33841
+ continue;
33392
33842
  }
33393
33843
  }
33844
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33845
+ return {
33846
+ restoredCount: restored.size,
33847
+ failedCount: failures.length
33848
+ };
33394
33849
  }
33395
33850
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33396
33851
  toSummary(device) {
@@ -34117,6 +34572,12 @@ Object.freeze({
34117
34572
  addonId: null,
34118
34573
  access: "create"
34119
34574
  },
34575
+ "backup.cancel": {
34576
+ capName: "backup",
34577
+ capScope: "system",
34578
+ addonId: null,
34579
+ access: "create"
34580
+ },
34120
34581
  "backup.delete": {
34121
34582
  capName: "backup",
34122
34583
  capScope: "system",
@@ -34159,6 +34620,12 @@ Object.freeze({
34159
34620
  addonId: null,
34160
34621
  access: "view"
34161
34622
  },
34623
+ "backup.listRuns": {
34624
+ capName: "backup",
34625
+ capScope: "system",
34626
+ addonId: null,
34627
+ access: "view"
34628
+ },
34162
34629
  "backup.listSchedules": {
34163
34630
  capName: "backup",
34164
34631
  capScope: "system",
@@ -35359,6 +35826,12 @@ Object.freeze({
35359
35826
  addonId: null,
35360
35827
  access: "view"
35361
35828
  },
35829
+ "deviceProvider.reloadDevice": {
35830
+ capName: "device-provider",
35831
+ capScope: "system",
35832
+ addonId: null,
35833
+ access: "create"
35834
+ },
35362
35835
  "deviceProvider.start": {
35363
35836
  capName: "device-provider",
35364
35837
  capScope: "system",
@@ -38725,6 +39198,12 @@ Object.freeze({
38725
39198
  addonId: null,
38726
39199
  access: "create"
38727
39200
  },
39201
+ "streamBroker.forgetDeviceHardware": {
39202
+ capName: "stream-broker",
39203
+ capScope: "system",
39204
+ addonId: null,
39205
+ access: "delete"
39206
+ },
38728
39207
  "streamBroker.getAllRtspEntries": {
38729
39208
  capName: "stream-broker",
38730
39209
  capScope: "system",
@@ -41183,6 +41662,11 @@ Object.freeze({
41183
41662
  form: "single",
41184
41663
  optional: false
41185
41664
  }],
41665
+ "streamBroker.forgetDeviceHardware": [{
41666
+ name: "deviceId",
41667
+ form: "single",
41668
+ optional: false
41669
+ }],
41186
41670
  "streamBroker.getDeviceAudioMute": [{
41187
41671
  name: "deviceId",
41188
41672
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10804,6 +10804,89 @@ var LocationStatSchema = object({
10804
10804
  fileCount: number(),
10805
10805
  present: boolean()
10806
10806
  });
10807
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10808
+ var BackupRunStateSchema = _enum([
10809
+ "queued",
10810
+ "running",
10811
+ "succeeded",
10812
+ "failed",
10813
+ "cancelled"
10814
+ ]);
10815
+ /**
10816
+ * Where a running backup currently is. `queued` before it starts,
10817
+ * `building` while the tar.gz is being staged, `uploading` during the
10818
+ * per-destination fan-out, `done` once terminal.
10819
+ */
10820
+ var BackupRunPhaseSchema = _enum([
10821
+ "queued",
10822
+ "building",
10823
+ "uploading",
10824
+ "done"
10825
+ ]);
10826
+ /**
10827
+ * Observable state of one backup run — readable WHILE it runs via
10828
+ * `backup.listRuns`. This is what makes the execution queue and
10829
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10830
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10831
+ * diagnosable with `du` because nothing reported that runs existed or
10832
+ * how large the staged archive had grown.
10833
+ */
10834
+ var BackupRunSchema = object({
10835
+ /** Stable run id — the handle `backup.cancel` takes. */
10836
+ id: string(),
10837
+ state: BackupRunStateSchema,
10838
+ phase: BackupRunPhaseSchema,
10839
+ /**
10840
+ * Resolved destination location ids. Empty while queued (targets are
10841
+ * resolved when the run starts, against the then-current policies).
10842
+ */
10843
+ destinationIds: array(string()).readonly(),
10844
+ label: string().optional(),
10845
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10846
+ requestedAt: number(),
10847
+ /** ms-epoch when the run left the queue and started building. */
10848
+ startedAt: number().optional(),
10849
+ /** ms-epoch when the run reached a terminal state. */
10850
+ finishedAt: number().optional(),
10851
+ /** Compressed bytes of the staging archive written so far. */
10852
+ stagedBytes: number(),
10853
+ /** Final staged archive size, once the build phase completes. */
10854
+ archiveSizeBytes: number().optional(),
10855
+ /** Bytes pushed to the destination currently uploading. */
10856
+ uploadedBytes: number(),
10857
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10858
+ completedDestinationIds: array(string()).readonly(),
10859
+ /** Destinations that failed during the fan-out. */
10860
+ failedDestinationIds: array(string()).readonly(),
10861
+ /** Failure message when `state === 'failed'`. */
10862
+ error: string().optional(),
10863
+ /**
10864
+ * 1-based place in the execution queue — 1 = runs next. Present only
10865
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10866
+ * queue's OWN pending order, never derived from timestamps, so the
10867
+ * UI cannot show an order the executor will not honour.
10868
+ */
10869
+ queuePosition: number().int().min(1).optional()
10870
+ });
10871
+ /**
10872
+ * Result of `backup.trigger`. The call still resolves when the run
10873
+ * terminates (compat with schedule-driven runs and the admin UI), but
10874
+ * it now names the run and says whether it had to WAIT: a trigger that
10875
+ * arrives while another run is in flight is enqueued (or joined onto
10876
+ * an identical already-queued run), never started concurrently.
10877
+ */
10878
+ var BackupTriggerResultSchema = object({
10879
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10880
+ runId: string(),
10881
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10882
+ queued: boolean(),
10883
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10884
+ joined: boolean(),
10885
+ /** True when the run was cancelled before completing every destination. */
10886
+ cancelled: boolean(),
10887
+ /** One entry per destination the archive landed at (partial on cancel). */
10888
+ entries: array(BackupEntrySchema).readonly()
10889
+ });
10807
10890
  /**
10808
10891
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10809
10892
  * SET of destination locations. Supersedes the per-location cron on
@@ -10851,7 +10934,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10851
10934
  * retention (manual runs).
10852
10935
  */
10853
10936
  retentionCount: number().int().min(1).max(1e3).optional()
10854
- }).optional(), array(BackupEntrySchema).readonly(), {
10937
+ }).optional(), BackupTriggerResultSchema, {
10938
+ kind: "mutation",
10939
+ auth: "admin"
10940
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10855
10941
  kind: "mutation",
10856
10942
  auth: "admin"
10857
10943
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11568,6 +11654,14 @@ method(object({
11568
11654
  }), object({ success: literal(true) }), {
11569
11655
  kind: "mutation",
11570
11656
  auth: "admin"
11657
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11658
+ derivedStreamsDeleted: array(string()).readonly(),
11659
+ assignmentsPurged: boolean(),
11660
+ probeSnapshotsDropped: number().int().nonnegative(),
11661
+ rtspTokenRowsDeleted: number().int().nonnegative()
11662
+ }), {
11663
+ kind: "mutation",
11664
+ auth: "admin"
11571
11665
  }), method(object({
11572
11666
  deviceId: number(),
11573
11667
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -13007,6 +13101,35 @@ var deviceProviderCapability = {
13007
13101
  name: string(),
13008
13102
  type: string()
13009
13103
  }))),
13104
+ /**
13105
+ * Tear down and reconstruct ONE device in place from its persisted rows —
13106
+ * touching no other device this provider owns.
13107
+ *
13108
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
13109
+ * migrated numbers: after `swapIds` the runner's live instance still
13110
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
13111
+ * registrations and its log tags), and a live object cannot be renumbered.
13112
+ * Before this method the only flush was restarting the whole owning addon
13113
+ * — which took every camera the provider owns down with it (28 devices
13114
+ * for one migrated camera, measured 2026-09-04, and the morning of the
13115
+ * same day ~27 devices' native caps did not come back on their own).
13116
+ *
13117
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
13118
+ * that changes. The reply carries the id the device answers on NOW.
13119
+ * Implemented once in `BaseDeviceProvider` — decommission the live
13120
+ * instance (if any), then re-create from the persisted row: the same
13121
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
13122
+ * An RPC, never an event: a dropped event would leave the runner writing
13123
+ * against the wrong camera (D8).
13124
+ *
13125
+ * Construction can dial hardware, and the migrated source is
13126
+ * characteristically dead — the timeout covers a full activate window
13127
+ * rather than the 60 s default.
13128
+ */
13129
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
13130
+ kind: "mutation",
13131
+ timeoutMs: 3 * 6e4
13132
+ }),
13010
13133
  supportsDiscovery: method(object({}), boolean()),
13011
13134
  /**
13012
13135
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13334,7 +13457,8 @@ method(object({
13334
13457
  targetId: number()
13335
13458
  }), MigrateDeviceResultSchema, {
13336
13459
  kind: "mutation",
13337
- auth: "admin"
13460
+ auth: "admin",
13461
+ timeoutMs: 12 * 6e4
13338
13462
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13339
13463
  deviceId: number(),
13340
13464
  name: string()
@@ -33159,6 +33283,147 @@ var BaseDevice = class {
33159
33283
  }
33160
33284
  };
33161
33285
  /**
33286
+ * Delays before retry rounds 1..N — the round count IS the bound.
33287
+ * 10 s catches "the hub was busy for a moment"; the full schedule
33288
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
33289
+ * per attempt) covers a device-manager lock held for minutes — the
33290
+ * 2026-09-04 outage's migration hold was ~3.5 min.
33291
+ */
33292
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
33293
+ 1e4,
33294
+ 3e4,
33295
+ 9e4
33296
+ ];
33297
+ /** Abortable sleep — resolves early (never rejects) on abort. */
33298
+ function sleep$1(ms, signal) {
33299
+ return new Promise((resolve) => {
33300
+ if (signal.aborted) {
33301
+ resolve();
33302
+ return;
33303
+ }
33304
+ const onAbort = () => {
33305
+ clearTimeout(timer);
33306
+ resolve();
33307
+ };
33308
+ const timer = setTimeout(() => {
33309
+ signal.removeEventListener("abort", onAbort);
33310
+ resolve();
33311
+ }, ms);
33312
+ timer.unref?.();
33313
+ signal.addEventListener("abort", onAbort, { once: true });
33314
+ });
33315
+ }
33316
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
33317
+ * not reject (callers wrap their own try/catch). */
33318
+ async function runWithConcurrency(items, width, fn) {
33319
+ const queue = [...items];
33320
+ const laneCount = Math.max(1, Math.min(width, queue.length));
33321
+ const lane = async () => {
33322
+ for (;;) {
33323
+ const item = queue.shift();
33324
+ if (item === void 0) return;
33325
+ await fn(item);
33326
+ }
33327
+ };
33328
+ await Promise.all(Array.from({ length: laneCount }, lane));
33329
+ }
33330
+ var DeviceRestoreRetryScheduler = class {
33331
+ #logger;
33332
+ #attempt;
33333
+ #onPermanentFailure;
33334
+ #delaysMs;
33335
+ #concurrency;
33336
+ #now;
33337
+ #abort = new AbortController();
33338
+ constructor(options) {
33339
+ this.#logger = options.logger;
33340
+ this.#attempt = options.attempt;
33341
+ this.#onPermanentFailure = options.onPermanentFailure;
33342
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
33343
+ this.#concurrency = options.concurrency ?? 4;
33344
+ this.#now = options.now ?? Date.now;
33345
+ }
33346
+ /** Stop retrying (shutdown). Pending entries are NOT marked
33347
+ * permanently failed — the next boot restores them from disk. */
33348
+ cancel() {
33349
+ this.#abort.abort();
33350
+ }
33351
+ /**
33352
+ * Run the bounded retry rounds. Resolves when every entry has either
33353
+ * restored, been marked permanently failed, or the scheduler was
33354
+ * cancelled. Never rejects.
33355
+ */
33356
+ async run(initialFailures) {
33357
+ let pending = initialFailures.map((failure) => ({
33358
+ saved: failure.saved,
33359
+ lastError: failure.error,
33360
+ attempts: 1
33361
+ }));
33362
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
33363
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
33364
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
33365
+ if (this.#abort.signal.aborted) break;
33366
+ pending = await this.#runRound(pending, round);
33367
+ }
33368
+ if (this.#abort.signal.aborted) return [];
33369
+ const terminal = pending.map((entry) => ({
33370
+ deviceId: entry.saved.id,
33371
+ stableId: entry.saved.stableId,
33372
+ type: String(entry.saved.type),
33373
+ attempts: entry.attempts,
33374
+ lastError: entry.lastError,
33375
+ failedAt: this.#now()
33376
+ }));
33377
+ for (const failure of terminal) this.#onPermanentFailure(failure);
33378
+ return terminal;
33379
+ }
33380
+ /** One retry round: parents first (phase 0), then hub-adopted
33381
+ * children (phase 1) — a child's attempt depends on its parent
33382
+ * having landed, exactly like the initial two-pass restore. */
33383
+ async #runRound(pending, round) {
33384
+ const next = [];
33385
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
33386
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
33387
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
33388
+ if (this.#abort.signal.aborted) {
33389
+ next.push(entry);
33390
+ return;
33391
+ }
33392
+ const attemptNo = entry.attempts + 1;
33393
+ try {
33394
+ await this.#attempt(entry.saved);
33395
+ this.#logger.info("Device restored on retry", {
33396
+ tags: {
33397
+ deviceId: entry.saved.id,
33398
+ stableId: entry.saved.stableId
33399
+ },
33400
+ meta: { attempt: attemptNo }
33401
+ });
33402
+ } catch (err) {
33403
+ const lastError = err instanceof Error ? err.message : String(err);
33404
+ const remainingRetries = this.#delaysMs.length - (round + 1);
33405
+ this.#logger.warn("Device restore retry failed", {
33406
+ tags: {
33407
+ deviceId: entry.saved.id,
33408
+ stableId: entry.saved.stableId
33409
+ },
33410
+ meta: {
33411
+ attempt: attemptNo,
33412
+ remainingRetries,
33413
+ error: lastError
33414
+ }
33415
+ });
33416
+ next.push({
33417
+ saved: entry.saved,
33418
+ lastError,
33419
+ attempts: attemptNo
33420
+ });
33421
+ }
33422
+ });
33423
+ return next;
33424
+ }
33425
+ };
33426
+ /**
33162
33427
  * Convert an IDevice to the flat DeviceSummary shape expected by the
33163
33428
  * device-provider cap router. Shared across all providers.
33164
33429
  */
@@ -33207,6 +33472,7 @@ var BaseDeviceProvider = class extends BaseAddon {
33207
33472
  }];
33208
33473
  }
33209
33474
  async onShutdown() {
33475
+ this.cancelRestoreRetries();
33210
33476
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
33211
33477
  for (const device of devices) try {
33212
33478
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -33224,9 +33490,16 @@ var BaseDeviceProvider = class extends BaseAddon {
33224
33490
  async start() {}
33225
33491
  async stop() {}
33226
33492
  async getStatus() {
33493
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
33494
+ const summary = this.restoreFailureSummary();
33495
+ if (summary === null) return {
33496
+ connected: true,
33497
+ deviceCount: all.length
33498
+ };
33227
33499
  return {
33228
33500
  connected: true,
33229
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
33501
+ deviceCount: all.length,
33502
+ error: summary
33230
33503
  };
33231
33504
  }
33232
33505
  async getDevices() {
@@ -33316,8 +33589,137 @@ var BaseDeviceProvider = class extends BaseAddon {
33316
33589
  };
33317
33590
  }
33318
33591
  async restoreDevices(savedDevices) {
33319
- await this.onRestoreDevices(savedDevices);
33320
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
33592
+ const report = await this.onRestoreDevices(savedDevices);
33593
+ if (savedDevices.length === 0) return;
33594
+ if (report && report.failedCount > 0) {
33595
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
33596
+ return;
33597
+ }
33598
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
33599
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
33600
+ }
33601
+ /** Retry schedule. Overridable (tests use millisecond delays). */
33602
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
33603
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
33604
+ * never re-stampede full-width while the initial pass does (D167). */
33605
+ restoreRetryConcurrency = 4;
33606
+ _restoreRetryScheduler = null;
33607
+ _restoreRetryCompletion = null;
33608
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
33609
+ /** Settles when the background retry rounds finish (or `null` when
33610
+ * nothing failed). Exposed for tests and subclass diagnostics —
33611
+ * boot NEVER awaits this: the runner's post-init handshake goes out
33612
+ * with the devices that restored, and a late success is announced
33613
+ * through the `native-cap-change` → `updateCaps` path. */
33614
+ get restoreRetryCompletion() {
33615
+ return this._restoreRetryCompletion;
33616
+ }
33617
+ /** Devices that exhausted the retry bound this process lifetime. */
33618
+ get permanentRestoreFailures() {
33619
+ return [...this._permanentRestoreFailures.values()];
33620
+ }
33621
+ /** One-line operator-facing summary for `getStatus().error`, or
33622
+ * `null` when every device restored. */
33623
+ restoreFailureSummary() {
33624
+ if (this._permanentRestoreFailures.size === 0) return null;
33625
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
33626
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
33627
+ }
33628
+ cancelRestoreRetries() {
33629
+ this._restoreRetryScheduler?.cancel();
33630
+ this._restoreRetryScheduler = null;
33631
+ }
33632
+ recordPermanentRestoreFailure(failure) {
33633
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
33634
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
33635
+ tags: {
33636
+ deviceId: failure.deviceId,
33637
+ stableId: failure.stableId
33638
+ },
33639
+ meta: {
33640
+ type: failure.type,
33641
+ attempts: failure.attempts,
33642
+ error: failure.lastError
33643
+ }
33644
+ });
33645
+ }
33646
+ scheduleRestoreRetries(failures, attempt) {
33647
+ const scheduler = new DeviceRestoreRetryScheduler({
33648
+ logger: this.ctx.logger,
33649
+ delaysMs: this.restoreRetryDelaysMs,
33650
+ concurrency: this.restoreRetryConcurrency,
33651
+ attempt,
33652
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
33653
+ });
33654
+ this._restoreRetryScheduler = scheduler;
33655
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
33656
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
33657
+ });
33658
+ }
33659
+ /**
33660
+ * Tear down and reconstruct ONE device from its persisted rows — the
33661
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
33662
+ * and no other device this provider owns is disturbed.
33663
+ *
33664
+ * Keyed by `stableId` because the caller's whole reason to be here is that
33665
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
33666
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
33667
+ * whatever number the row carries NOW. The teardown is `decommission` —
33668
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
33669
+ * unregisters native caps, drops the registry entry) — and the rebuild is
33670
+ * the boot restore's own `create()` path, including its pass 2: first-class
33671
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
33672
+ * parent by the cascade and must be re-created explicitly, because only
33673
+ * accessory children come back through `getAccessoryChildren()`.
33674
+ *
33675
+ * Reloading an accessory child directly is refused (no device class) —
33676
+ * reload its parent instead.
33677
+ */
33678
+ async reloadDevice(input) {
33679
+ const { stableId } = input;
33680
+ const devices = this.ctx.kernel.devices;
33681
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
33682
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
33683
+ if (live) await devices.decommission(live.id);
33684
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
33685
+ addonId: this.addonId,
33686
+ stableId
33687
+ });
33688
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
33689
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
33690
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
33691
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
33692
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
33693
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
33694
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
33695
+ for (const row of rows) {
33696
+ if (row.parentDeviceId !== id) continue;
33697
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
33698
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
33699
+ if (!ChildClass) continue;
33700
+ try {
33701
+ await devices.create(row.stableId, ChildClass, {}, id);
33702
+ } catch (err) {
33703
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
33704
+ tags: {
33705
+ deviceId: row.id,
33706
+ stableId: row.stableId
33707
+ },
33708
+ meta: {
33709
+ parentDeviceId: id,
33710
+ error: err instanceof Error ? err.message : String(err)
33711
+ }
33712
+ });
33713
+ }
33714
+ }
33715
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
33716
+ tags: { deviceId: id },
33717
+ meta: {
33718
+ stableId,
33719
+ type: meta.type
33720
+ }
33721
+ });
33722
+ return { deviceId: id };
33321
33723
  }
33322
33724
  /**
33323
33725
  * Restore devices from persisted state. Two-pass:
@@ -33343,55 +33745,108 @@ var BaseDeviceProvider = class extends BaseAddon {
33343
33745
  * accessory-spawn flow handles via the parent's
33344
33746
  * `getAccessoryChildren()`. Override only when the default doesn't
33345
33747
  * fit.
33748
+ *
33749
+ * A row that fails either pass is NOT terminal (D347): it is handed
33750
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
33751
+ * Only after the bound is exhausted is the device marked permanently
33752
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
33753
+ * `getStatus().error`.
33346
33754
  */
33347
33755
  async onRestoreDevices(savedDevices) {
33348
33756
  const restored = /* @__PURE__ */ new Set();
33757
+ const failures = [];
33758
+ const attemptRestore = async (saved) => {
33759
+ if (restored.has(saved.id)) return;
33760
+ const Class = this.deviceClasses[saved.type];
33761
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33762
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33763
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33764
+ restored.add(saved.id);
33765
+ };
33349
33766
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
33350
33767
  const restoreOne = async (saved) => {
33351
- const Class = this.deviceClasses[saved.type];
33352
- if (!Class) {
33768
+ if (!this.deviceClasses[saved.type]) {
33353
33769
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
33354
- tags: { stableId: saved.stableId },
33770
+ tags: {
33771
+ deviceId: saved.id,
33772
+ stableId: saved.stableId
33773
+ },
33355
33774
  meta: { type: saved.type }
33356
33775
  });
33357
33776
  return;
33358
33777
  }
33359
33778
  try {
33360
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
33361
- restored.add(saved.id);
33779
+ await attemptRestore(saved);
33362
33780
  } catch (err) {
33363
- this.ctx.logger.warn("Failed to restore device", {
33364
- tags: { stableId: saved.stableId },
33781
+ const error = err instanceof Error ? err.message : String(err);
33782
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33783
+ tags: {
33784
+ deviceId: saved.id,
33785
+ stableId: saved.stableId
33786
+ },
33365
33787
  meta: {
33366
33788
  type: saved.type,
33367
- error: err instanceof Error ? err.message : String(err)
33789
+ attempt: 1,
33790
+ error
33368
33791
  }
33369
33792
  });
33793
+ failures.push({
33794
+ saved,
33795
+ error
33796
+ });
33370
33797
  }
33371
33798
  };
33372
33799
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33800
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
33373
33801
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
33374
33802
  for (const saved of childRows) {
33375
- const Class = this.deviceClasses[saved.type];
33376
- if (!Class) continue;
33803
+ if (!this.deviceClasses[saved.type]) continue;
33377
33804
  if (saved.parentDeviceId === null) continue;
33378
- if (!restored.has(saved.parentDeviceId)) continue;
33379
- try {
33380
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33381
- restored.add(saved.id);
33382
- } catch (err) {
33383
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33805
+ if (restored.has(saved.parentDeviceId)) {
33806
+ try {
33807
+ await attemptRestore(saved);
33808
+ } catch (err) {
33809
+ const error = err instanceof Error ? err.message : String(err);
33810
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33811
+ tags: {
33812
+ deviceId: saved.id,
33813
+ stableId: saved.stableId,
33814
+ parentDeviceId: saved.parentDeviceId
33815
+ },
33816
+ meta: {
33817
+ type: saved.type,
33818
+ attempt: 1,
33819
+ error
33820
+ }
33821
+ });
33822
+ failures.push({
33823
+ saved,
33824
+ error
33825
+ });
33826
+ }
33827
+ continue;
33828
+ }
33829
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33830
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
33384
33831
  tags: {
33832
+ deviceId: saved.id,
33385
33833
  stableId: saved.stableId,
33386
33834
  parentDeviceId: saved.parentDeviceId
33387
33835
  },
33388
- meta: {
33389
- type: saved.type,
33390
- error: err instanceof Error ? err.message : String(err)
33391
- }
33836
+ meta: { type: saved.type }
33392
33837
  });
33838
+ failures.push({
33839
+ saved,
33840
+ error: `parent device ${saved.parentDeviceId} not restored`
33841
+ });
33842
+ continue;
33393
33843
  }
33394
33844
  }
33845
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33846
+ return {
33847
+ restoredCount: restored.size,
33848
+ failedCount: failures.length
33849
+ };
33395
33850
  }
33396
33851
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
33397
33852
  toSummary(device) {
@@ -34118,6 +34573,12 @@ Object.freeze({
34118
34573
  addonId: null,
34119
34574
  access: "create"
34120
34575
  },
34576
+ "backup.cancel": {
34577
+ capName: "backup",
34578
+ capScope: "system",
34579
+ addonId: null,
34580
+ access: "create"
34581
+ },
34121
34582
  "backup.delete": {
34122
34583
  capName: "backup",
34123
34584
  capScope: "system",
@@ -34160,6 +34621,12 @@ Object.freeze({
34160
34621
  addonId: null,
34161
34622
  access: "view"
34162
34623
  },
34624
+ "backup.listRuns": {
34625
+ capName: "backup",
34626
+ capScope: "system",
34627
+ addonId: null,
34628
+ access: "view"
34629
+ },
34163
34630
  "backup.listSchedules": {
34164
34631
  capName: "backup",
34165
34632
  capScope: "system",
@@ -35360,6 +35827,12 @@ Object.freeze({
35360
35827
  addonId: null,
35361
35828
  access: "view"
35362
35829
  },
35830
+ "deviceProvider.reloadDevice": {
35831
+ capName: "device-provider",
35832
+ capScope: "system",
35833
+ addonId: null,
35834
+ access: "create"
35835
+ },
35363
35836
  "deviceProvider.start": {
35364
35837
  capName: "device-provider",
35365
35838
  capScope: "system",
@@ -38726,6 +39199,12 @@ Object.freeze({
38726
39199
  addonId: null,
38727
39200
  access: "create"
38728
39201
  },
39202
+ "streamBroker.forgetDeviceHardware": {
39203
+ capName: "stream-broker",
39204
+ capScope: "system",
39205
+ addonId: null,
39206
+ access: "delete"
39207
+ },
38729
39208
  "streamBroker.getAllRtspEntries": {
38730
39209
  capName: "stream-broker",
38731
39210
  capScope: "system",
@@ -41184,6 +41663,11 @@ Object.freeze({
41184
41663
  form: "single",
41185
41664
  optional: false
41186
41665
  }],
41666
+ "streamBroker.forgetDeviceHardware": [{
41667
+ name: "deviceId",
41668
+ form: "single",
41669
+ optional: false
41670
+ }],
41187
41671
  "streamBroker.getDeviceAudioMute": [{
41188
41672
  name: "deviceId",
41189
41673
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-hikvision",
3
- "version": "1.2.66",
3
+ "version": "1.2.68",
4
4
  "description": "Hikvision camera device provider addon for CamStack — ISAPI over HTTP(S) with digest auth (snapshot, alarm stream, RTSP discovery)",
5
5
  "keywords": [
6
6
  "camstack",