@camstack/addon-provider-gree 0.2.57 → 0.2.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/dist/addon.js +509 -25
  2. package/dist/addon.mjs +509 -25
  3. package/package.json +1 -1
package/dist/addon.js CHANGED
@@ -10530,6 +10530,89 @@ var LocationStatSchema = object({
10530
10530
  fileCount: number(),
10531
10531
  present: boolean()
10532
10532
  });
10533
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10534
+ var BackupRunStateSchema = _enum([
10535
+ "queued",
10536
+ "running",
10537
+ "succeeded",
10538
+ "failed",
10539
+ "cancelled"
10540
+ ]);
10541
+ /**
10542
+ * Where a running backup currently is. `queued` before it starts,
10543
+ * `building` while the tar.gz is being staged, `uploading` during the
10544
+ * per-destination fan-out, `done` once terminal.
10545
+ */
10546
+ var BackupRunPhaseSchema = _enum([
10547
+ "queued",
10548
+ "building",
10549
+ "uploading",
10550
+ "done"
10551
+ ]);
10552
+ /**
10553
+ * Observable state of one backup run — readable WHILE it runs via
10554
+ * `backup.listRuns`. This is what makes the execution queue and
10555
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10556
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10557
+ * diagnosable with `du` because nothing reported that runs existed or
10558
+ * how large the staged archive had grown.
10559
+ */
10560
+ var BackupRunSchema = object({
10561
+ /** Stable run id — the handle `backup.cancel` takes. */
10562
+ id: string(),
10563
+ state: BackupRunStateSchema,
10564
+ phase: BackupRunPhaseSchema,
10565
+ /**
10566
+ * Resolved destination location ids. Empty while queued (targets are
10567
+ * resolved when the run starts, against the then-current policies).
10568
+ */
10569
+ destinationIds: array(string()).readonly(),
10570
+ label: string().optional(),
10571
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10572
+ requestedAt: number(),
10573
+ /** ms-epoch when the run left the queue and started building. */
10574
+ startedAt: number().optional(),
10575
+ /** ms-epoch when the run reached a terminal state. */
10576
+ finishedAt: number().optional(),
10577
+ /** Compressed bytes of the staging archive written so far. */
10578
+ stagedBytes: number(),
10579
+ /** Final staged archive size, once the build phase completes. */
10580
+ archiveSizeBytes: number().optional(),
10581
+ /** Bytes pushed to the destination currently uploading. */
10582
+ uploadedBytes: number(),
10583
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10584
+ completedDestinationIds: array(string()).readonly(),
10585
+ /** Destinations that failed during the fan-out. */
10586
+ failedDestinationIds: array(string()).readonly(),
10587
+ /** Failure message when `state === 'failed'`. */
10588
+ error: string().optional(),
10589
+ /**
10590
+ * 1-based place in the execution queue — 1 = runs next. Present only
10591
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10592
+ * queue's OWN pending order, never derived from timestamps, so the
10593
+ * UI cannot show an order the executor will not honour.
10594
+ */
10595
+ queuePosition: number().int().min(1).optional()
10596
+ });
10597
+ /**
10598
+ * Result of `backup.trigger`. The call still resolves when the run
10599
+ * terminates (compat with schedule-driven runs and the admin UI), but
10600
+ * it now names the run and says whether it had to WAIT: a trigger that
10601
+ * arrives while another run is in flight is enqueued (or joined onto
10602
+ * an identical already-queued run), never started concurrently.
10603
+ */
10604
+ var BackupTriggerResultSchema = object({
10605
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10606
+ runId: string(),
10607
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10608
+ queued: boolean(),
10609
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10610
+ joined: boolean(),
10611
+ /** True when the run was cancelled before completing every destination. */
10612
+ cancelled: boolean(),
10613
+ /** One entry per destination the archive landed at (partial on cancel). */
10614
+ entries: array(BackupEntrySchema).readonly()
10615
+ });
10533
10616
  /**
10534
10617
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10535
10618
  * SET of destination locations. Supersedes the per-location cron on
@@ -10577,7 +10660,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10577
10660
  * retention (manual runs).
10578
10661
  */
10579
10662
  retentionCount: number().int().min(1).max(1e3).optional()
10580
- }).optional(), array(BackupEntrySchema).readonly(), {
10663
+ }).optional(), BackupTriggerResultSchema, {
10664
+ kind: "mutation",
10665
+ auth: "admin"
10666
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10581
10667
  kind: "mutation",
10582
10668
  auth: "admin"
10583
10669
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11294,6 +11380,14 @@ method(object({
11294
11380
  }), object({ success: literal(true) }), {
11295
11381
  kind: "mutation",
11296
11382
  auth: "admin"
11383
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11384
+ derivedStreamsDeleted: array(string()).readonly(),
11385
+ assignmentsPurged: boolean(),
11386
+ probeSnapshotsDropped: number().int().nonnegative(),
11387
+ rtspTokenRowsDeleted: number().int().nonnegative()
11388
+ }), {
11389
+ kind: "mutation",
11390
+ auth: "admin"
11297
11391
  }), method(object({
11298
11392
  deviceId: number(),
11299
11393
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12721,6 +12815,35 @@ var deviceProviderCapability = {
12721
12815
  name: string(),
12722
12816
  type: string()
12723
12817
  }))),
12818
+ /**
12819
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12820
+ * touching no other device this provider owns.
12821
+ *
12822
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12823
+ * migrated numbers: after `swapIds` the runner's live instance still
12824
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12825
+ * registrations and its log tags), and a live object cannot be renumbered.
12826
+ * Before this method the only flush was restarting the whole owning addon
12827
+ * — which took every camera the provider owns down with it (28 devices
12828
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12829
+ * same day ~27 devices' native caps did not come back on their own).
12830
+ *
12831
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12832
+ * that changes. The reply carries the id the device answers on NOW.
12833
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12834
+ * instance (if any), then re-create from the persisted row: the same
12835
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12836
+ * An RPC, never an event: a dropped event would leave the runner writing
12837
+ * against the wrong camera (D8).
12838
+ *
12839
+ * Construction can dial hardware, and the migrated source is
12840
+ * characteristically dead — the timeout covers a full activate window
12841
+ * rather than the 60 s default.
12842
+ */
12843
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12844
+ kind: "mutation",
12845
+ timeoutMs: 3 * 6e4
12846
+ }),
12724
12847
  supportsDiscovery: method(object({}), boolean()),
12725
12848
  /**
12726
12849
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13048,7 +13171,8 @@ method(object({
13048
13171
  targetId: number()
13049
13172
  }), MigrateDeviceResultSchema, {
13050
13173
  kind: "mutation",
13051
- auth: "admin"
13174
+ auth: "admin",
13175
+ timeoutMs: 12 * 6e4
13052
13176
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13053
13177
  deviceId: number(),
13054
13178
  name: string()
@@ -32404,6 +32528,147 @@ var BaseDevice$1 = class {
32404
32528
  }
32405
32529
  };
32406
32530
  /**
32531
+ * Delays before retry rounds 1..N — the round count IS the bound.
32532
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32533
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32534
+ * per attempt) covers a device-manager lock held for minutes — the
32535
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32536
+ */
32537
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32538
+ 1e4,
32539
+ 3e4,
32540
+ 9e4
32541
+ ];
32542
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32543
+ function sleep$1(ms, signal) {
32544
+ return new Promise((resolve) => {
32545
+ if (signal.aborted) {
32546
+ resolve();
32547
+ return;
32548
+ }
32549
+ const onAbort = () => {
32550
+ clearTimeout(timer);
32551
+ resolve();
32552
+ };
32553
+ const timer = setTimeout(() => {
32554
+ signal.removeEventListener("abort", onAbort);
32555
+ resolve();
32556
+ }, ms);
32557
+ timer.unref?.();
32558
+ signal.addEventListener("abort", onAbort, { once: true });
32559
+ });
32560
+ }
32561
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32562
+ * not reject (callers wrap their own try/catch). */
32563
+ async function runWithConcurrency(items, width, fn) {
32564
+ const queue = [...items];
32565
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32566
+ const lane = async () => {
32567
+ for (;;) {
32568
+ const item = queue.shift();
32569
+ if (item === void 0) return;
32570
+ await fn(item);
32571
+ }
32572
+ };
32573
+ await Promise.all(Array.from({ length: laneCount }, lane));
32574
+ }
32575
+ var DeviceRestoreRetryScheduler = class {
32576
+ #logger;
32577
+ #attempt;
32578
+ #onPermanentFailure;
32579
+ #delaysMs;
32580
+ #concurrency;
32581
+ #now;
32582
+ #abort = new AbortController();
32583
+ constructor(options) {
32584
+ this.#logger = options.logger;
32585
+ this.#attempt = options.attempt;
32586
+ this.#onPermanentFailure = options.onPermanentFailure;
32587
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32588
+ this.#concurrency = options.concurrency ?? 4;
32589
+ this.#now = options.now ?? Date.now;
32590
+ }
32591
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32592
+ * permanently failed — the next boot restores them from disk. */
32593
+ cancel() {
32594
+ this.#abort.abort();
32595
+ }
32596
+ /**
32597
+ * Run the bounded retry rounds. Resolves when every entry has either
32598
+ * restored, been marked permanently failed, or the scheduler was
32599
+ * cancelled. Never rejects.
32600
+ */
32601
+ async run(initialFailures) {
32602
+ let pending = initialFailures.map((failure) => ({
32603
+ saved: failure.saved,
32604
+ lastError: failure.error,
32605
+ attempts: 1
32606
+ }));
32607
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32608
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32609
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32610
+ if (this.#abort.signal.aborted) break;
32611
+ pending = await this.#runRound(pending, round);
32612
+ }
32613
+ if (this.#abort.signal.aborted) return [];
32614
+ const terminal = pending.map((entry) => ({
32615
+ deviceId: entry.saved.id,
32616
+ stableId: entry.saved.stableId,
32617
+ type: String(entry.saved.type),
32618
+ attempts: entry.attempts,
32619
+ lastError: entry.lastError,
32620
+ failedAt: this.#now()
32621
+ }));
32622
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32623
+ return terminal;
32624
+ }
32625
+ /** One retry round: parents first (phase 0), then hub-adopted
32626
+ * children (phase 1) — a child's attempt depends on its parent
32627
+ * having landed, exactly like the initial two-pass restore. */
32628
+ async #runRound(pending, round) {
32629
+ const next = [];
32630
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32631
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32632
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32633
+ if (this.#abort.signal.aborted) {
32634
+ next.push(entry);
32635
+ return;
32636
+ }
32637
+ const attemptNo = entry.attempts + 1;
32638
+ try {
32639
+ await this.#attempt(entry.saved);
32640
+ this.#logger.info("Device restored on retry", {
32641
+ tags: {
32642
+ deviceId: entry.saved.id,
32643
+ stableId: entry.saved.stableId
32644
+ },
32645
+ meta: { attempt: attemptNo }
32646
+ });
32647
+ } catch (err) {
32648
+ const lastError = err instanceof Error ? err.message : String(err);
32649
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32650
+ this.#logger.warn("Device restore retry failed", {
32651
+ tags: {
32652
+ deviceId: entry.saved.id,
32653
+ stableId: entry.saved.stableId
32654
+ },
32655
+ meta: {
32656
+ attempt: attemptNo,
32657
+ remainingRetries,
32658
+ error: lastError
32659
+ }
32660
+ });
32661
+ next.push({
32662
+ saved: entry.saved,
32663
+ lastError,
32664
+ attempts: attemptNo
32665
+ });
32666
+ }
32667
+ });
32668
+ return next;
32669
+ }
32670
+ };
32671
+ /**
32407
32672
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32408
32673
  * device-provider cap router. Shared across all providers.
32409
32674
  */
@@ -32452,6 +32717,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32452
32717
  }];
32453
32718
  }
32454
32719
  async onShutdown() {
32720
+ this.cancelRestoreRetries();
32455
32721
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32456
32722
  for (const device of devices) try {
32457
32723
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32469,9 +32735,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32469
32735
  async start() {}
32470
32736
  async stop() {}
32471
32737
  async getStatus() {
32738
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32739
+ const summary = this.restoreFailureSummary();
32740
+ if (summary === null) return {
32741
+ connected: true,
32742
+ deviceCount: all.length
32743
+ };
32472
32744
  return {
32473
32745
  connected: true,
32474
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32746
+ deviceCount: all.length,
32747
+ error: summary
32475
32748
  };
32476
32749
  }
32477
32750
  async getDevices() {
@@ -32561,8 +32834,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32561
32834
  };
32562
32835
  }
32563
32836
  async restoreDevices(savedDevices) {
32564
- await this.onRestoreDevices(savedDevices);
32565
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32837
+ const report = await this.onRestoreDevices(savedDevices);
32838
+ if (savedDevices.length === 0) return;
32839
+ if (report && report.failedCount > 0) {
32840
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32841
+ return;
32842
+ }
32843
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32844
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32845
+ }
32846
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32847
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32848
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32849
+ * never re-stampede full-width while the initial pass does (D167). */
32850
+ restoreRetryConcurrency = 4;
32851
+ _restoreRetryScheduler = null;
32852
+ _restoreRetryCompletion = null;
32853
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32854
+ /** Settles when the background retry rounds finish (or `null` when
32855
+ * nothing failed). Exposed for tests and subclass diagnostics —
32856
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32857
+ * with the devices that restored, and a late success is announced
32858
+ * through the `native-cap-change` → `updateCaps` path. */
32859
+ get restoreRetryCompletion() {
32860
+ return this._restoreRetryCompletion;
32861
+ }
32862
+ /** Devices that exhausted the retry bound this process lifetime. */
32863
+ get permanentRestoreFailures() {
32864
+ return [...this._permanentRestoreFailures.values()];
32865
+ }
32866
+ /** One-line operator-facing summary for `getStatus().error`, or
32867
+ * `null` when every device restored. */
32868
+ restoreFailureSummary() {
32869
+ if (this._permanentRestoreFailures.size === 0) return null;
32870
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32871
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32872
+ }
32873
+ cancelRestoreRetries() {
32874
+ this._restoreRetryScheduler?.cancel();
32875
+ this._restoreRetryScheduler = null;
32876
+ }
32877
+ recordPermanentRestoreFailure(failure) {
32878
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32879
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32880
+ tags: {
32881
+ deviceId: failure.deviceId,
32882
+ stableId: failure.stableId
32883
+ },
32884
+ meta: {
32885
+ type: failure.type,
32886
+ attempts: failure.attempts,
32887
+ error: failure.lastError
32888
+ }
32889
+ });
32890
+ }
32891
+ scheduleRestoreRetries(failures, attempt) {
32892
+ const scheduler = new DeviceRestoreRetryScheduler({
32893
+ logger: this.ctx.logger,
32894
+ delaysMs: this.restoreRetryDelaysMs,
32895
+ concurrency: this.restoreRetryConcurrency,
32896
+ attempt,
32897
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32898
+ });
32899
+ this._restoreRetryScheduler = scheduler;
32900
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32901
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32902
+ });
32903
+ }
32904
+ /**
32905
+ * Tear down and reconstruct ONE device from its persisted rows — the
32906
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32907
+ * and no other device this provider owns is disturbed.
32908
+ *
32909
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32910
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32911
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32912
+ * whatever number the row carries NOW. The teardown is `decommission` —
32913
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32914
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32915
+ * the boot restore's own `create()` path, including its pass 2: first-class
32916
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32917
+ * parent by the cascade and must be re-created explicitly, because only
32918
+ * accessory children come back through `getAccessoryChildren()`.
32919
+ *
32920
+ * Reloading an accessory child directly is refused (no device class) —
32921
+ * reload its parent instead.
32922
+ */
32923
+ async reloadDevice(input) {
32924
+ const { stableId } = input;
32925
+ const devices = this.ctx.kernel.devices;
32926
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32927
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32928
+ if (live) await devices.decommission(live.id);
32929
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32930
+ addonId: this.addonId,
32931
+ stableId
32932
+ });
32933
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32934
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32935
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32936
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32937
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32938
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32939
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32940
+ for (const row of rows) {
32941
+ if (row.parentDeviceId !== id) continue;
32942
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32943
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32944
+ if (!ChildClass) continue;
32945
+ try {
32946
+ await devices.create(row.stableId, ChildClass, {}, id);
32947
+ } catch (err) {
32948
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32949
+ tags: {
32950
+ deviceId: row.id,
32951
+ stableId: row.stableId
32952
+ },
32953
+ meta: {
32954
+ parentDeviceId: id,
32955
+ error: err instanceof Error ? err.message : String(err)
32956
+ }
32957
+ });
32958
+ }
32959
+ }
32960
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32961
+ tags: { deviceId: id },
32962
+ meta: {
32963
+ stableId,
32964
+ type: meta.type
32965
+ }
32966
+ });
32967
+ return { deviceId: id };
32566
32968
  }
32567
32969
  /**
32568
32970
  * Restore devices from persisted state. Two-pass:
@@ -32588,55 +32990,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32588
32990
  * accessory-spawn flow handles via the parent's
32589
32991
  * `getAccessoryChildren()`. Override only when the default doesn't
32590
32992
  * fit.
32993
+ *
32994
+ * A row that fails either pass is NOT terminal (D347): it is handed
32995
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
32996
+ * Only after the bound is exhausted is the device marked permanently
32997
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
32998
+ * `getStatus().error`.
32591
32999
  */
32592
33000
  async onRestoreDevices(savedDevices) {
32593
33001
  const restored = /* @__PURE__ */ new Set();
33002
+ const failures = [];
33003
+ const attemptRestore = async (saved) => {
33004
+ if (restored.has(saved.id)) return;
33005
+ const Class = this.deviceClasses[saved.type];
33006
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33007
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33008
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33009
+ restored.add(saved.id);
33010
+ };
32594
33011
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32595
33012
  const restoreOne = async (saved) => {
32596
- const Class = this.deviceClasses[saved.type];
32597
- if (!Class) {
33013
+ if (!this.deviceClasses[saved.type]) {
32598
33014
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32599
- tags: { stableId: saved.stableId },
33015
+ tags: {
33016
+ deviceId: saved.id,
33017
+ stableId: saved.stableId
33018
+ },
32600
33019
  meta: { type: saved.type }
32601
33020
  });
32602
33021
  return;
32603
33022
  }
32604
33023
  try {
32605
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32606
- restored.add(saved.id);
33024
+ await attemptRestore(saved);
32607
33025
  } catch (err) {
32608
- this.ctx.logger.warn("Failed to restore device", {
32609
- tags: { stableId: saved.stableId },
33026
+ const error = err instanceof Error ? err.message : String(err);
33027
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33028
+ tags: {
33029
+ deviceId: saved.id,
33030
+ stableId: saved.stableId
33031
+ },
32610
33032
  meta: {
32611
33033
  type: saved.type,
32612
- error: err instanceof Error ? err.message : String(err)
33034
+ attempt: 1,
33035
+ error
32613
33036
  }
32614
33037
  });
33038
+ failures.push({
33039
+ saved,
33040
+ error
33041
+ });
32615
33042
  }
32616
33043
  };
32617
33044
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33045
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32618
33046
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32619
33047
  for (const saved of childRows) {
32620
- const Class = this.deviceClasses[saved.type];
32621
- if (!Class) continue;
33048
+ if (!this.deviceClasses[saved.type]) continue;
32622
33049
  if (saved.parentDeviceId === null) continue;
32623
- if (!restored.has(saved.parentDeviceId)) continue;
32624
- try {
32625
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32626
- restored.add(saved.id);
32627
- } catch (err) {
32628
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33050
+ if (restored.has(saved.parentDeviceId)) {
33051
+ try {
33052
+ await attemptRestore(saved);
33053
+ } catch (err) {
33054
+ const error = err instanceof Error ? err.message : String(err);
33055
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33056
+ tags: {
33057
+ deviceId: saved.id,
33058
+ stableId: saved.stableId,
33059
+ parentDeviceId: saved.parentDeviceId
33060
+ },
33061
+ meta: {
33062
+ type: saved.type,
33063
+ attempt: 1,
33064
+ error
33065
+ }
33066
+ });
33067
+ failures.push({
33068
+ saved,
33069
+ error
33070
+ });
33071
+ }
33072
+ continue;
33073
+ }
33074
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33075
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32629
33076
  tags: {
33077
+ deviceId: saved.id,
32630
33078
  stableId: saved.stableId,
32631
33079
  parentDeviceId: saved.parentDeviceId
32632
33080
  },
32633
- meta: {
32634
- type: saved.type,
32635
- error: err instanceof Error ? err.message : String(err)
32636
- }
33081
+ meta: { type: saved.type }
32637
33082
  });
33083
+ failures.push({
33084
+ saved,
33085
+ error: `parent device ${saved.parentDeviceId} not restored`
33086
+ });
33087
+ continue;
32638
33088
  }
32639
33089
  }
33090
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33091
+ return {
33092
+ restoredCount: restored.size,
33093
+ failedCount: failures.length
33094
+ };
32640
33095
  }
32641
33096
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32642
33097
  toSummary(device) {
@@ -33145,6 +33600,12 @@ Object.freeze({
33145
33600
  addonId: null,
33146
33601
  access: "create"
33147
33602
  },
33603
+ "backup.cancel": {
33604
+ capName: "backup",
33605
+ capScope: "system",
33606
+ addonId: null,
33607
+ access: "create"
33608
+ },
33148
33609
  "backup.delete": {
33149
33610
  capName: "backup",
33150
33611
  capScope: "system",
@@ -33187,6 +33648,12 @@ Object.freeze({
33187
33648
  addonId: null,
33188
33649
  access: "view"
33189
33650
  },
33651
+ "backup.listRuns": {
33652
+ capName: "backup",
33653
+ capScope: "system",
33654
+ addonId: null,
33655
+ access: "view"
33656
+ },
33190
33657
  "backup.listSchedules": {
33191
33658
  capName: "backup",
33192
33659
  capScope: "system",
@@ -34387,6 +34854,12 @@ Object.freeze({
34387
34854
  addonId: null,
34388
34855
  access: "view"
34389
34856
  },
34857
+ "deviceProvider.reloadDevice": {
34858
+ capName: "device-provider",
34859
+ capScope: "system",
34860
+ addonId: null,
34861
+ access: "create"
34862
+ },
34390
34863
  "deviceProvider.start": {
34391
34864
  capName: "device-provider",
34392
34865
  capScope: "system",
@@ -37753,6 +38226,12 @@ Object.freeze({
37753
38226
  addonId: null,
37754
38227
  access: "create"
37755
38228
  },
38229
+ "streamBroker.forgetDeviceHardware": {
38230
+ capName: "stream-broker",
38231
+ capScope: "system",
38232
+ addonId: null,
38233
+ access: "delete"
38234
+ },
37756
38235
  "streamBroker.getAllRtspEntries": {
37757
38236
  capName: "stream-broker",
37758
38237
  capScope: "system",
@@ -40211,6 +40690,11 @@ Object.freeze({
40211
40690
  form: "single",
40212
40691
  optional: false
40213
40692
  }],
40693
+ "streamBroker.forgetDeviceHardware": [{
40694
+ name: "deviceId",
40695
+ form: "single",
40696
+ optional: false
40697
+ }],
40214
40698
  "streamBroker.getDeviceAudioMute": [{
40215
40699
  name: "deviceId",
40216
40700
  form: "single",
package/dist/addon.mjs CHANGED
@@ -10529,6 +10529,89 @@ var LocationStatSchema = object({
10529
10529
  fileCount: number(),
10530
10530
  present: boolean()
10531
10531
  });
10532
+ /** Lifecycle of a backup run. Terminal states: succeeded / failed / cancelled. */
10533
+ var BackupRunStateSchema = _enum([
10534
+ "queued",
10535
+ "running",
10536
+ "succeeded",
10537
+ "failed",
10538
+ "cancelled"
10539
+ ]);
10540
+ /**
10541
+ * Where a running backup currently is. `queued` before it starts,
10542
+ * `building` while the tar.gz is being staged, `uploading` during the
10543
+ * per-destination fan-out, `done` once terminal.
10544
+ */
10545
+ var BackupRunPhaseSchema = _enum([
10546
+ "queued",
10547
+ "building",
10548
+ "uploading",
10549
+ "done"
10550
+ ]);
10551
+ /**
10552
+ * Observable state of one backup run — readable WHILE it runs via
10553
+ * `backup.listRuns`. This is what makes the execution queue and
10554
+ * `backup.cancel` usable: the 2026-09-04 incident (two concurrent
10555
+ * multi-GB builds, staging 5.1 GB → 18 GB, load 62) was only
10556
+ * diagnosable with `du` because nothing reported that runs existed or
10557
+ * how large the staged archive had grown.
10558
+ */
10559
+ var BackupRunSchema = object({
10560
+ /** Stable run id — the handle `backup.cancel` takes. */
10561
+ id: string(),
10562
+ state: BackupRunStateSchema,
10563
+ phase: BackupRunPhaseSchema,
10564
+ /**
10565
+ * Resolved destination location ids. Empty while queued (targets are
10566
+ * resolved when the run starts, against the then-current policies).
10567
+ */
10568
+ destinationIds: array(string()).readonly(),
10569
+ label: string().optional(),
10570
+ /** ms-epoch when the run was submitted (trigger call / schedule fire). */
10571
+ requestedAt: number(),
10572
+ /** ms-epoch when the run left the queue and started building. */
10573
+ startedAt: number().optional(),
10574
+ /** ms-epoch when the run reached a terminal state. */
10575
+ finishedAt: number().optional(),
10576
+ /** Compressed bytes of the staging archive written so far. */
10577
+ stagedBytes: number(),
10578
+ /** Final staged archive size, once the build phase completes. */
10579
+ archiveSizeBytes: number().optional(),
10580
+ /** Bytes pushed to the destination currently uploading. */
10581
+ uploadedBytes: number(),
10582
+ /** Destinations where the archive fully landed (uploaded + indexed). */
10583
+ completedDestinationIds: array(string()).readonly(),
10584
+ /** Destinations that failed during the fan-out. */
10585
+ failedDestinationIds: array(string()).readonly(),
10586
+ /** Failure message when `state === 'failed'`. */
10587
+ error: string().optional(),
10588
+ /**
10589
+ * 1-based place in the execution queue — 1 = runs next. Present only
10590
+ * while `state === 'queued'`. Stamped by the orchestrator from the
10591
+ * queue's OWN pending order, never derived from timestamps, so the
10592
+ * UI cannot show an order the executor will not honour.
10593
+ */
10594
+ queuePosition: number().int().min(1).optional()
10595
+ });
10596
+ /**
10597
+ * Result of `backup.trigger`. The call still resolves when the run
10598
+ * terminates (compat with schedule-driven runs and the admin UI), but
10599
+ * it now names the run and says whether it had to WAIT: a trigger that
10600
+ * arrives while another run is in flight is enqueued (or joined onto
10601
+ * an identical already-queued run), never started concurrently.
10602
+ */
10603
+ var BackupTriggerResultSchema = object({
10604
+ /** The run this trigger mapped to — poll it via `listRuns`, stop it via `cancel`. */
10605
+ runId: string(),
10606
+ /** True when the run waited behind an in-flight run instead of starting immediately. */
10607
+ queued: boolean(),
10608
+ /** True when this trigger was coalesced onto an identical already-queued run. */
10609
+ joined: boolean(),
10610
+ /** True when the run was cancelled before completing every destination. */
10611
+ cancelled: boolean(),
10612
+ /** One entry per destination the archive landed at (partial on cancel). */
10613
+ entries: array(BackupEntrySchema).readonly()
10614
+ });
10532
10615
  /**
10533
10616
  * A backup schedule — the N:M "entry" that binds one cron cadence to a
10534
10617
  * SET of destination locations. Supersedes the per-location cron on
@@ -10576,7 +10659,10 @@ method(_void(), array(BackupDestinationInfoSchema).readonly(), { auth: "admin" }
10576
10659
  * retention (manual runs).
10577
10660
  */
10578
10661
  retentionCount: number().int().min(1).max(1e3).optional()
10579
- }).optional(), array(BackupEntrySchema).readonly(), {
10662
+ }).optional(), BackupTriggerResultSchema, {
10663
+ kind: "mutation",
10664
+ auth: "admin"
10665
+ }), method(_void(), array(BackupRunSchema).readonly(), { auth: "admin" }), method(object({ runId: string() }), object({ cancelled: boolean() }), {
10580
10666
  kind: "mutation",
10581
10667
  auth: "admin"
10582
10668
  }), method(_void(), array(BackupEntrySchema).readonly(), { auth: "admin" }), method(_void(), array(LocationStatSchema).readonly(), { auth: "admin" }), method(object({
@@ -11293,6 +11379,14 @@ method(object({
11293
11379
  }), object({ success: literal(true) }), {
11294
11380
  kind: "mutation",
11295
11381
  auth: "admin"
11382
+ }), method(object({ deviceId: number().int().nonnegative() }), object({
11383
+ derivedStreamsDeleted: array(string()).readonly(),
11384
+ assignmentsPurged: boolean(),
11385
+ probeSnapshotsDropped: number().int().nonnegative(),
11386
+ rtspTokenRowsDeleted: number().int().nonnegative()
11387
+ }), {
11388
+ kind: "mutation",
11389
+ auth: "admin"
11296
11390
  }), method(object({
11297
11391
  deviceId: number(),
11298
11392
  /** Absent = the LOWEST assigned profile — a notification attachment is
@@ -12720,6 +12814,35 @@ var deviceProviderCapability = {
12720
12814
  name: string(),
12721
12815
  type: string()
12722
12816
  }))),
12817
+ /**
12818
+ * Tear down and reconstruct ONE device in place from its persisted rows —
12819
+ * touching no other device this provider owns.
12820
+ *
12821
+ * The primitive `deviceManager.migrateDevice` uses to flush the two
12822
+ * migrated numbers: after `swapIds` the runner's live instance still
12823
+ * carries the PRE-swap numeric id (baked into the object, its native-cap
12824
+ * registrations and its log tags), and a live object cannot be renumbered.
12825
+ * Before this method the only flush was restarting the whole owning addon
12826
+ * — which took every camera the provider owns down with it (28 devices
12827
+ * for one migrated camera, measured 2026-09-04, and the morning of the
12828
+ * same day ~27 devices' native caps did not come back on their own).
12829
+ *
12830
+ * Keyed by `stableId`, deliberately: the numeric id is exactly the thing
12831
+ * that changes. The reply carries the id the device answers on NOW.
12832
+ * Implemented once in `BaseDeviceProvider` — decommission the live
12833
+ * instance (if any), then re-create from the persisted row: the same
12834
+ * teardown/rehydrate pair every graceful shutdown + boot already uses.
12835
+ * An RPC, never an event: a dropped event would leave the runner writing
12836
+ * against the wrong camera (D8).
12837
+ *
12838
+ * Construction can dial hardware, and the migrated source is
12839
+ * characteristically dead — the timeout covers a full activate window
12840
+ * rather than the 60 s default.
12841
+ */
12842
+ reloadDevice: method(object({ stableId: string() }), object({ deviceId: number() }), {
12843
+ kind: "mutation",
12844
+ timeoutMs: 3 * 6e4
12845
+ }),
12723
12846
  supportsDiscovery: method(object({}), boolean()),
12724
12847
  /**
12725
12848
  * Run a network scan. `params` carries optional provider-specific scan
@@ -13047,7 +13170,8 @@ method(object({
13047
13170
  targetId: number()
13048
13171
  }), MigrateDeviceResultSchema, {
13049
13172
  kind: "mutation",
13050
- auth: "admin"
13173
+ auth: "admin",
13174
+ timeoutMs: 12 * 6e4
13051
13175
  }), method(DeviceRegisterPayloadSchema, _void(), { kind: "mutation" }), method(DeviceRemovePayloadSchema, _void(), { kind: "mutation" }), method(DevicePersistConfigPayloadSchema, _void(), { kind: "mutation" }), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), record(string(), unknown())), method(object({ deviceId: number() }), DeviceMetaSchema.nullable()), method(object({
13052
13176
  deviceId: number(),
13053
13177
  name: string()
@@ -32403,6 +32527,147 @@ var BaseDevice$1 = class {
32403
32527
  }
32404
32528
  };
32405
32529
  /**
32530
+ * Delays before retry rounds 1..N — the round count IS the bound.
32531
+ * 10 s catches "the hub was busy for a moment"; the full schedule
32532
+ * (10 + 30 + 90 s of waiting, plus up to one 60 s transport timeout
32533
+ * per attempt) covers a device-manager lock held for minutes — the
32534
+ * 2026-09-04 outage's migration hold was ~3.5 min.
32535
+ */
32536
+ var DEVICE_RESTORE_RETRY_DELAYS_MS = [
32537
+ 1e4,
32538
+ 3e4,
32539
+ 9e4
32540
+ ];
32541
+ /** Abortable sleep — resolves early (never rejects) on abort. */
32542
+ function sleep$1(ms, signal) {
32543
+ return new Promise((resolve) => {
32544
+ if (signal.aborted) {
32545
+ resolve();
32546
+ return;
32547
+ }
32548
+ const onAbort = () => {
32549
+ clearTimeout(timer);
32550
+ resolve();
32551
+ };
32552
+ const timer = setTimeout(() => {
32553
+ signal.removeEventListener("abort", onAbort);
32554
+ resolve();
32555
+ }, ms);
32556
+ timer.unref?.();
32557
+ signal.addEventListener("abort", onAbort, { once: true });
32558
+ });
32559
+ }
32560
+ /** Drain `items` through at most `width` concurrent lanes. `fn` must
32561
+ * not reject (callers wrap their own try/catch). */
32562
+ async function runWithConcurrency(items, width, fn) {
32563
+ const queue = [...items];
32564
+ const laneCount = Math.max(1, Math.min(width, queue.length));
32565
+ const lane = async () => {
32566
+ for (;;) {
32567
+ const item = queue.shift();
32568
+ if (item === void 0) return;
32569
+ await fn(item);
32570
+ }
32571
+ };
32572
+ await Promise.all(Array.from({ length: laneCount }, lane));
32573
+ }
32574
+ var DeviceRestoreRetryScheduler = class {
32575
+ #logger;
32576
+ #attempt;
32577
+ #onPermanentFailure;
32578
+ #delaysMs;
32579
+ #concurrency;
32580
+ #now;
32581
+ #abort = new AbortController();
32582
+ constructor(options) {
32583
+ this.#logger = options.logger;
32584
+ this.#attempt = options.attempt;
32585
+ this.#onPermanentFailure = options.onPermanentFailure;
32586
+ this.#delaysMs = options.delaysMs ?? DEVICE_RESTORE_RETRY_DELAYS_MS;
32587
+ this.#concurrency = options.concurrency ?? 4;
32588
+ this.#now = options.now ?? Date.now;
32589
+ }
32590
+ /** Stop retrying (shutdown). Pending entries are NOT marked
32591
+ * permanently failed — the next boot restores them from disk. */
32592
+ cancel() {
32593
+ this.#abort.abort();
32594
+ }
32595
+ /**
32596
+ * Run the bounded retry rounds. Resolves when every entry has either
32597
+ * restored, been marked permanently failed, or the scheduler was
32598
+ * cancelled. Never rejects.
32599
+ */
32600
+ async run(initialFailures) {
32601
+ let pending = initialFailures.map((failure) => ({
32602
+ saved: failure.saved,
32603
+ lastError: failure.error,
32604
+ attempts: 1
32605
+ }));
32606
+ for (let round = 0; round < this.#delaysMs.length; round += 1) {
32607
+ if (pending.length === 0 || this.#abort.signal.aborted) break;
32608
+ await sleep$1(this.#delaysMs[round] ?? 0, this.#abort.signal);
32609
+ if (this.#abort.signal.aborted) break;
32610
+ pending = await this.#runRound(pending, round);
32611
+ }
32612
+ if (this.#abort.signal.aborted) return [];
32613
+ const terminal = pending.map((entry) => ({
32614
+ deviceId: entry.saved.id,
32615
+ stableId: entry.saved.stableId,
32616
+ type: String(entry.saved.type),
32617
+ attempts: entry.attempts,
32618
+ lastError: entry.lastError,
32619
+ failedAt: this.#now()
32620
+ }));
32621
+ for (const failure of terminal) this.#onPermanentFailure(failure);
32622
+ return terminal;
32623
+ }
32624
+ /** One retry round: parents first (phase 0), then hub-adopted
32625
+ * children (phase 1) — a child's attempt depends on its parent
32626
+ * having landed, exactly like the initial two-pass restore. */
32627
+ async #runRound(pending, round) {
32628
+ const next = [];
32629
+ const parents = pending.filter((entry) => entry.saved.parentDeviceId === null);
32630
+ const children = pending.filter((entry) => entry.saved.parentDeviceId !== null);
32631
+ for (const phase of [parents, children]) await runWithConcurrency(phase, this.#concurrency, async (entry) => {
32632
+ if (this.#abort.signal.aborted) {
32633
+ next.push(entry);
32634
+ return;
32635
+ }
32636
+ const attemptNo = entry.attempts + 1;
32637
+ try {
32638
+ await this.#attempt(entry.saved);
32639
+ this.#logger.info("Device restored on retry", {
32640
+ tags: {
32641
+ deviceId: entry.saved.id,
32642
+ stableId: entry.saved.stableId
32643
+ },
32644
+ meta: { attempt: attemptNo }
32645
+ });
32646
+ } catch (err) {
32647
+ const lastError = err instanceof Error ? err.message : String(err);
32648
+ const remainingRetries = this.#delaysMs.length - (round + 1);
32649
+ this.#logger.warn("Device restore retry failed", {
32650
+ tags: {
32651
+ deviceId: entry.saved.id,
32652
+ stableId: entry.saved.stableId
32653
+ },
32654
+ meta: {
32655
+ attempt: attemptNo,
32656
+ remainingRetries,
32657
+ error: lastError
32658
+ }
32659
+ });
32660
+ next.push({
32661
+ saved: entry.saved,
32662
+ lastError,
32663
+ attempts: attemptNo
32664
+ });
32665
+ }
32666
+ });
32667
+ return next;
32668
+ }
32669
+ };
32670
+ /**
32406
32671
  * Convert an IDevice to the flat DeviceSummary shape expected by the
32407
32672
  * device-provider cap router. Shared across all providers.
32408
32673
  */
@@ -32451,6 +32716,7 @@ var BaseDeviceProvider = class extends BaseAddon {
32451
32716
  }];
32452
32717
  }
32453
32718
  async onShutdown() {
32719
+ this.cancelRestoreRetries();
32454
32720
  const devices = await this.ctx.kernel.devices?.getAll() ?? [];
32455
32721
  for (const device of devices) try {
32456
32722
  await this.ctx.kernel.devices?.decommission(device.id);
@@ -32468,9 +32734,16 @@ var BaseDeviceProvider = class extends BaseAddon {
32468
32734
  async start() {}
32469
32735
  async stop() {}
32470
32736
  async getStatus() {
32737
+ const all = await this.ctx.kernel.devices?.getAll() ?? [];
32738
+ const summary = this.restoreFailureSummary();
32739
+ if (summary === null) return {
32740
+ connected: true,
32741
+ deviceCount: all.length
32742
+ };
32471
32743
  return {
32472
32744
  connected: true,
32473
- deviceCount: (await this.ctx.kernel.devices?.getAll() ?? []).length
32745
+ deviceCount: all.length,
32746
+ error: summary
32474
32747
  };
32475
32748
  }
32476
32749
  async getDevices() {
@@ -32560,8 +32833,137 @@ var BaseDeviceProvider = class extends BaseAddon {
32560
32833
  };
32561
32834
  }
32562
32835
  async restoreDevices(savedDevices) {
32563
- await this.onRestoreDevices(savedDevices);
32564
- if (savedDevices.length > 0) this.ctx.logger.info(`Restored ${savedDevices.length} ${this.providerName} device(s)`);
32836
+ const report = await this.onRestoreDevices(savedDevices);
32837
+ if (savedDevices.length === 0) return;
32838
+ if (report && report.failedCount > 0) {
32839
+ this.ctx.logger.warn(`Restored ${report.restoredCount}/${savedDevices.length} ${this.providerName} device(s) — ${report.failedCount} failed, bounded retry scheduled`);
32840
+ return;
32841
+ }
32842
+ const restoredCount = report ? report.restoredCount : savedDevices.length;
32843
+ this.ctx.logger.info(`Restored ${restoredCount} ${this.providerName} device(s)`);
32844
+ }
32845
+ /** Retry schedule. Overridable (tests use millisecond delays). */
32846
+ restoreRetryDelaysMs = DEVICE_RESTORE_RETRY_DELAYS_MS;
32847
+ /** Retry lane width. See `device-restore-retry.ts` for why retries
32848
+ * never re-stampede full-width while the initial pass does (D167). */
32849
+ restoreRetryConcurrency = 4;
32850
+ _restoreRetryScheduler = null;
32851
+ _restoreRetryCompletion = null;
32852
+ _permanentRestoreFailures = /* @__PURE__ */ new Map();
32853
+ /** Settles when the background retry rounds finish (or `null` when
32854
+ * nothing failed). Exposed for tests and subclass diagnostics —
32855
+ * boot NEVER awaits this: the runner's post-init handshake goes out
32856
+ * with the devices that restored, and a late success is announced
32857
+ * through the `native-cap-change` → `updateCaps` path. */
32858
+ get restoreRetryCompletion() {
32859
+ return this._restoreRetryCompletion;
32860
+ }
32861
+ /** Devices that exhausted the retry bound this process lifetime. */
32862
+ get permanentRestoreFailures() {
32863
+ return [...this._permanentRestoreFailures.values()];
32864
+ }
32865
+ /** One-line operator-facing summary for `getStatus().error`, or
32866
+ * `null` when every device restored. */
32867
+ restoreFailureSummary() {
32868
+ if (this._permanentRestoreFailures.size === 0) return null;
32869
+ const ids = [...this._permanentRestoreFailures.keys()].join(", ");
32870
+ return `${this._permanentRestoreFailures.size} device(s) permanently failed restore (deviceIds: ${ids}) — restart the ${this.providerName} provider to retry`;
32871
+ }
32872
+ cancelRestoreRetries() {
32873
+ this._restoreRetryScheduler?.cancel();
32874
+ this._restoreRetryScheduler = null;
32875
+ }
32876
+ recordPermanentRestoreFailure(failure) {
32877
+ this._permanentRestoreFailures.set(failure.deviceId, failure);
32878
+ this.ctx.logger.error("Device restore permanently failed — its capabilities will not register until the provider restarts", {
32879
+ tags: {
32880
+ deviceId: failure.deviceId,
32881
+ stableId: failure.stableId
32882
+ },
32883
+ meta: {
32884
+ type: failure.type,
32885
+ attempts: failure.attempts,
32886
+ error: failure.lastError
32887
+ }
32888
+ });
32889
+ }
32890
+ scheduleRestoreRetries(failures, attempt) {
32891
+ const scheduler = new DeviceRestoreRetryScheduler({
32892
+ logger: this.ctx.logger,
32893
+ delaysMs: this.restoreRetryDelaysMs,
32894
+ concurrency: this.restoreRetryConcurrency,
32895
+ attempt,
32896
+ onPermanentFailure: (failure) => this.recordPermanentRestoreFailure(failure)
32897
+ });
32898
+ this._restoreRetryScheduler = scheduler;
32899
+ this._restoreRetryCompletion = scheduler.run(failures).then(() => void 0).catch((err) => {
32900
+ this.ctx.logger.error("Restore retry scheduler crashed", { meta: { error: err instanceof Error ? err.message : String(err) } });
32901
+ });
32902
+ }
32903
+ /**
32904
+ * Tear down and reconstruct ONE device from its persisted rows — the
32905
+ * `deviceProvider.reloadDevice` cap method. Persistence is never touched,
32906
+ * and no other device this provider owns is disturbed.
32907
+ *
32908
+ * Keyed by `stableId` because the caller's whole reason to be here is that
32909
+ * the NUMERIC id changed (`deviceManager.migrateDevice` swapped it): the
32910
+ * fresh instance resolves its id through `allocateDeviceId`, which returns
32911
+ * whatever number the row carries NOW. The teardown is `decommission` —
32912
+ * exactly what a graceful shutdown runs per device (fires `removeDevice()`,
32913
+ * unregisters native caps, drops the registry entry) — and the rebuild is
32914
+ * the boot restore's own `create()` path, including its pass 2: first-class
32915
+ * children (hub-adopted cameras under an NVR) are decommissioned with the
32916
+ * parent by the cascade and must be re-created explicitly, because only
32917
+ * accessory children come back through `getAccessoryChildren()`.
32918
+ *
32919
+ * Reloading an accessory child directly is refused (no device class) —
32920
+ * reload its parent instead.
32921
+ */
32922
+ async reloadDevice(input) {
32923
+ const { stableId } = input;
32924
+ const devices = this.ctx.kernel.devices;
32925
+ if (!devices) throw new Error(`${this.providerName}: kernel.devices unavailable — cannot reload`);
32926
+ const live = (await devices.getAll()).find((d) => d.stableId === stableId);
32927
+ if (live) await devices.decommission(live.id);
32928
+ const { id } = await this.ctx.api.deviceManager.allocateDeviceId.mutate({
32929
+ addonId: this.addonId,
32930
+ stableId
32931
+ });
32932
+ const meta = await this.ctx.api.deviceManager.loadMeta.query({ deviceId: id });
32933
+ if (meta === null) throw new Error(`${this.providerName}: no persisted meta for "${stableId}" (id ${id}) — cannot reload`);
32934
+ const deviceType = Object.values(DeviceType).find((t) => t === meta.type);
32935
+ const Class = deviceType !== void 0 ? this.deviceClasses[deviceType] : void 0;
32936
+ if (!Class) throw new Error(`${this.providerName}: no device class for type "${meta.type}" — "${stableId}" is an accessory child; reload its parent instead`);
32937
+ await devices.create(stableId, Class, {}, meta.parentDeviceId ?? null);
32938
+ const rows = await this.ctx.api.deviceManager.listPersistedByAddon.query({ addonId: this.addonId });
32939
+ for (const row of rows) {
32940
+ if (row.parentDeviceId !== id) continue;
32941
+ const childType = Object.values(DeviceType).find((t) => t === row.type);
32942
+ const ChildClass = childType !== void 0 ? this.deviceClasses[childType] : void 0;
32943
+ if (!ChildClass) continue;
32944
+ try {
32945
+ await devices.create(row.stableId, ChildClass, {}, id);
32946
+ } catch (err) {
32947
+ this.ctx.logger.warn("reloadDevice: failed to re-create first-class child", {
32948
+ tags: {
32949
+ deviceId: row.id,
32950
+ stableId: row.stableId
32951
+ },
32952
+ meta: {
32953
+ parentDeviceId: id,
32954
+ error: err instanceof Error ? err.message : String(err)
32955
+ }
32956
+ });
32957
+ }
32958
+ }
32959
+ this.ctx.logger.info("device reloaded in place from persisted rows", {
32960
+ tags: { deviceId: id },
32961
+ meta: {
32962
+ stableId,
32963
+ type: meta.type
32964
+ }
32965
+ });
32966
+ return { deviceId: id };
32565
32967
  }
32566
32968
  /**
32567
32969
  * Restore devices from persisted state. Two-pass:
@@ -32587,55 +32989,108 @@ var BaseDeviceProvider = class extends BaseAddon {
32587
32989
  * accessory-spawn flow handles via the parent's
32588
32990
  * `getAccessoryChildren()`. Override only when the default doesn't
32589
32991
  * fit.
32992
+ *
32993
+ * A row that fails either pass is NOT terminal (D347): it is handed
32994
+ * to a bounded background retry (`DeviceRestoreRetryScheduler`).
32995
+ * Only after the bound is exhausted is the device marked permanently
32996
+ * failed — logged at ERROR with `tags.deviceId` and surfaced via
32997
+ * `getStatus().error`.
32590
32998
  */
32591
32999
  async onRestoreDevices(savedDevices) {
32592
33000
  const restored = /* @__PURE__ */ new Set();
33001
+ const failures = [];
33002
+ const attemptRestore = async (saved) => {
33003
+ if (restored.has(saved.id)) return;
33004
+ const Class = this.deviceClasses[saved.type];
33005
+ if (!Class) throw new Error(`no device class registered for type "${saved.type}"`);
33006
+ if (saved.parentDeviceId !== null && !restored.has(saved.parentDeviceId)) throw new Error(`parent device ${saved.parentDeviceId} not restored`);
33007
+ await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
33008
+ restored.add(saved.id);
33009
+ };
32593
33010
  const topLevel = savedDevices.filter((saved) => saved.parentDeviceId === null);
32594
33011
  const restoreOne = async (saved) => {
32595
- const Class = this.deviceClasses[saved.type];
32596
- if (!Class) {
33012
+ if (!this.deviceClasses[saved.type]) {
32597
33013
  this.ctx.logger.warn("No device class registered for restored type — skipping", {
32598
- tags: { stableId: saved.stableId },
33014
+ tags: {
33015
+ deviceId: saved.id,
33016
+ stableId: saved.stableId
33017
+ },
32599
33018
  meta: { type: saved.type }
32600
33019
  });
32601
33020
  return;
32602
33021
  }
32603
33022
  try {
32604
- await this.ctx.kernel.devices.create(saved.stableId, Class, {});
32605
- restored.add(saved.id);
33023
+ await attemptRestore(saved);
32606
33024
  } catch (err) {
32607
- this.ctx.logger.warn("Failed to restore device", {
32608
- tags: { stableId: saved.stableId },
33025
+ const error = err instanceof Error ? err.message : String(err);
33026
+ this.ctx.logger.warn("Failed to restore device — bounded retry scheduled", {
33027
+ tags: {
33028
+ deviceId: saved.id,
33029
+ stableId: saved.stableId
33030
+ },
32609
33031
  meta: {
32610
33032
  type: saved.type,
32611
- error: err instanceof Error ? err.message : String(err)
33033
+ attempt: 1,
33034
+ error
32612
33035
  }
32613
33036
  });
33037
+ failures.push({
33038
+ saved,
33039
+ error
33040
+ });
32614
33041
  }
32615
33042
  };
32616
33043
  await Promise.all(topLevel.map((saved) => restoreOne(saved)));
33044
+ const failedTopLevelIds = new Set(failures.map((failure) => failure.saved.id));
32617
33045
  const childRows = savedDevices.filter((s) => s.parentDeviceId !== null);
32618
33046
  for (const saved of childRows) {
32619
- const Class = this.deviceClasses[saved.type];
32620
- if (!Class) continue;
33047
+ if (!this.deviceClasses[saved.type]) continue;
32621
33048
  if (saved.parentDeviceId === null) continue;
32622
- if (!restored.has(saved.parentDeviceId)) continue;
32623
- try {
32624
- await this.ctx.kernel.devices.create(saved.stableId, Class, {}, saved.parentDeviceId);
32625
- restored.add(saved.id);
32626
- } catch (err) {
32627
- this.ctx.logger.warn("Failed to restore hub-adopted child", {
33049
+ if (restored.has(saved.parentDeviceId)) {
33050
+ try {
33051
+ await attemptRestore(saved);
33052
+ } catch (err) {
33053
+ const error = err instanceof Error ? err.message : String(err);
33054
+ this.ctx.logger.warn("Failed to restore hub-adopted child — bounded retry scheduled", {
33055
+ tags: {
33056
+ deviceId: saved.id,
33057
+ stableId: saved.stableId,
33058
+ parentDeviceId: saved.parentDeviceId
33059
+ },
33060
+ meta: {
33061
+ type: saved.type,
33062
+ attempt: 1,
33063
+ error
33064
+ }
33065
+ });
33066
+ failures.push({
33067
+ saved,
33068
+ error
33069
+ });
33070
+ }
33071
+ continue;
33072
+ }
33073
+ if (failedTopLevelIds.has(saved.parentDeviceId)) {
33074
+ this.ctx.logger.warn("Hub-adopted child deferred — parent failed initial restore", {
32628
33075
  tags: {
33076
+ deviceId: saved.id,
32629
33077
  stableId: saved.stableId,
32630
33078
  parentDeviceId: saved.parentDeviceId
32631
33079
  },
32632
- meta: {
32633
- type: saved.type,
32634
- error: err instanceof Error ? err.message : String(err)
32635
- }
33080
+ meta: { type: saved.type }
32636
33081
  });
33082
+ failures.push({
33083
+ saved,
33084
+ error: `parent device ${saved.parentDeviceId} not restored`
33085
+ });
33086
+ continue;
32637
33087
  }
32638
33088
  }
33089
+ if (failures.length > 0) this.scheduleRestoreRetries(failures, attemptRestore);
33090
+ return {
33091
+ restoredCount: restored.size,
33092
+ failedCount: failures.length
33093
+ };
32639
33094
  }
32640
33095
  /** Convert an IDevice to the flat DeviceSummary for the cap router. */
32641
33096
  toSummary(device) {
@@ -33144,6 +33599,12 @@ Object.freeze({
33144
33599
  addonId: null,
33145
33600
  access: "create"
33146
33601
  },
33602
+ "backup.cancel": {
33603
+ capName: "backup",
33604
+ capScope: "system",
33605
+ addonId: null,
33606
+ access: "create"
33607
+ },
33147
33608
  "backup.delete": {
33148
33609
  capName: "backup",
33149
33610
  capScope: "system",
@@ -33186,6 +33647,12 @@ Object.freeze({
33186
33647
  addonId: null,
33187
33648
  access: "view"
33188
33649
  },
33650
+ "backup.listRuns": {
33651
+ capName: "backup",
33652
+ capScope: "system",
33653
+ addonId: null,
33654
+ access: "view"
33655
+ },
33189
33656
  "backup.listSchedules": {
33190
33657
  capName: "backup",
33191
33658
  capScope: "system",
@@ -34386,6 +34853,12 @@ Object.freeze({
34386
34853
  addonId: null,
34387
34854
  access: "view"
34388
34855
  },
34856
+ "deviceProvider.reloadDevice": {
34857
+ capName: "device-provider",
34858
+ capScope: "system",
34859
+ addonId: null,
34860
+ access: "create"
34861
+ },
34389
34862
  "deviceProvider.start": {
34390
34863
  capName: "device-provider",
34391
34864
  capScope: "system",
@@ -37752,6 +38225,12 @@ Object.freeze({
37752
38225
  addonId: null,
37753
38226
  access: "create"
37754
38227
  },
38228
+ "streamBroker.forgetDeviceHardware": {
38229
+ capName: "stream-broker",
38230
+ capScope: "system",
38231
+ addonId: null,
38232
+ access: "delete"
38233
+ },
37755
38234
  "streamBroker.getAllRtspEntries": {
37756
38235
  capName: "stream-broker",
37757
38236
  capScope: "system",
@@ -40210,6 +40689,11 @@ Object.freeze({
40210
40689
  form: "single",
40211
40690
  optional: false
40212
40691
  }],
40692
+ "streamBroker.forgetDeviceHardware": [{
40693
+ name: "deviceId",
40694
+ form: "single",
40695
+ optional: false
40696
+ }],
40213
40697
  "streamBroker.getDeviceAudioMute": [{
40214
40698
  name: "deviceId",
40215
40699
  form: "single",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-provider-gree",
3
- "version": "0.2.57",
3
+ "version": "0.2.59",
4
4
  "description": "Gree air-conditioner device-provider addon for CamStack — wraps the @apocaliss92/nodegree local-UDP client (LAN discovery + AES control), exposing climate-control and fan-control",
5
5
  "keywords": [
6
6
  "camstack",