@spooky-sync/core 0.0.1-canary.161 → 0.0.1-canary.162

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/sp00ky.ts CHANGED
@@ -11,7 +11,12 @@ import type {
11
11
  SyncHealth,
12
12
  StorageHealth,
13
13
  } from './types';
14
- import { LocalMigrator, RemoteDatabaseService, createLocalEngine } from './services/database/index';
14
+ import {
15
+ ConnectionSupervisor,
16
+ LocalMigrator,
17
+ RemoteDatabaseService,
18
+ createLocalEngine,
19
+ } from './services/database/index';
15
20
  import type { LocalStore } from './services/database/index';
16
21
  import { StaleEpochError } from './services/database/index';
17
22
  import type { UpEvent } from './modules/sync/index';
@@ -148,6 +153,7 @@ function writeBootBucketHint(bucketId: string): void {
148
153
  export class Sp00kyClient<S extends SchemaStructure> {
149
154
  private local: LocalStore;
150
155
  private remote: RemoteDatabaseService;
156
+ private connectionSupervisor: ConnectionSupervisor;
151
157
  private persistenceClient: PersistenceClient;
152
158
 
153
159
  private migrator: LocalMigrator;
@@ -267,6 +273,10 @@ export class Sp00kyClient<S extends SchemaStructure> {
267
273
  shared: tabsSupport.supported,
268
274
  });
269
275
  this.remote = new RemoteDatabaseService(this.config.database, logger);
276
+ // Owns socket liveness for the life of the client: re-opens the connection
277
+ // when the SDK's own reconnect gives up, and heartbeats to catch a socket
278
+ // that died without ever firing a `close` event.
279
+ this.connectionSupervisor = new ConnectionSupervisor(this.remote, logger);
270
280
 
271
281
  if (config.persistenceClient === 'surrealdb') {
272
282
  this.persistenceClient = new SurrealDBPersistenceClient(this.local, logger);
@@ -343,6 +353,9 @@ export class Sp00kyClient<S extends SchemaStructure> {
343
353
  this.config.syncHealth === false
344
354
  ? 0
345
355
  : (this.config.syncHealth?.degradeAfterConsecutiveFailures ?? 3),
356
+ pushTimeoutMs: this.config.pushTimeoutMs,
357
+ // Read-only: sync mirrors the transport state into `SyncHealth`.
358
+ connectionSupervisor: this.connectionSupervisor,
346
359
  }
347
360
  );
348
361
 
@@ -616,6 +629,10 @@ export class Sp00kyClient<S extends SchemaStructure> {
616
629
  }
617
630
 
618
631
  await this.remote.connect();
632
+ // Start supervising only after the first connect succeeds, so a boot-time
633
+ // failure surfaces as a thrown init() rather than being silently absorbed
634
+ // into a background retry loop.
635
+ this.connectionSupervisor.start();
619
636
  this.logger.debug(
620
637
  { Category: 'sp00ky-client::Sp00kyClient::init' },
621
638
  'Remote database connected'
@@ -638,6 +655,16 @@ export class Sp00kyClient<S extends SchemaStructure> {
638
655
  // for the same user don't collide on shared `_00_query` rows. The same
639
656
  // session id is the `session_id` key in `_00_cursor` rows, so the
640
657
  // CrdtManager needs it too.
658
+ //
659
+ // Deliberately NOT refreshed on reconnect, even though `session::id()`
660
+ // does change with every WebSocket session. The salt keys `_00_query`
661
+ // rows and local cache entries, so rotating it would invalidate every
662
+ // query hash and force a full re-register plus a cache-key migration on
663
+ // each blip. Those rows stay valid because the TTL heartbeat keeps them
664
+ // alive, not because the session that created them is still open — so a
665
+ // stable salt across reconnects is the correct behavior. Auth flips are
666
+ // the one case that must rotate it (below): a sign-in is a different
667
+ // principal, not the same session on a new socket.
641
668
  const sessionId = await this.fetchSessionId();
642
669
  await this.dataModule.init(sessionId);
643
670
  this.crdtManager.setSessionId(sessionId);
@@ -868,9 +895,13 @@ export class Sp00kyClient<S extends SchemaStructure> {
868
895
  }
869
896
 
870
897
  async close() {
898
+ // Before anything else: a live supervisor would read the intentional
899
+ // `remote.close()` below as a failure and immediately reconnect.
900
+ this.connectionSupervisor.dispose();
871
901
  await this.featureFlags.closeAll();
872
902
  await this.appReleases.closeAll();
873
903
  this.crdtManager.closeAll();
904
+ this.crdtManager.dispose();
874
905
  // Leaving the broker first hands leadership to another tab (and releases
875
906
  // the OPFS handles via the worker shutdown) before the store closes.
876
907
  if (this.tabsCoordinator) await this.tabsCoordinator.stop();
package/src/types.ts CHANGED
@@ -124,6 +124,22 @@ export interface Sp00kyConfig<S extends SchemaStructure> {
124
124
  * multi-hop path (escape hatch while the worker-side path beds in).
125
125
  */
126
126
  workerSelect?: boolean;
127
+ /**
128
+ * WebSocket reconnect + liveness tuning. All fields optional; the defaults
129
+ * keep the connection alive indefinitely without configuration. See
130
+ * {@link ReconnectConfig}.
131
+ */
132
+ reconnect?: ReconnectConfig;
133
+ /**
134
+ * Deadline (ms) for every remote RPC. Remote queries are serialized through
135
+ * a single promise chain, so one call that never settles (half-open socket:
136
+ * the WebSocket looks open, the peer is gone, no `close` event fires) would
137
+ * otherwise wedge ALL later remote traffic behind it — including the sync
138
+ * poll's own health probe, leaving health pinned at `healthy` with no
139
+ * banner and no self-heal. The deadline turns that into an ordinary network
140
+ * failure the queue retries. `0` disables. Defaults to `60_000`.
141
+ */
142
+ queryTimeoutMs?: number;
127
143
  };
128
144
  /** The schema definition. */
129
145
  schema: S;
@@ -254,6 +270,15 @@ export interface Sp00kyConfig<S extends SchemaStructure> {
254
270
  * (or `degradeAfterConsecutiveFailures: 0`) to never report degraded.
255
271
  */
256
272
  syncHealth?: SyncHealthConfig | false;
273
+ /**
274
+ * Deadline (ms) for a single outgoing mutation push. Tighter than
275
+ * {@link Sp00kyConfig.database.queryTimeoutMs} because the up-queue drains
276
+ * one mutation at a time behind an `isSyncingUp` flag: a push that never
277
+ * settles stops every later mutation for the session, with no retry and no
278
+ * error. On expiry the push is treated as a network failure and re-queued.
279
+ * `0` disables. Defaults to `30_000`.
280
+ */
281
+ pushTimeoutMs?: number;
257
282
  }
258
283
 
259
284
  /** Tunables for sync-health reporting. See {@link Sp00kyConfig.syncHealth}. */
@@ -267,6 +292,55 @@ export interface SyncHealthConfig {
267
292
  degradeAfterConsecutiveFailures?: number;
268
293
  }
269
294
 
295
+ /**
296
+ * Tunables for WebSocket reconnect and liveness detection. See
297
+ * {@link Sp00kyConfig.database.reconnect}.
298
+ *
299
+ * Two independent mechanisms cooperate here. The SurrealDB SDK reconnects on
300
+ * its own after a socket `close` (`attempts` / `retryDelayMax`), and a
301
+ * supervisor above it re-opens the connection from scratch whenever the SDK
302
+ * gives up or its post-reconnect handshake fails — the SDK terminates the
303
+ * engine permanently in that case, so a supervisor is required, not optional.
304
+ * The heartbeat covers the third case: a socket that never closes at all.
305
+ */
306
+ export interface ReconnectConfig {
307
+ /**
308
+ * SDK reconnect attempts after a socket close. `-1` retries forever.
309
+ * Defaults to `-1` (the SDK's own default is `5`, which caps recovery at a
310
+ * ~62s outage and then gives up for the life of the page).
311
+ */
312
+ attempts?: number;
313
+ /** Cap on the SDK's exponential backoff delay. Defaults to `15_000`. */
314
+ retryDelayMax?: number;
315
+ /**
316
+ * Cadence of the application-level liveness probe (`RETURN true`) that
317
+ * detects a half-open socket the transport never reports as closed.
318
+ * `0` disables the heartbeat. Defaults to `20_000`.
319
+ */
320
+ heartbeatIntervalMs?: number;
321
+ /**
322
+ * Deadline for a heartbeat response. Exceeding it means the socket is dead
323
+ * regardless of what its `readyState` claims, so the connection is torn down
324
+ * and rebuilt. Defaults to `10_000`.
325
+ */
326
+ heartbeatTimeoutMs?: number;
327
+ /**
328
+ * Cap on the supervisor's own backoff between `connect()` retries once the
329
+ * SDK has given up. Defaults to `15_000`.
330
+ */
331
+ superviseRetryDelayMaxMs?: number;
332
+ }
333
+
334
+ /**
335
+ * Transport-level connection state, independent of {@link SyncHealthStatus}.
336
+ *
337
+ * These answer different questions: `connection` is about the socket,
338
+ * `status` is about whether sync rounds are succeeding. A `connected` socket
339
+ * can still be `degraded` (server erroring), and a `reconnecting` socket is
340
+ * usually still `healthy` for the first few seconds.
341
+ */
342
+ export type ConnectionState = 'connecting' | 'connected' | 'reconnecting' | 'disconnected';
343
+
270
344
  export type SyncHealthStatus = 'healthy' | 'degraded';
271
345
 
272
346
  /** Snapshot of sync health delivered to `subscribeToSyncHealth` subscribers. */
@@ -286,6 +360,13 @@ export interface SyncHealth {
286
360
  * a working session. Never resets back to `false` once set.
287
361
  */
288
362
  everConnected: boolean;
363
+ /**
364
+ * Live transport state of the remote WebSocket. Distinct from `status`: this
365
+ * one flips the instant the socket drops, whereas `status` only degrades
366
+ * after a sustained run of failed sync rounds. Use it to show "reconnecting…"
367
+ * immediately without waiting for the degrade threshold.
368
+ */
369
+ connection: ConnectionState;
289
370
  }
290
371
 
291
372
  export type StorageHealthStatus = 'unknown' | 'persistent' | 'memory';
@@ -370,6 +451,24 @@ export interface QueryConfig {
370
451
  * `rowCount` / `localArray`.
371
452
  */
372
453
  subqueryRemoteArray?: RecordVersionArray;
454
+ /**
455
+ * Whether authoritative membership (`remoteArray`) has ever been established
456
+ * for this query — either fetched from `_00_list_ref` this session, or read
457
+ * back from the durable `_00_window` row on a cold start.
458
+ *
459
+ * Tri-state matters: "known and empty" must render an empty list, while
460
+ * "never established" has to fall back to a predicate scan of the local store
461
+ * so a query first run on this device still paints offline. A
462
+ * `remoteArray.length === 0` check cannot tell those apart.
463
+ */
464
+ membershipKnown?: boolean;
465
+ /**
466
+ * Key of this query's durable `_00_window` membership row: a hash of
467
+ * `{surql, params}` WITHOUT the `session::id()` salt that `id` carries, so it
468
+ * survives a reload (which mints a new session id) and a bucket switch.
469
+ * In-memory only.
470
+ */
471
+ membershipKey?: string;
373
472
  /** Time-To-Live for this query. */
374
473
  ttl: QueryTimeToLive;
375
474
  /** Timestamp when the query was last accessed/active. */
@@ -199,3 +199,41 @@ export async function withRetry<T>(
199
199
  }
200
200
  throw lastError;
201
201
  }
202
+
203
+ /**
204
+ * Reject after `timeoutMs` if `promise` hasn't settled.
205
+ *
206
+ * Exists because a WebSocket RPC has no deadline of its own: the SurrealDB SDK
207
+ * parks each call in a pending map and only rejects it when the socket reports
208
+ * a close. On a half-open socket (peer gone, no `close` event, `readyState`
209
+ * still OPEN) the call never settles at all, which wedges anything serialized
210
+ * behind it.
211
+ *
212
+ * `message` MUST contain "timed out" so `classifySyncError` treats the
213
+ * rejection as `network` — that's what makes the sync queues retry the
214
+ * operation instead of rolling the mutation back as an application error.
215
+ *
216
+ * Non-positive `timeoutMs` returns `promise` unchanged. The underlying promise
217
+ * is NOT cancelled (nothing can cancel an in-flight RPC); its eventual
218
+ * settlement is absorbed, so attach no expectations to it after a timeout.
219
+ */
220
+ export function withTimeout<T>(
221
+ promise: Promise<T>,
222
+ timeoutMs: number,
223
+ message: string
224
+ ): Promise<T> {
225
+ if (!(timeoutMs > 0)) return promise;
226
+ return new Promise<T>((resolve, reject) => {
227
+ const timer = setTimeout(() => reject(new Error(message)), timeoutMs);
228
+ promise.then(
229
+ (value) => {
230
+ clearTimeout(timer);
231
+ resolve(value);
232
+ },
233
+ (err) => {
234
+ clearTimeout(timer);
235
+ reject(err);
236
+ }
237
+ );
238
+ });
239
+ }