taladb 0.9.3 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,6 +36,67 @@ interface VectorSearchResult<T extends Document = Document> {
36
36
  */
37
37
  score: number;
38
38
  }
39
+ interface TextSearchResult<T extends Document = Document> {
40
+ /** The matched document. */
41
+ document: T;
42
+ /**
43
+ * BM25 relevance score — higher means more relevant. Unbounded above, and
44
+ * only meaningful for ordering within a single query's result set.
45
+ */
46
+ score: number;
47
+ }
48
+ /** Tuning for `searchText`'s BM25 ranking. */
49
+ interface TextSearchOptions {
50
+ /**
51
+ * Term-frequency saturation (BM25 `k1`, default `1.2`). Higher values let a
52
+ * repeated term keep adding relevance for longer.
53
+ */
54
+ k1?: number;
55
+ /**
56
+ * Length normalisation (BM25 `b`, default `0.75`). `0` ignores document
57
+ * length; `1` normalises fully by length relative to the corpus average.
58
+ */
59
+ b?: number;
60
+ }
61
+ interface HybridSearchResult<T extends Document = Document> {
62
+ /** The matched document. */
63
+ document: T;
64
+ /**
65
+ * Fused reciprocal-rank-fusion score. Small by construction and meaningful
66
+ * only as an ordering within one result set — never a similarity or a
67
+ * confidence.
68
+ */
69
+ score: number;
70
+ /**
71
+ * Zero-based position in the text ranking, or `null` if the text retriever
72
+ * did not return this document.
73
+ */
74
+ textRank: number | null;
75
+ /**
76
+ * Zero-based position in the vector ranking, or `null` if the vector
77
+ * retriever did not return this document.
78
+ */
79
+ vectorRank: number | null;
80
+ }
81
+ /** Tuning for `hybridSearch`'s fusion and per-retriever scoring. */
82
+ interface HybridSearchOptions extends TextSearchOptions {
83
+ /**
84
+ * Reciprocal rank fusion smoothing constant (default `60`). Larger values
85
+ * flatten the advantage of the very top ranks.
86
+ */
87
+ rrfK?: number;
88
+ /** Relative weight of the text ranking (default `1`). Set `0` to disable it. */
89
+ textWeight?: number;
90
+ /** Relative weight of the vector ranking (default `1`). Set `0` to disable it. */
91
+ vectorWeight?: number;
92
+ /**
93
+ * How many candidates to pull from each retriever before fusing
94
+ * (default `max(topK * 4, 20)`). Raise it for better recall at more cost;
95
+ * fusing only `topK` from each side drops documents that rank just outside
96
+ * one retriever but high in the other.
97
+ */
98
+ candidates?: number;
99
+ }
39
100
  type Value = null | boolean | number | string | Uint8Array | Value[] | {
40
101
  [key: string]: Value;
41
102
  };
@@ -43,6 +104,16 @@ type Document = {
43
104
  _id?: string;
44
105
  [key: string]: Value | undefined;
45
106
  };
107
+ /**
108
+ * Who authored a write, and therefore whether it replicates outward.
109
+ *
110
+ * - `'local'` *(default)* — an ordinary user write. Replicates to peers as usual.
111
+ * - `'remote'` — a row replicated **in** from an authoritative origin. The origin
112
+ * already has it, so it must never go back out: rows written this way fire no
113
+ * sync events and never appear in `exportChanges()`, and deletes made this way
114
+ * leave no tombstone. Enforced in the engine, not by convention.
115
+ */
116
+ type WriteOrigin = 'local' | 'remote';
46
117
  /**
47
118
  * The operators available on a single field.
48
119
  *
@@ -211,12 +282,42 @@ interface CollectionOptions<T extends Document = Document> {
211
282
  * });
212
283
  */
213
284
  migrateDocument?: (doc: T, fromVersion: number) => T;
285
+ /**
286
+ * Lazy, read-time **downcast** — the mirror of {@link migrateDocument}, for a
287
+ * document written by a *newer* peer. When set, every document returned by
288
+ * `find` / `findOne` whose `_v` is **above** `syncSchema.version` is passed
289
+ * through `downgradeDocument(doc, fromVersion)` and projected into the shape
290
+ * this build understands, so application code on an old client sees a shape
291
+ * it can actually read instead of an unexpected future one.
292
+ *
293
+ * Requires `syncSchema.version`. Must be pure and deterministic.
294
+ *
295
+ * **The projection is view-only and is never persisted**, regardless of
296
+ * {@link persistMigrations}. The stored document keeps its original `_v` and
297
+ * its newer fields intact, because this replica must continue to replicate
298
+ * that document faithfully to other peers — an old client is a *reader* of a
299
+ * newer shape, never its editor. For the same reason the returned document
300
+ * keeps its original (higher) `_v`: it is a projection of a v-N document, not
301
+ * a v-M one, and writing it back wholesale would tell the fleet otherwise.
302
+ *
303
+ * @example
304
+ * // This build understands v1. A v2 peer split `name` into first/last.
305
+ * const users = db.collection<User>('users', {
306
+ * syncSchema: { version: 1 },
307
+ * downgradeDocument: (doc) => ({ ...doc, name: `${doc.first} ${doc.last}` }),
308
+ * });
309
+ */
310
+ downgradeDocument?: (doc: Readonly<T>, fromVersion: number) => T;
214
311
  /**
215
312
  * When `true`, a document upgraded by {@link migrateDocument} on read is
216
- * **written back** to storage (a best-effort `updateOne` computing the
217
- * `$set`/`$unset` diff) so the migration becomes permanent — after which
218
- * filters and indexes on the new shape match it. Default `false` (the
219
- * migrated shape is returned but not persisted).
313
+ * **written back** to storage (a best-effort `updateOne` computing a `$set`
314
+ * diff) so the migration becomes permanent — after which filters and indexes
315
+ * on the new shape match it. Default `false` (the migrated shape is returned
316
+ * but not persisted).
317
+ *
318
+ * The write-back is **additive-only** except for fields explicitly listed in
319
+ * {@link retiredFields}. A field present in storage but absent from the
320
+ * migrated document is otherwise left alone rather than `$unset`.
220
321
  *
221
322
  * Trade-offs: reads that encounter un-migrated documents now issue writes
222
323
  * (which fire live-query and sync-hook notifications like any other write);
@@ -224,6 +325,35 @@ interface CollectionOptions<T extends Document = Document> {
224
325
  * one-shot eager rewrite instead, prefer `openDB({ migrations })`.
225
326
  */
226
327
  persistMigrations?: boolean;
328
+ /**
329
+ * Fields an upcast is explicitly allowed to remove during persist-on-read.
330
+ * Prefer this precise list to {@link allowFieldRemoval}: omissions of any
331
+ * other field remain additive and are preserved.
332
+ */
333
+ retiredFields?: (keyof T & string)[];
334
+ /**
335
+ * How to read a field that {@link migrateDocument} left out of its output:
336
+ * as an intentional removal (`true`), or as a field the migration simply
337
+ * never heard of (`false`, the default).
338
+ *
339
+ * Default `false` — omitted fields are **preserved**: kept on the document
340
+ * returned to your code, and left in storage by the {@link persistMigrations}
341
+ * write-back rather than `$unset`.
342
+ *
343
+ * This default exists because on a synced collection the two cases are
344
+ * indistinguishable from the migration's output, and guessing "removal" is
345
+ * the destructive guess. A migration written today cannot mention a field a
346
+ * *newer* peer will add tomorrow, so an innocent `(doc) => ({ id, name })`
347
+ * becomes a deletion of a field its author never heard of — and under
348
+ * whole-document LWW that deletion replicates to the whole fleet. An old
349
+ * replica has to stay a faithful carrier of shapes it does not understand.
350
+ *
351
+ * Set `true` only on a collection that never syncs, or during a deliberate
352
+ * add → backfill → dual-read → **retire** rollout, where you already know the
353
+ * whole fleet has stopped writing the field.
354
+ */
355
+ /** @deprecated Prefer {@link retiredFields}; this treats every omission as removal. */
356
+ allowFieldRemoval?: boolean;
227
357
  }
228
358
  /** A single MongoDB-style aggregation stage. */
229
359
  type AggregateStage<T extends Document = Document> = {
@@ -255,6 +385,39 @@ type AggregatePipeline<T extends Document = Document> = AggregateStage<T>[];
255
385
  interface Collection<T extends Document = Document> {
256
386
  insert(doc: Omit<T, '_id'>): Promise<string>;
257
387
  insertMany(docs: Omit<T, '_id'>[]): Promise<string[]>;
388
+ /**
389
+ * Upsert many documents **by `_id`**, in a single commit. Existing rows are
390
+ * replaced in place, absent rows are created, and rows not named in `docs` are
391
+ * left alone — so writing page 2 never disturbs page 1.
392
+ *
393
+ * Unlike {@link insertMany}, which discards `_id` and mints a fresh ULID, this
394
+ * *honours* the id you supply. That is the whole point: for a row replicated
395
+ * from a remote origin, pass `_id: deriveDocId(collection, remoteKey)` and every
396
+ * later fetch of that row converges on the same document instead of duplicating
397
+ * it. Idempotent, and safe to run concurrently from a background hydration walk
398
+ * and an on-demand fetch.
399
+ *
400
+ * `origin: 'remote'` marks the rows as replicated in from an authoritative
401
+ * origin, which means they are **never replicated back out** — they will not
402
+ * fire sync events and will not appear in `exportChanges()`. Use it for anything
403
+ * the origin already knows about. Defaults to `'local'`.
404
+ *
405
+ * @example
406
+ * await products.replaceManyWithIds(
407
+ * rows.map((r) => ({ ...r, _id: deriveDocId('products', r.id) })),
408
+ * 'remote',
409
+ * );
410
+ */
411
+ replaceManyWithIds(docs: T[], origin?: WriteOrigin): Promise<string[]>;
412
+ /**
413
+ * Delete many documents by `_id`, in a single commit. Returns how many were
414
+ * present and removed; unknown ids are skipped.
415
+ *
416
+ * `origin: 'remote'` deletes **without a tombstone**, so the deletion is not
417
+ * replicated outward — correct when the origin is the one that told you the row
418
+ * was deleted. Defaults to `'local'`, which tombstones as usual.
419
+ */
420
+ deleteManyWithIds(ids: string[], origin?: WriteOrigin): Promise<number>;
258
421
  find(filter?: Filter<T>): Promise<T[]>;
259
422
  findOne(filter: Filter<T>): Promise<T | null>;
260
423
  updateOne(filter: Filter<T>, update: Update<T>): Promise<boolean>;
@@ -265,7 +428,12 @@ interface Collection<T extends Document = Document> {
265
428
  /**
266
429
  * Run a MongoDB-style aggregation pipeline (`$match`, `$group`, `$sort`,
267
430
  * `$skip`, `$limit`, `$project`) inside the engine. Returns the resulting
268
- * documents. Currently available on Node.js and the in-memory browser build.
431
+ * documents. Available on every runtime: Node, the OPFS worker, the in-memory
432
+ * browser build, and React Native.
433
+ *
434
+ * This is also how you page a collection locally — `find()` has no sort/skip/
435
+ * limit — but note it returns a **snapshot**. For a paged read that stays live
436
+ * as rows land, use {@link subscribeAggregate}.
269
437
  *
270
438
  * @example
271
439
  * const byStatus = await orders.aggregate([
@@ -303,6 +471,44 @@ interface Collection<T extends Document = Document> {
303
471
  createFtsIndex(field: keyof Omit<T, '_id'> & string): Promise<void>;
304
472
  /** Drop a full-text search index. */
305
473
  dropFtsIndex(field: keyof Omit<T, '_id'> & string): Promise<void>;
474
+ /**
475
+ * Rank documents against a free-text `query` using BM25, most relevant first.
476
+ *
477
+ * Unlike the `$contains` filter, which requires **every** token to be
478
+ * present, this uses OR semantics — a document that matches more of the
479
+ * query simply scores higher. Requires an FTS index on `field`.
480
+ *
481
+ * @example
482
+ * const hits = await articles.searchText('body', 'reset my password', 5);
483
+ * // hits: Array<{ document: Article, score: number }>
484
+ */
485
+ searchText(field: keyof Omit<T, '_id'> & string, query: string, topK: number, filter?: Filter<T>, options?: TextSearchOptions): Promise<TextSearchResult<T>[]>;
486
+ /**
487
+ * Hybrid retrieval: rank by keyword relevance (BM25) **and** vector
488
+ * similarity, then fuse the two rankings with reciprocal rank fusion.
489
+ *
490
+ * The two retrievers fail differently — keyword search misses paraphrases,
491
+ * vector search misses exact identifiers and rare proper nouns — so fusing
492
+ * them recovers both. A document both retrievers rank well outranks one that
493
+ * only a single retriever found. Requires an FTS index on `textField` and a
494
+ * vector index on `vectorField`.
495
+ *
496
+ * The optional `filter` is applied to both retrievers before ranking.
497
+ *
498
+ * @example
499
+ * const hits = await articles.hybridSearch(
500
+ * { textField: 'body', text: 'reset my password' },
501
+ * { vectorField: 'embedding', vector: queryVec },
502
+ * 5,
503
+ * );
504
+ */
505
+ hybridSearch(text: {
506
+ textField: keyof Omit<T, '_id'> & string;
507
+ text: string;
508
+ }, vector: {
509
+ vectorField: keyof Omit<T, '_id'> & string;
510
+ vector: number[];
511
+ }, topK: number, filter?: Filter<T>, options?: HybridSearchOptions): Promise<HybridSearchResult<T>[]>;
306
512
  /**
307
513
  * Return the indexes that currently exist on this collection.
308
514
  *
@@ -363,6 +569,28 @@ interface Collection<T extends Document = Document> {
363
569
  * unsub();
364
570
  */
365
571
  subscribe(filter: Filter<T>, callback: (docs: T[]) => void, onError?: (error: unknown) => void): () => void;
572
+ /**
573
+ * Subscribe to a live **aggregation** — the same as {@link subscribe}, but the
574
+ * result set is produced by a pipeline rather than a filter.
575
+ *
576
+ * This exists because {@link aggregate} is the only way to sort/skip/limit, and
577
+ * on its own it returns a dead snapshot: a paged read built on it would never
578
+ * re-run when new rows land, so a page would sit frozen while a background
579
+ * hydration filled the collection underneath it. Anything that pages locally
580
+ * should subscribe here instead of calling `aggregate` in an effect.
581
+ *
582
+ * The callback receives a snapshot immediately and again after every write that
583
+ * could affect the result.
584
+ *
585
+ * @returns An unsubscribe function.
586
+ *
587
+ * @example
588
+ * const unsub = products.subscribeAggregate(
589
+ * [{ $match: { category: 'kitchen' } }, { $sort: { price: 1 } }, { $limit: 20 }],
590
+ * (page) => render(page),
591
+ * );
592
+ */
593
+ subscribeAggregate<R extends Document = Document>(pipeline: AggregatePipeline<T>, callback: (docs: R[]) => void, onError?: (error: unknown) => void): () => void;
366
594
  }
367
595
  /**
368
596
  * A JSON-encoded changeset — the opaque payload exchanged between peers. Produced
@@ -384,9 +612,58 @@ interface SyncAdapter {
384
612
  * Fetch remote changes with `changed_at` after `sinceMs` (ms epoch), as a
385
613
  * serialized changeset. Return `'[]'` when there is nothing new. Required for
386
614
  * `'pull'` / `'both'`.
615
+ *
616
+ * @deprecated in spirit, not in support — wall-clock timestamps are not safe
617
+ * cursors (see {@link CursorSyncAdapter}), which is why every pass built on this
618
+ * method replays the whole collection from zero. Implement
619
+ * {@link CursorSyncAdapter.pullWithCursor} instead when your origin can issue a
620
+ * cursor. Adapters that only implement `pull` keep working unchanged.
387
621
  */
388
622
  pull?(sinceMs: number): Promise<SerializedChangeset>;
389
623
  }
624
+ /** One page of remote changes, plus where to resume from. */
625
+ interface PullResult {
626
+ /** The changes themselves. `'[]'` when there is nothing new. */
627
+ changeset: SerializedChangeset;
628
+ /**
629
+ * Opaque resume token, issued by the origin. **Never parse this.** It may be a
630
+ * timestamp, a sequence number, an LSN, a snapshot id — that is the origin's
631
+ * business, and treating it as a number is how clients reintroduce the
632
+ * clock-skew bug this type exists to kill.
633
+ */
634
+ cursor: string;
635
+ /** `true` when more pages remain; call again with the returned `cursor`. */
636
+ hasMore: boolean;
637
+ }
638
+ /**
639
+ * A {@link SyncAdapter} whose origin can issue a resume cursor.
640
+ *
641
+ * ## Why this exists
642
+ *
643
+ * The original contract is `pull(sinceMs)`, and it cannot be made correct. Author
644
+ * wall-clock timestamps are not safe cursors: a write can commit *after* an export
645
+ * yet carry an *earlier* timestamp, so resuming from "the newest timestamp I saw"
646
+ * silently drops rows. TalaDB's answer was to give up on cursors entirely and
647
+ * replay from zero on every pass — correct, but it re-downloads the whole
648
+ * collection forever, which makes a full local replica of a real catalog
649
+ * unaffordable.
650
+ *
651
+ * The fix is to stop inventing the cursor on the client. The origin issues an
652
+ * opaque token; we store it and hand it back. Whatever ordering guarantee the
653
+ * origin has (a sequence, an LSN, a snapshot) travels with the token, and the
654
+ * client never has to reason about clocks at all.
655
+ *
656
+ * `runSync` feature-detects `pullWithCursor` and prefers it. Adapters that only
657
+ * implement `pull(sinceMs)` are untouched and keep their replay-from-zero
658
+ * behavior.
659
+ */
660
+ interface CursorSyncAdapter extends SyncAdapter {
661
+ /**
662
+ * Fetch changes after `cursor`, or from the beginning when it is `null`.
663
+ * Returns the changes plus the token to resume from next time.
664
+ */
665
+ pullWithCursor(cursor: string | null): Promise<PullResult>;
666
+ }
390
667
  interface SyncOptions {
391
668
  /**
392
669
  * Collections to sync. Omit to sync **all** user collections (reserved
@@ -593,6 +870,504 @@ declare class HttpSyncAdapter implements SyncAdapter {
593
870
  pull(sinceMs: number): Promise<SerializedChangeset>;
594
871
  }
595
872
 
873
+ /**
874
+ * Deterministic document ids for replicated rows.
875
+ *
876
+ * The engine assigns ULIDs and **ignores a caller-supplied `_id`** — it silently
877
+ * becomes an ordinary field, so `find({ _id: 'sku-1' })` then matches nothing.
878
+ * That leaves a document replicated from a remote origin with no stable local
879
+ * identity to merge on: re-fetching the same row would insert a duplicate.
880
+ *
881
+ * Hashing the origin's primary key into the ULID gives that identity back. The
882
+ * same `(collection, key)` always maps to the same document, which is what makes
883
+ * replication upserts **idempotent** (re-applying a page is a no-op), **resumable**
884
+ * (a bootstrap walk can restart mid-way), and **safe to run concurrently** (an
885
+ * on-demand fetch and the background walk can touch the same row and converge on
886
+ * one document rather than two).
887
+ *
888
+ * ## This must stay byte-identical to the Rust `derive_doc_id`
889
+ *
890
+ * The same rows are addressed from both sides. If the two implementations ever
891
+ * disagree, two clients assign different `_id`s to the same remote row and the
892
+ * replica silently forks into duplicates — with no error anywhere. The shared
893
+ * test vectors in `derive-id.test.ts` and `packages/core/src/document.rs` exist to
894
+ * make that impossible to do by accident; keep them in lockstep.
895
+ *
896
+ * FNV-1a is used over a stronger hash precisely *because* it is short enough to
897
+ * port between the two languages without ambiguity. It is non-cryptographic, which
898
+ * is fine here: the input is a primary key from an origin the client already
899
+ * trusts, not adversarial input.
900
+ */
901
+ /**
902
+ * Derive a stable `_id` for a row replicated from a remote origin.
903
+ *
904
+ * `collection` is part of the preimage, so the same remote id in two different
905
+ * collections cannot collide.
906
+ *
907
+ * @example
908
+ * deriveDocId('products', 'sku-123') // → '56GC678DQYWW1Z98HPYJ90WVKH', always
909
+ *
910
+ * ## Ordering caveat
911
+ *
912
+ * The result is a hash, so its ULID timestamp prefix is **not** chronological.
913
+ * Documents written with a derived id do not come back in insertion order from an
914
+ * unsorted `find()`; reads over replicated collections must carry an explicit
915
+ * sort. Documents written via `insert`/`insertMany` are unaffected — they still
916
+ * get monotonic ULIDs.
917
+ */
918
+ declare function deriveDocId(collection: string, key: string): string;
919
+
920
+ /**
921
+ * Coverage — "is this collection complete enough, locally, to answer a query
922
+ * without the network?"
923
+ *
924
+ * This is the question the whole coverage-first design turns on, and it is *not*
925
+ * "have I fetched this page?". A replica assembled from whichever pages a user
926
+ * happened to visit is an arbitrary partial subset: it cannot answer a query
927
+ * nobody has asked yet ("products under ₱500" may live on page 43), so every new
928
+ * filter or sort still goes to the network and the local database buys you almost
929
+ * nothing. Coverage is what licenses a purely local read.
930
+ *
931
+ * Two things make it trustworthy:
932
+ *
933
+ * 1. **It is scoped, not per-collection.** `complete` for a bare collection name
934
+ * would leak across users: log in as someone else and you inherit the previous
935
+ * user's "complete" flag *and* their rows. The key is a tuple.
936
+ * 2. **It is a state machine, not a boolean.** Only `complete` authorizes a
937
+ * local-only read. `best-effort` exists precisely so an origin that *cannot*
938
+ * give us a consistent snapshot degrades honestly instead of claiming a
939
+ * completeness it never established.
940
+ */
941
+
942
+ /** Reserved collection holding one coverage document per replicated scope. */
943
+ declare const COVERAGE_COLLECTION = "__taladb_replica";
944
+ /**
945
+ * What identifies a replicated scope. Every component must be part of the key,
946
+ * because each one changes what "complete" means:
947
+ *
948
+ * - `origin` — two origins are two different datasets.
949
+ * - `collection` — the local collection being filled.
950
+ * - `scope` — the *authorization* slice (a user, a tenant, a store). This is the
951
+ * one that bites: without it, user B logging in inherits user A's completeness.
952
+ * - `projectionVersion` — a replica hydrated with a slimmer projection is not
953
+ * complete for a query that needs the dropped fields.
954
+ * - `schemaVersion` — rows hydrated under an older shape may not satisfy today's.
955
+ */
956
+ interface CoverageKey {
957
+ origin: string;
958
+ collection: string;
959
+ scope: string;
960
+ projectionVersion: number;
961
+ schemaVersion: number;
962
+ }
963
+ type CoverageState =
964
+ /** Nothing local. */
965
+ {
966
+ status: 'empty';
967
+ }
968
+ /**
969
+ * A bootstrap walk is in progress. `snapshot` pins every page to one logical
970
+ * view of the origin; `nextPage` is the durable resume point.
971
+ */
972
+ | {
973
+ status: 'hydrating';
974
+ snapshot: string;
975
+ nextPage: string | number;
976
+ rowsApplied: number;
977
+ deltaCursor?: string;
978
+ total?: number;
979
+ }
980
+ /**
981
+ * The scope is fully local as of `cursor`. **The only state that permits a
982
+ * local-only read.**
983
+ */
984
+ | {
985
+ status: 'complete';
986
+ cursor: string;
987
+ completedAt: number;
988
+ rowsApplied: number;
989
+ total?: number;
990
+ }
991
+ /**
992
+ * Every row the origin offered was applied, but the origin could not pin a
993
+ * snapshot, so we cannot *prove* we saw a consistent view — a row that shifted
994
+ * between pages mid-walk may have been missed. Reads must not treat this as
995
+ * authoritative.
996
+ */
997
+ | {
998
+ status: 'best-effort';
999
+ cursor: string;
1000
+ reason: string;
1001
+ rowsApplied: number;
1002
+ total?: number;
1003
+ }
1004
+ /** Complete once, but known to have fallen behind (e.g. a projection change). */
1005
+ | {
1006
+ status: 'stale';
1007
+ cursor: string;
1008
+ reason: string;
1009
+ }
1010
+ /** The walk failed. `resumeFrom` is where to pick it up. */
1011
+ | {
1012
+ status: 'error';
1013
+ resumeFrom: string | number;
1014
+ snapshot?: string;
1015
+ deltaCursor?: string;
1016
+ rowsApplied?: number;
1017
+ total?: number;
1018
+ error: string;
1019
+ };
1020
+ /**
1021
+ * Serialize a {@link CoverageKey} into a stable string.
1022
+ *
1023
+ * Field order is fixed rather than derived from `Object.keys`, so the key cannot
1024
+ * change meaning if someone reorders the interface — a silent coverage reset,
1025
+ * which would look like "the app re-downloads everything for no reason".
1026
+ */
1027
+ declare function coverageKey(key: CoverageKey): string;
1028
+ /**
1029
+ * Persistent coverage state, one document per scope.
1030
+ *
1031
+ * The state is stored as a JSON string rather than as structured fields: it is a
1032
+ * discriminated union whose shape varies per variant, and TalaDB documents are
1033
+ * flat. Writing it whole also makes each transition a single atomic write, which
1034
+ * is what lets `markComplete` be the durable commit point of a bootstrap.
1035
+ */
1036
+ declare class CoverageStore {
1037
+ private readonly col;
1038
+ constructor(db: TalaDB);
1039
+ read(key: CoverageKey): Promise<CoverageState>;
1040
+ write(key: CoverageKey, state: CoverageState): Promise<void>;
1041
+ /** Drop a scope's coverage, forcing a fresh bootstrap on next use. */
1042
+ clear(key: CoverageKey): Promise<void>;
1043
+ }
1044
+ /**
1045
+ * Whether a local-only read is authorized for this state.
1046
+ *
1047
+ * Deliberately strict: **only `complete`**. `best-effort` is the interesting
1048
+ * exclusion — it means we applied everything the origin gave us, but the origin
1049
+ * could not pin a snapshot, so a row that moved between pages during the walk may
1050
+ * never have been seen. Serving that as authoritative would silently return
1051
+ * incomplete results, which is worse than going to the network.
1052
+ */
1053
+ declare function isAuthoritative(state: CoverageState): boolean;
1054
+ /** Rows applied so far, for progress reporting. */
1055
+ declare function rowsApplied(state: CoverageState): number;
1056
+ /** Fractional hydration progress, when the origin told us the total. */
1057
+ declare function progress(state: CoverageState): number | undefined;
1058
+
1059
+ /**
1060
+ * The replication *source* — wire translation, and nothing else.
1061
+ *
1062
+ * A source knows how to talk to one origin: how to ask for a page, how to ask for
1063
+ * changes since a cursor, how to find a row's primary key, and how to shape a row
1064
+ * into a document. It owns **no orchestration**: no batching, no yielding, no
1065
+ * cursor persistence, no coverage transitions, no retry, no dedup. All of that
1066
+ * belongs to the coordinator, which is generic over sources.
1067
+ *
1068
+ * That split is deliberate. The obvious alternative — make the REST origin a
1069
+ * `SyncAdapter` and let `db.sync()` drive it — does not work: a bootstrap of 100k
1070
+ * rows would sit inside a single `pull()` call with no way to report progress,
1071
+ * pause, resume, or yield to the UI between pages. Orchestration has to live one
1072
+ * level up, or it cannot be orchestrated at all.
1073
+ */
1074
+
1075
+ /** The origin's primary key for a row. Stringified before hashing into an id. */
1076
+ type RemoteKey = string;
1077
+ /** A request for one page of the initial bootstrap walk. */
1078
+ interface BootstrapRequest {
1079
+ /**
1080
+ * Where to resume. `null` on the first call — which is also when the origin is
1081
+ * expected to *issue* the snapshot and delta cursor.
1082
+ */
1083
+ page: string | number | null;
1084
+ /**
1085
+ * The snapshot token from the first page, echoed back on every subsequent one.
1086
+ * `null` on the first call, and on origins that don't support snapshots.
1087
+ */
1088
+ snapshot: string | null;
1089
+ /** Rows per page. */
1090
+ limit: number;
1091
+ }
1092
+ /** One page of the bootstrap walk. */
1093
+ interface BootstrapPage<RemoteRow> {
1094
+ rows: RemoteRow[];
1095
+ /** Resume token for the next page; `null` when the walk is done. */
1096
+ nextPage: string | number | null;
1097
+ /**
1098
+ * An opaque token pinning every page of this walk to one logical view of the
1099
+ * origin.
1100
+ *
1101
+ * **Omit it and you get `best-effort` coverage, not `complete`.** Without a
1102
+ * snapshot, a page walk over live data is not a consistent read: fetch page 1,
1103
+ * a row is inserted, everything shifts, and the row that was going to be on
1104
+ * page 20 is now on page 19 — which you already passed. It is never seen. The
1105
+ * walk still "succeeds", and the replica silently has a hole in it. Since
1106
+ * nothing detects that, the honest response is to refuse to call the result
1107
+ * complete, and to keep serving reads from the network.
1108
+ */
1109
+ snapshot?: string;
1110
+ /**
1111
+ * The cursor to begin the *delta* stream from once the walk finishes. Issued on
1112
+ * the first page — i.e. as of the snapshot — so no change made during the walk
1113
+ * can slip between "bootstrap ended" and "delta began".
1114
+ */
1115
+ deltaCursor?: string;
1116
+ /** Total rows in scope, when the origin knows it. Drives progress reporting. */
1117
+ total?: number;
1118
+ }
1119
+ /** One batch of incremental changes since a cursor. */
1120
+ interface DeltaPage<RemoteRow> {
1121
+ changed: RemoteRow[];
1122
+ /**
1123
+ * Primary keys the origin has deleted.
1124
+ *
1125
+ * This is the only way a REST replica learns about deletions. A plain paged GET
1126
+ * returns survivors, and a row's *absence* from a response is ambiguous — it may
1127
+ * have been deleted, or it may merely have shifted to another page. Guessing
1128
+ * would eventually delete live data, so we never infer; the origin must say so.
1129
+ */
1130
+ deleted: RemoteKey[];
1131
+ cursor: string;
1132
+ hasMore: boolean;
1133
+ }
1134
+ /**
1135
+ * Everything the coordinator needs to replicate one collection from one origin.
1136
+ *
1137
+ * @typeParam RemoteRow - the row shape the origin returns, before mapping.
1138
+ * @typeParam T - the local document shape.
1139
+ */
1140
+ interface ReplicationSource<RemoteRow = unknown, T extends Document = Document> {
1141
+ /** Bump when a custom source's behavior changes without changing its metadata. */
1142
+ readonly configVersion?: string | number;
1143
+ /** Stable identity for this origin. Part of the coverage key. */
1144
+ readonly origin: string;
1145
+ /** The local collection this source fills. */
1146
+ readonly collection: string;
1147
+ /**
1148
+ * The authorization slice these rows belong to — a user, tenant, or store.
1149
+ * Part of the coverage key, so one user's completeness never licenses another's
1150
+ * reads. Use a constant for genuinely global data.
1151
+ */
1152
+ readonly scope: string;
1153
+ /** Bump when {@link mapRow} starts producing a different shape. */
1154
+ readonly projectionVersion: number;
1155
+ /** Bump when the local schema changes in a way hydrated rows must match. */
1156
+ readonly schemaVersion: number;
1157
+ /** Fetch one page of the initial walk. */
1158
+ bootstrap(request: BootstrapRequest): Promise<BootstrapPage<RemoteRow>>;
1159
+ /** Fetch changes since `cursor`. Absent when the origin has no delta feed. */
1160
+ delta?(cursor: string): Promise<DeltaPage<RemoteRow>>;
1161
+ /**
1162
+ * Fetch exactly the rows a specific query needs, for the cold-start bridge.
1163
+ *
1164
+ * Optional. When absent, a query against an un-hydrated scope simply waits for
1165
+ * coverage rather than short-circuiting to the network.
1166
+ */
1167
+ fetchQuery?(query: BridgeQuery): Promise<RemoteRow[]>;
1168
+ /** The origin's primary key for a row. Must be stable across fetches. */
1169
+ keyOf(row: RemoteRow): RemoteKey;
1170
+ /**
1171
+ * Monotonic authoritative revision for stale-response protection. Strongly
1172
+ * recommended whenever bridge/bootstrap/delta requests may overlap.
1173
+ */
1174
+ revisionOf(row: RemoteRow): number;
1175
+ /** Shape a remote row into a local document (minus `_id`, which is derived). */
1176
+ mapRow(row: RemoteRow): Omit<T, '_id'>;
1177
+ }
1178
+ /**
1179
+ * A local query, handed to the bridge so it can ask the origin for the same rows.
1180
+ *
1181
+ * Deliberately loose: every REST API spells pagination and filtering differently,
1182
+ * so translating this into a query string is the source's job, not ours.
1183
+ */
1184
+ interface BridgeQuery {
1185
+ filter?: Record<string, unknown>;
1186
+ sort?: Record<string, 1 | -1>;
1187
+ page?: number;
1188
+ limit?: number;
1189
+ }
1190
+
1191
+ /**
1192
+ * The replication coordinator — all orchestration, no wire format.
1193
+ *
1194
+ * Owns: the bootstrap walk, resume-after-crash, delta refresh, the cold-start
1195
+ * bridge, batching, yielding, coverage transitions, and in-flight dedup. The
1196
+ * {@link ReplicationSource} it drives owns only wire translation.
1197
+ *
1198
+ * ## The two mechanisms are one mechanism
1199
+ *
1200
+ * "Fetch the page the user is looking at" and "import the whole catalog in the
1201
+ * background" look like separate features. They are the same primitive with two
1202
+ * schedulers: *fetch rows → upsert them by derived id*. Because both write the
1203
+ * **same rows under the same ids**, they compose for free — a bridged fetch is not
1204
+ * a throwaway cache entry, it is a down payment on the replica, and when the walk
1205
+ * later reaches those rows it overwrites them in place instead of duplicating
1206
+ * them. Nothing has to reconcile the two.
1207
+ *
1208
+ * The one thing they do *not* share is coverage. A bridge fetch must never advance
1209
+ * the bootstrap cursor, because it did not come from the walk's snapshot and
1210
+ * proves nothing about completeness. Trading a little duplicate network for a
1211
+ * trustworthy completeness proof is the right side of that bargain.
1212
+ */
1213
+
1214
+ interface CoordinatorOptions<T extends Document = Document> {
1215
+ /** Rows per bootstrap page. Larger = fewer commits, longer stalls. */
1216
+ pageSize?: number;
1217
+ /**
1218
+ * Called between pages so the walk yields. Defaults to a macrotask.
1219
+ *
1220
+ * This matters more than it looks. Live queries re-run on a 300 ms poll, and on
1221
+ * React Native every write is *synchronous on the JS thread* — a tight bootstrap
1222
+ * loop starves both, and the UI freezes for the duration of the import.
1223
+ */
1224
+ yieldFn?: () => Promise<void>;
1225
+ /** Fired after each committed page, for progress UI. */
1226
+ onProgress?: (state: CoverageState) => void;
1227
+ /** Collection schema/migration options registered by the host application. */
1228
+ collectionOptions?: CollectionOptions<T>;
1229
+ }
1230
+ declare const REPLICA_SCOPE_FIELD = "_replica_scope";
1231
+ declare const REPLICA_REVISION_FIELD = "_remote_rev";
1232
+ interface BridgeResult {
1233
+ count: number;
1234
+ ids: string[];
1235
+ }
1236
+ declare class ReplicationCoordinator<RemoteRow, T extends Document> {
1237
+ private readonly db;
1238
+ private readonly source;
1239
+ private readonly coverage;
1240
+ private readonly key;
1241
+ private readonly pageSize;
1242
+ private readonly yieldFn;
1243
+ private readonly onProgress?;
1244
+ private readonly collectionOptions?;
1245
+ /**
1246
+ * In-flight passes, keyed by intent. Two components mounting the same query must
1247
+ * fire one request, and the background walk must not race the bridge for the
1248
+ * same rows — both join the existing promise instead.
1249
+ */
1250
+ private readonly inflight;
1251
+ constructor(db: TalaDB, source: ReplicationSource<RemoteRow, T>, options?: CoordinatorOptions<T>);
1252
+ get replicaScope(): string;
1253
+ private get identityNamespace();
1254
+ getCoverage(): Promise<CoverageState>;
1255
+ /** Whether a purely local read is authorized right now. */
1256
+ isReady(): Promise<boolean>;
1257
+ /** Dedup by intent: identical concurrent work joins rather than duplicating. */
1258
+ private dedup;
1259
+ /**
1260
+ * Write a batch of remote rows into the local collection.
1261
+ *
1262
+ * One commit for the whole batch, ids derived from the origin's primary key, and
1263
+ * `origin: 'remote'` so the rows can never replicate back out at the origin they
1264
+ * came from. This is the *only* write path in the coordinator — bootstrap, delta
1265
+ * and bridge all funnel through it, which is precisely why they converge instead
1266
+ * of conflicting.
1267
+ */
1268
+ private applyRows;
1269
+ /**
1270
+ * Hydrate the scope: walk the origin page by page until the whole collection is
1271
+ * local, then mark it complete.
1272
+ *
1273
+ * Resumable and idempotent. If the walk is interrupted — a reload, a crash, a
1274
+ * dead network — the next call picks up from the last committed page, and
1275
+ * re-applying a page it already wrote is a no-op because the ids are derived.
1276
+ */
1277
+ hydrate(): Promise<CoverageState>;
1278
+ private runHydrate;
1279
+ /**
1280
+ * Apply incremental changes since the stored cursor.
1281
+ *
1282
+ * Deletions are applied by mapping the origin's primary keys through the same
1283
+ * `deriveDocId`, and are written with `origin: 'remote'` so they leave no
1284
+ * tombstone — the origin already knows it deleted these, and a tombstone would
1285
+ * push its own deletion back at it.
1286
+ */
1287
+ refresh(): Promise<CoverageState>;
1288
+ private runRefresh;
1289
+ /**
1290
+ * Cold-start bridge: fetch exactly the rows one query needs, right now.
1291
+ *
1292
+ * Needed because a SPA or React Native app has no server render to paint behind
1293
+ * while the replica fills. The rows land in the same collection under the same
1294
+ * derived ids as the walk's, so this is not a cache — it is the replica, arriving
1295
+ * early.
1296
+ *
1297
+ * **Does not advance coverage.** These rows did not come from the bootstrap
1298
+ * snapshot and prove nothing about completeness; treating them as progress would
1299
+ * let a page-1 fetch masquerade as a hydrated catalog.
1300
+ */
1301
+ bridge(query: BridgeQuery): Promise<BridgeResult>;
1302
+ /** Drop coverage and force a fresh bootstrap. Local rows are left alone. */
1303
+ reset(): Promise<void>;
1304
+ }
1305
+
1306
+ /**
1307
+ * A {@link ReplicationSource} for an ordinary paged JSON API.
1308
+ *
1309
+ * This is the adoption path: point it at `GET /api/products?page=1&limit=500` and
1310
+ * a team on Express + Postgres gets a local replica without rewriting their API to
1311
+ * speak TalaDB's sync contract. Everything here is wire translation — the
1312
+ * coordinator owns the walk, the coverage, and the retries.
1313
+ *
1314
+ * ## What the origin has to provide, and what happens when it doesn't
1315
+ *
1316
+ * | Feature | Endpoint | Without it |
1317
+ * |---|---|---|
1318
+ * | Paged list | `?page=&limit=` | Nothing works. Required. |
1319
+ * | Snapshot token | `snapshot` in the response | Coverage caps at `best-effort`; reads keep hitting the network |
1320
+ * | Delta feed | `?since=<cursor>` | No incremental refresh, and **deletions never propagate** |
1321
+ *
1322
+ * The snapshot and the delta feed are each about twenty minutes of Express work
1323
+ * (a monotonic `updated_at`/revision column, a soft-delete table, and a
1324
+ * `rev <= snapshotRev` predicate). They are worth it: without a snapshot the
1325
+ * replica can never be trusted for a local-only read, which is the entire point.
1326
+ */
1327
+
1328
+ interface RestSourceOptions<RemoteRow, T extends Document> {
1329
+ /** Base URL, e.g. `/api/products`. */
1330
+ endpoint: string;
1331
+ /** The local collection to fill. */
1332
+ collection: string;
1333
+ /** Stable identity for the origin. Defaults to `endpoint`. */
1334
+ origin?: string;
1335
+ /**
1336
+ * The authorization slice these rows belong to — a user id, tenant, or store.
1337
+ * Part of the coverage key, so one user's completeness never licenses another
1338
+ * user's reads. Defaults to `'global'`; **set it for anything user-scoped.**
1339
+ */
1340
+ scope?: string;
1341
+ /** Bump when {@link mapRow} starts producing a different shape. Default 1. */
1342
+ projectionVersion?: number;
1343
+ /** Bump when the local schema changes. Default 1. */
1344
+ schemaVersion?: number;
1345
+ /** Field on the remote row holding its primary key. Default `'id'`. */
1346
+ key?: string;
1347
+ /** Field/callback yielding a monotonic numeric row revision. Default `'rev'`. */
1348
+ revision?: string | ((row: RemoteRow) => number | undefined);
1349
+ /** Shape a remote row into a local document. Default: identity, minus `_id`. */
1350
+ mapRow?: (row: RemoteRow) => Omit<T, '_id'>;
1351
+ /** Per-request headers, resolved **at send time** so a refreshed token is used. */
1352
+ getAuth?: () => Promise<Record<string, string>> | Record<string, string>;
1353
+ /** `fetch` implementation. Defaults to the global. */
1354
+ fetch?: typeof fetch;
1355
+ /** Sub-paths appended to `endpoint`. */
1356
+ paths?: {
1357
+ bootstrap?: string;
1358
+ delta?: string;
1359
+ };
1360
+ /** Enable delta polling. Defaults to true only when `paths.delta` is set. */
1361
+ delta?: boolean;
1362
+ /** Meaning of the fallback `page` parameter when no next token is returned. */
1363
+ pagination?: 'page' | 'offset';
1364
+ /** Translate a local query into this API's query-string conventions. */
1365
+ toParams?: (query: BridgeQuery) => Record<string, string>;
1366
+ /** Pull the row array out of a response whose envelope we don't recognize. */
1367
+ parse?: (body: unknown) => unknown[];
1368
+ }
1369
+ declare function createRestSource<RemoteRow = Record<string, unknown>, T extends Document = Document>(options: RestSourceOptions<RemoteRow, T>): ReplicationSource<RemoteRow, T>;
1370
+
596
1371
  /**
597
1372
  * Thrown when a document fails schema validation on `insert` or `insertMany`.
598
1373
  * The `cause` property holds the original error thrown by the schema library.
@@ -704,4 +1479,4 @@ interface OpenDBOptions {
704
1479
  */
705
1480
  declare function openDB(dbName?: string, options?: OpenDBOptions): Promise<TalaDB>;
706
1481
 
707
- export { type AggregatePipeline, type AggregateStage, type Collection, type CollectionIndexInfo, type CollectionOptions, type Document, type DurabilityConfig, type Filter, HttpSyncAdapter, type Migration, type OpenDBOptions, type Schema, type SerializedChangeset, type SyncAdapter, type SyncConfig, type SyncDirection, type SyncOptions, type SyncResult, type TalaDB, type TalaDbConfig, TalaDbValidationError, type Update, type Value, type VectorIndexOptions, type VectorMetric, type VectorSearchResult, applySchema, openDB, runMigrations };
1482
+ export { type AggregatePipeline, type AggregateStage, type BootstrapPage, type BootstrapRequest, type BridgeQuery, type BridgeResult, COVERAGE_COLLECTION, type Collection, type CollectionIndexInfo, type CollectionOptions, type CoordinatorOptions, type CoverageKey, type CoverageState, CoverageStore, type CursorSyncAdapter, type DeltaPage, type Document, type DurabilityConfig, type Filter, HttpSyncAdapter, type HybridSearchOptions, type HybridSearchResult, type Migration, type OpenDBOptions, type PullResult, REPLICA_REVISION_FIELD, REPLICA_SCOPE_FIELD, type RemoteKey, ReplicationCoordinator, type ReplicationSource, type RestSourceOptions, type Schema, type SerializedChangeset, type SyncAdapter, type SyncConfig, type SyncDirection, type SyncOptions, type SyncResult, type TalaDB, type TalaDbConfig, TalaDbValidationError, type TextSearchOptions, type TextSearchResult, type Update, type Value, type VectorIndexOptions, type VectorMetric, type VectorSearchResult, type WriteOrigin, applySchema, coverageKey, createRestSource, deriveDocId, isAuthoritative, openDB, progress, rowsApplied, runMigrations };