@pylonsync/sync 0.3.365 → 0.3.366

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -249,20 +249,29 @@ export declare class SyncEngine {
249
249
  */
250
250
  private applyQueue;
251
251
  /**
252
- * Live-event hold buffer, active ONLY while a from-zero snapshot pull is in
253
- * flight. A snapshot is full state as-of `snapshot_seq` S; its rows arrive
254
- * tagged `seq = S`. If a live WS frame (or a tab broadcast) at `seq = S+k`
255
- * applies FIRST — during the snapshot's (possibly multi-page) HTTP fetch — it
256
- * advances the cursor past S, and then EVERY snapshot row (seq ≤ S) is
257
- * dropped by the monotonic filter in enqueueApply, leaving a near-empty
258
- * replica with the cursor persisted ahead (no 410, no heal until a reconcile
259
- * happens to fire). The store has no per-row seq guard, so we can't just
260
- * apply the snapshot unconditionally — an older snapshot row would clobber a
261
- * newer live update. So we instead ORDER them: hold live/broadcast applies
262
- * here while snapshotting, then replay them (seq-filtered) AFTER the snapshot
263
- * lands. null = not snapshotting → normal apply.
252
+ * Live-event hold buffer, active while ANY pull is in flight.
253
+ *
254
+ * From-zero snapshot: a snapshot is full state as-of `snapshot_seq` S; its
255
+ * rows arrive tagged `seq = S`. If a live WS frame (or a tab broadcast) at
256
+ * `seq = S+k` applies FIRST — during the snapshot's (possibly multi-page)
257
+ * HTTP fetch — it advances the cursor past S, and then EVERY snapshot row
258
+ * (seq ≤ S) is dropped by the monotonic filter in enqueueApply, leaving a
259
+ * near-empty replica with the cursor persisted ahead (no 410, no heal until
260
+ * a reconcile happens to fire).
261
+ *
262
+ * Delta catch-up has the SAME leapfrog: a live frame at seq S+k landing
263
+ * mid-catch-up advances the cursor past the not-yet-pulled gap (cursor, S+k),
264
+ * and when the pull then delivers those gap events the monotonic filter
265
+ * drops them — silently missing rows until the next reconcile sweeps them.
266
+ *
267
+ * The store has no per-row seq guard, so we can't just apply out-of-order —
268
+ * an older row would clobber a newer live update. So we ORDER them: hold
269
+ * live/broadcast applies here while pulling, then replay them (seq-filtered)
270
+ * AFTER the pull lands. On a failed pull the buffer is DISCARDED — the
271
+ * cursor didn't reach the held seqs, so the next pull re-fetches those
272
+ * events from the log. null = no pull in flight → normal apply.
264
273
  */
265
- private snapshotHold;
274
+ private pullHold;
266
275
  /**
267
276
  * Serialized channel for outbound network ops (pull, push, reconcile,
268
277
  * refresh, resetReplica). Replaces the per-op `inFlightX` mutexes +
package/package.json CHANGED
@@ -3,7 +3,7 @@
3
3
  "publishConfig": {
4
4
  "access": "public"
5
5
  },
6
- "version": "0.3.365",
6
+ "version": "0.3.366",
7
7
  "type": "module",
8
8
  "main": "./src/index.ts",
9
9
  "types": "./dist/index.d.ts",
package/src/index.ts CHANGED
@@ -375,20 +375,29 @@ export class SyncEngine {
375
375
  private applyQueue: Promise<void> = Promise.resolve();
376
376
 
377
377
  /**
378
- * Live-event hold buffer, active ONLY while a from-zero snapshot pull is in
379
- * flight. A snapshot is full state as-of `snapshot_seq` S; its rows arrive
380
- * tagged `seq = S`. If a live WS frame (or a tab broadcast) at `seq = S+k`
381
- * applies FIRST — during the snapshot's (possibly multi-page) HTTP fetch — it
382
- * advances the cursor past S, and then EVERY snapshot row (seq ≤ S) is
383
- * dropped by the monotonic filter in enqueueApply, leaving a near-empty
384
- * replica with the cursor persisted ahead (no 410, no heal until a reconcile
385
- * happens to fire). The store has no per-row seq guard, so we can't just
386
- * apply the snapshot unconditionally — an older snapshot row would clobber a
387
- * newer live update. So we instead ORDER them: hold live/broadcast applies
388
- * here while snapshotting, then replay them (seq-filtered) AFTER the snapshot
389
- * lands. null = not snapshotting → normal apply.
378
+ * Live-event hold buffer, active while ANY pull is in flight.
379
+ *
380
+ * From-zero snapshot: a snapshot is full state as-of `snapshot_seq` S; its
381
+ * rows arrive tagged `seq = S`. If a live WS frame (or a tab broadcast) at
382
+ * `seq = S+k` applies FIRST — during the snapshot's (possibly multi-page)
383
+ * HTTP fetch — it advances the cursor past S, and then EVERY snapshot row
384
+ * (seq ≤ S) is dropped by the monotonic filter in enqueueApply, leaving a
385
+ * near-empty replica with the cursor persisted ahead (no 410, no heal until
386
+ * a reconcile happens to fire).
387
+ *
388
+ * Delta catch-up has the SAME leapfrog: a live frame at seq S+k landing
389
+ * mid-catch-up advances the cursor past the not-yet-pulled gap (cursor, S+k),
390
+ * and when the pull then delivers those gap events the monotonic filter
391
+ * drops them — silently missing rows until the next reconcile sweeps them.
392
+ *
393
+ * The store has no per-row seq guard, so we can't just apply out-of-order —
394
+ * an older row would clobber a newer live update. So we ORDER them: hold
395
+ * live/broadcast applies here while pulling, then replay them (seq-filtered)
396
+ * AFTER the pull lands. On a failed pull the buffer is DISCARDED — the
397
+ * cursor didn't reach the held seqs, so the next pull re-fetches those
398
+ * events from the log. null = no pull in flight → normal apply.
390
399
  */
391
- private snapshotHold: ChangeEvent[] | null = null;
400
+ private pullHold: ChangeEvent[] | null = null;
392
401
 
393
402
  /**
394
403
  * Serialized channel for outbound network ops (pull, push, reconcile,
@@ -1305,14 +1314,15 @@ export class SyncEngine {
1305
1314
  targetCursor?: SyncCursor,
1306
1315
  opts: { fromBroadcast?: boolean; isPull?: boolean } = {},
1307
1316
  ): Promise<void> {
1308
- // Snapshot fence: while a from-zero snapshot is in flight, hold live WS
1309
- // frames + tab broadcasts so they can't advance the cursor past the
1310
- // snapshot's rows and filter them out (see `snapshotHold`). The pull's own
1311
- // apply (`isPull`) is exempt — it IS the snapshot. Held events are replayed
1312
- // in arrival (≈seq) order once the snapshot lands. Synchronous + before the
1313
- // queue chain so held events never interleave into the applyQueue.
1314
- if (this.snapshotHold !== null && !opts.isPull) {
1315
- this.snapshotHold.push(...changes);
1317
+ // Pull fence: while any pull is in flight, hold live WS frames + tab
1318
+ // broadcasts so they can't advance the cursor past rows the pull hasn't
1319
+ // delivered yet and get them filtered out (see `pullHold`). The pull's
1320
+ // own applies (`isPull`) are exempt — they ARE the pull. Held events are
1321
+ // replayed in arrival (≈seq) order once the pull lands. Synchronous +
1322
+ // before the queue chain so held events never interleave into the
1323
+ // applyQueue.
1324
+ if (this.pullHold !== null && !opts.isPull) {
1325
+ this.pullHold.push(...changes);
1316
1326
  return Promise.resolve();
1317
1327
  }
1318
1328
  const prev = this.applyQueue;
@@ -1708,12 +1718,14 @@ export class SyncEngine {
1708
1718
  // bootstrap reconcile (the snapshot path already returned every
1709
1719
  // policy-visible row, per-entity refetch right after is waste).
1710
1720
  const startedFromZero = this.cursor.last_seq === 0;
1711
- // A from-zero pull is a SNAPSHOT — open the live-event hold for its whole
1712
- // (possibly multi-page) duration so a racing WS frame can't leapfrog the
1713
- // cursor and filter the snapshot rows out. Nested pulls (delta tail /
1714
- // has_more, 410 recursion) run at a non-zero cursor → they don't touch
1715
- // this, and their applies pass `isPull` so they're never held.
1716
- if (startedFromZero) this.snapshotHold = [];
1721
+ // Open the live-event hold for the pull's whole (possibly multi-page)
1722
+ // duration so a racing WS frame can't leapfrog the cursor — past the
1723
+ // snapshot's rows on a from-zero pull, or past the not-yet-pulled gap
1724
+ // on a delta catch-up (see `pullHold`). The pull's own applies pass
1725
+ // `isPull` so they're never held. The 410 handler discards this
1726
+ // buffer before recursing, so the nested pull always opens (and owns)
1727
+ // a fresh hold.
1728
+ this.pullHold = [];
1717
1729
  // Declared outside the try: on a mid-loop fetch error the previous
1718
1730
  // page's apply may still be in flight, and the catch (410 reset →
1719
1731
  // re-pull) MUST let it settle first — otherwise the orphaned apply
@@ -1803,14 +1815,14 @@ export class SyncEngine {
1803
1815
  // truth. Record it so onConnected skips the reconcile that would
1804
1816
  // otherwise re-fetch every entity via cursor pagination.
1805
1817
  this.lastPullStartedFromZero = startedFromZero;
1806
- // Snapshot landed cleanly → replay the live events we held during it, in
1807
- // arrival (≈seq) order. They filter against the now-correct cursor
1808
- // (snapshot_seq), so events newer than the snapshot apply and older ones
1809
- // (already in the snapshot) are deduped. Clearing `snapshotHold` first
1810
- // means this replay applies normally (it isn't re-held).
1811
- if (startedFromZero && this.snapshotHold) {
1812
- const held = this.snapshotHold;
1813
- this.snapshotHold = null;
1818
+ // Pull landed cleanly → replay the live events we held during it, in
1819
+ // arrival (≈seq) order. They filter against the now-advanced cursor,
1820
+ // so events newer than what the pull delivered apply and older ones
1821
+ // (already delivered by the pull) are deduped. Clearing `pullHold`
1822
+ // first means this replay applies normally (it isn't re-held).
1823
+ if (this.pullHold) {
1824
+ const held = this.pullHold;
1825
+ this.pullHold = null;
1814
1826
  if (held.length > 0) await this.enqueueApply(held);
1815
1827
  }
1816
1828
  } catch (err) {
@@ -1856,6 +1868,10 @@ export class SyncEngine {
1856
1868
  // First resync of the episode — snapshot now. Bypass the queue
1857
1869
  // (we ARE the pull op holding the slot; the public pull()
1858
1870
  // would re-enqueue and share our own promise back → deadlock).
1871
+ // Discard this failed episode's held events first so the
1872
+ // nested pull opens (and owns) a fresh hold; the from-zero
1873
+ // snapshot it runs covers everything the buffer held.
1874
+ this.pullHold = null;
1859
1875
  await this.resetReplicaInner();
1860
1876
  await this.pullInner();
1861
1877
  } else {
@@ -1874,14 +1890,14 @@ export class SyncEngine {
1874
1890
  }
1875
1891
  }
1876
1892
  } finally {
1877
- // Snapshot pull failed (network error / 410 mid-fetch): DISCARD any
1878
- // still-held live events rather than applying them. The cursor stays at 0
1879
- // so the retry resnapshots and re-covers them; applying them here would
1880
- // advance the cursor and turn the retry into a gappy delta. On success
1881
- // the try already drained + nulled the hold, so this is a no-op there.
1882
- // Nested non-zero pulls never set `snapshotHold`, so this only fires for
1883
- // the from-zero snapshot that owns it.
1884
- if (startedFromZero && this.snapshotHold !== null) this.snapshotHold = null;
1893
+ // Pull failed (network error / 410 mid-fetch): DISCARD any still-held
1894
+ // live events rather than applying them. The cursor never reached the
1895
+ // held seqs, so the next pull re-fetches those events from the log;
1896
+ // applying them here would advance the cursor over the un-pulled gap
1897
+ // (the exact leapfrog the hold exists to prevent). On success the try
1898
+ // already drained + nulled the hold, so this is a no-op there — and
1899
+ // the 410 recursion drained or re-owned it before we got here.
1900
+ if (this.pullHold !== null) this.pullHold = null;
1885
1901
  }
1886
1902
  }
1887
1903
 
@@ -2153,16 +2169,20 @@ export class SyncEngine {
2153
2169
  ): Promise<{ rows: Row[]; truncated: boolean }> {
2154
2170
  const out: Row[] = [];
2155
2171
  let cursor: string | null = null;
2156
- // 200 pages × 100 per page = 20k rows. A `sync:true` entity larger than
2172
+ // 20 pages × 1000 per page = 20k rows. A `sync:true` entity larger than
2157
2173
  // this should switch to `sync: false` + search/by-id — see useInfiniteQuery.
2158
- for (let page = 0; page < 200; page++) {
2174
+ // 1000/page matches the sync-pull batch limits; the server only honors it
2175
+ // on replication fetches (`sync=1`), and each page costs a full round
2176
+ // trip, so the page size directly divides sweep latency (a 20k-row
2177
+ // entity was 200 sequential requests; now 20).
2178
+ for (let page = 0; page < 20; page++) {
2159
2179
  // `sync=1` marks this as a REPLICATION fetch, which is what makes an
2160
2180
  // entity's `sync` scope apply. Without it the server treats the request
2161
2181
  // as an ordinary app read and returns unscoped rows — correct for a
2162
2182
  // direct read, wrong here, since these rows land in the replica.
2163
2183
  const qs: string = cursor
2164
- ? `?limit=100&sync=1&after=${encodeURIComponent(cursor)}`
2165
- : `?limit=100&sync=1`;
2184
+ ? `?limit=1000&sync=1&after=${encodeURIComponent(cursor)}`
2185
+ : `?limit=1000&sync=1`;
2166
2186
  const resp: {
2167
2187
  data: Row[];
2168
2188
  next_cursor: string | null;
@@ -301,8 +301,9 @@ describe("SyncEngine.reconcile", () => {
301
301
 
302
302
  await engine.reconcile(["Event"]);
303
303
 
304
- // The fetch hit the 200-page safety cap — proof the set was truncated.
305
- expect(pages).toBe(200);
304
+ // The fetch hit the 20-page safety cap (20 × 1000/page = the same 20k
305
+ // row ceiling) — proof the set was truncated.
306
+ expect(pages).toBe(20);
306
307
  // The un-fetched local rows are NOT deleted (the bug would drop them).
307
308
  expect(engine.store.get("Event", "beyond_1")).not.toBeNull();
308
309
  expect(engine.store.get("Event", "beyond_2")).not.toBeNull();
@@ -623,6 +623,67 @@ describe("sync scenarios", () => {
623
623
  expect(overlapped).toBe(true);
624
624
  });
625
625
 
626
+ // WS LEAPFROG (pins the pullHold extension to delta pulls). A live WS
627
+ // frame landing MID-catch-up used to apply immediately and advance the
628
+ // cursor past the not-yet-pulled gap; when the pull then delivered the
629
+ // gap events, the monotonic seq filter dropped them — silently missing
630
+ // rows until a reconcile happened to sweep them. The hold buffers live
631
+ // frames for the pull's duration and replays them (seq-filtered) after
632
+ // it lands, so BOTH the gap rows and the live frame's row survive.
633
+ test("a live event mid-catch-up cannot leapfrog the cursor past unapplied pages", async () => {
634
+ let deltaFetches = 0;
635
+ let engineRef: SyncEngine | null = null;
636
+ env = createTestEnv({
637
+ transport: "poll",
638
+ beforePull: (_auth, since) => {
639
+ if (since > 0) {
640
+ deltaFetches += 1;
641
+ // Mid-catch-up (page 2 of 3), a "live WS frame" with a far
642
+ // newer seq arrives — exactly what onChangeEvent feeds into
643
+ // enqueueApply. Pre-fix this applied instantly, advanced the
644
+ // cursor to 10000, and pages 2–3 were filtered out on apply.
645
+ if (deltaFetches === 2 && engineRef) {
646
+ void (
647
+ engineRef as unknown as {
648
+ enqueueApply(c: unknown[]): Promise<void>;
649
+ }
650
+ ).enqueueApply([
651
+ {
652
+ seq: 10_000,
653
+ entity: "Note",
654
+ row_id: "live_1",
655
+ kind: "insert",
656
+ data: { id: "live_1", title: "live frame" },
657
+ timestamp: "",
658
+ },
659
+ ]);
660
+ }
661
+ }
662
+ },
663
+ });
664
+ env.signIn({ userId: "u1" });
665
+ env.server.seed("Note", [{ id: "n0", title: "seed" }]);
666
+ await env.start();
667
+ await env.flush();
668
+ engineRef = env.engine;
669
+
670
+ for (let i = 1; i <= 30; i++) {
671
+ env.server.insert("Note", { id: `n${i}`, title: `t${i}` });
672
+ }
673
+ env.server.deltaPageSize = 10;
674
+
675
+ await env.engine.pull();
676
+ await env.flush();
677
+
678
+ // Every gap page landed — including rows from the pages AFTER the
679
+ // live frame arrived (pre-fix these were silently dropped).
680
+ expect(env.engine.store.get("Note", "n15")).not.toBeNull();
681
+ expect(env.engine.store.get("Note", "n25")).not.toBeNull();
682
+ expect(env.engine.store.get("Note", "n30")).not.toBeNull();
683
+ // And the held live frame itself replayed after the pull.
684
+ expect(env.engine.store.get("Note", "live_1")).not.toBeNull();
685
+ });
686
+
626
687
  // OFFLINE WRITES (pins the transient/permanent split in pushInner).
627
688
  // A push that fails with a NETWORK error (offline — fetch rejects, no
628
689
  // HTTP status) must keep the mutation `pending` and the optimistic