cursedbelt-server 1.1.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,348 @@
1
+ /**
2
+ * **The peer says it has news. Nobody asks.**
3
+ *
4
+ * This is the half of the replication story that deleting the dialer (`./timer`,
5
+ * `./http`'s old `GET /requested`) would otherwise have left open, and it is the
6
+ * reason that deletion costs nothing.
7
+ *
8
+ * ## The three ways, and why this is the one
9
+ *
10
+ * Push-from-the-Mac answers Mac → peer completely: the side that has news calls
11
+ * `markDirty()` and pushes. It does not answer **peer → Mac**, and something must:
12
+ *
13
+ * | | |
14
+ * |---|---|
15
+ * | the Mac polls an outbox | ❌ polling. Ruled out 2026-09-15 |
16
+ * | the peer dials the Mac | ❌ needs the Mac reachable — a tunnel, an open port, a laptop that cannot sleep. That is the thing we are trying to delete |
17
+ * | 🟢 **the Mac holds a stream open and the peer pushes down it** | ✅ the Mac is still purely a client: it dials **out**, listens on nothing, and no timer fires |
18
+ *
19
+ * So: one long-lived outbound connection, opened by the dialing half, carrying a
20
+ * `news` frame whenever the receiver's op log grows. The Mac's reaction is
21
+ * `syncNow()` — the push it would have made anyway, at the moment there is a reason
22
+ * for it instead of 4,320 times a day on the chance.
23
+ *
24
+ * ## 🔴 Two honest caveats, neither of which is a poll
25
+ *
26
+ * · **A sleeping Mac drops the connection.** On wake it reconnects and catches up
27
+ * from `since` — the same catch-up a push model needs anyway, so it costs
28
+ * nothing extra. `since` is the receiver's head as the client last saw it.
29
+ * · **The reconnect backoff is a timer.** It runs *only while disconnected*, never
30
+ * against a healthy peer, and it is written below as an `await sleep()` in a
31
+ * loop rather than a `setTimeout` that calls `fetch` — the shape matters,
32
+ * because the second one is indistinguishable from the poll this file exists to
33
+ * replace. A connected client issues **exactly one** outbound request for as
34
+ * long as it stays connected, and `signal.spec.ts` measures that over 24
35
+ * simulated hours.
36
+ *
37
+ * ## Why this is not `cursedbelt-cc`'s `serverPush.ts`
38
+ *
39
+ * `libs/cursedbelt-cc/src/server/stream/serverPush.ts` is the real tenant-scoped
40
+ * push tier — heartbeats, backpressure, SSE — and on shape it is exactly this.
41
+ * It is not imported here for a reason that survives the preference:
42
+ * `cursedbelt-server/sync` is a **leaf** whose only runtime dependency is `hono`
43
+ * (see `./index.ts`), and `cursedbelt-cc` is a separate published package with zero
44
+ * consumers today, parked whole. Importing it would make every app that wants an
45
+ * op log install the cc platform to get one — the precise trade `./index.ts` was
46
+ * split to refuse, and the one `leafSubpathsImportNothing.spec.ts` now gates.
47
+ *
48
+ * The hub below is therefore the same model at a tenth the surface: this channel
49
+ * carries one frame kind, to one peer, with no tenant dimension. If `cursedbelt-cc`
50
+ * ever gains consumers and the two packages can depend on each other, {@link
51
+ * SyncSignalHub} is a strict subset of `ServerPushHub` and swaps in without
52
+ * touching a producer.
53
+ *
54
+ * ## The Workers path, named and not required yet
55
+ *
56
+ * On Cloudflare the durable form of "the Mac holds a connection all day" is a
57
+ * **Durable Object with WebSocket Hibernation**: the socket stays open while the
58
+ * object is evicted from memory, so an idle connection is not billed for duration.
59
+ * That is what keeps this free at rest, and it is the shape task `195` (the Mac is
60
+ * a client, never an origin) lands on. Nothing here needs it today — SSE over the
61
+ * existing receiver works on Bun and on Workers alike — but the frame format below
62
+ * is deliberately transport-free so the move is a transport swap, not a protocol
63
+ * change.
64
+ */
65
+ /** SSE frame headers. `x-accel-buffering` keeps nginx from holding the stream in a
66
+ * proxy buffer, which turns a live channel into a silent one — the same trap that
67
+ * makes a cached `200` sit over a dead origin elsewhere in this fleet. */
68
+ export const SIGNAL_SSE_HEADERS = Object.freeze({
69
+ "content-type": "text/event-stream",
70
+ "cache-control": "no-store",
71
+ connection: "keep-alive",
72
+ "x-accel-buffering": "no",
73
+ });
74
+ /**
75
+ * Comment-frame interval. Without it an idle stream dies silently behind a proxy
76
+ * with an idle timeout and the client never learns it should reconnect.
77
+ *
78
+ * 🔴 This is a heartbeat on an ALREADY-OPEN socket, not a request. It is row (c) of
79
+ * the 2026-09-15 ruling — a local housekeeping timer that issues no outbound
80
+ * request — and it is why that row had to be carved out explicitly: a rule that
81
+ * said "no `setInterval`" would have deleted the one mechanism that keeps a
82
+ * no-polling design alive.
83
+ */
84
+ const HEARTBEAT_MS = 15_000;
85
+ export function createSyncSignalHub() {
86
+ const listeners = new Set();
87
+ let closed = false;
88
+ return {
89
+ announce(news) {
90
+ if (closed)
91
+ return;
92
+ // Snapshot: a listener that unsubscribes itself mid-fanout (the normal
93
+ // teardown path) must not mutate the set being iterated.
94
+ for (const listener of [...listeners]) {
95
+ try {
96
+ listener(news);
97
+ }
98
+ catch {
99
+ // A broken transport is that subscriber's problem, never the
100
+ // producer's: drop it and keep fanning out. An `announce` may be
101
+ // called from inside an op-log write, so it can never throw upward.
102
+ listeners.delete(listener);
103
+ }
104
+ }
105
+ },
106
+ subscribe(listener) {
107
+ if (closed)
108
+ return () => undefined;
109
+ listeners.add(listener);
110
+ return () => {
111
+ listeners.delete(listener);
112
+ };
113
+ },
114
+ get subscriberCount() {
115
+ return listeners.size;
116
+ },
117
+ close() {
118
+ if (closed)
119
+ return;
120
+ closed = true;
121
+ listeners.clear();
122
+ },
123
+ };
124
+ }
125
+ /**
126
+ * Bridge a hub subscription onto a `ReadableStream` of SSE bytes.
127
+ *
128
+ * The stream ends when the client disconnects (`signal`) or the hub closes; both
129
+ * paths run the same teardown, so a dropped connection cannot leave a subscription
130
+ * — or an interval — behind.
131
+ */
132
+ export function signalStream(hub, options = {}) {
133
+ const heartbeatMs = options.heartbeatMs ?? HEARTBEAT_MS;
134
+ const setIntervalFn = options.setIntervalFn ?? ((fn, ms) => setInterval(fn, ms));
135
+ const clearIntervalFn = options.clearIntervalFn ??
136
+ ((handle) => clearInterval(handle));
137
+ const encoder = new TextEncoder();
138
+ let unsubscribe = () => undefined;
139
+ let heartbeat = null;
140
+ let done = false;
141
+ return new ReadableStream({
142
+ start(controller) {
143
+ const teardown = () => {
144
+ if (done)
145
+ return;
146
+ done = true;
147
+ unsubscribe();
148
+ if (heartbeat !== null)
149
+ clearIntervalFn(heartbeat);
150
+ heartbeat = null;
151
+ try {
152
+ controller.close();
153
+ }
154
+ catch {
155
+ // Already closed by the runtime when the socket died — expected.
156
+ }
157
+ };
158
+ const write = (chunk) => {
159
+ if (done)
160
+ return;
161
+ try {
162
+ controller.enqueue(encoder.encode(chunk));
163
+ }
164
+ catch {
165
+ teardown();
166
+ }
167
+ };
168
+ if (options.initial)
169
+ write(encodeNewsFrame(options.initial));
170
+ unsubscribe = hub.subscribe((news) => {
171
+ write(encodeNewsFrame(news));
172
+ });
173
+ heartbeat = setIntervalFn(() => {
174
+ // A comment frame. It wakes no client-side handler and issues no
175
+ // request — it keeps proxies from calling the connection idle.
176
+ write(": keep-alive\n\n");
177
+ }, heartbeatMs);
178
+ if (options.signal) {
179
+ if (options.signal.aborted)
180
+ teardown();
181
+ else
182
+ options.signal.addEventListener("abort", teardown, { once: true });
183
+ }
184
+ },
185
+ cancel() {
186
+ done = true;
187
+ unsubscribe();
188
+ if (heartbeat !== null)
189
+ clearIntervalFn(heartbeat);
190
+ heartbeat = null;
191
+ },
192
+ });
193
+ }
194
+ /** Serialize one announcement as an SSE frame. */
195
+ export const encodeNewsFrame = (news) => `event: news\ndata: ${JSON.stringify(news)}\n\n`;
196
+ /** A ready-to-return SSE `Response` — what the receiver's `GET /signal` answers. */
197
+ export const signalResponse = (hub, options = {}) => new Response(signalStream(hub, options), { headers: { ...SIGNAL_SSE_HEADERS } });
198
+ /**
199
+ * Hold one stream open against the peer, calling {@link SignalClientOptions.onNews}
200
+ * on every `news` frame, and reconnect with jittered backoff **while disconnected**.
201
+ *
202
+ * 🔴 Read the loop shape as load-bearing. The reconnect is `await sleep(ms)` inside
203
+ * an async loop, NOT a `setTimeout` whose callback calls `fetch`. The second shape
204
+ * is byte-for-byte what a poll looks like, and a reviewer — or a checker — cannot
205
+ * tell the two apart once it is written that way. Here, a connected client is
206
+ * parked on `reader.read()` and issues nothing; `requestCount` stops moving the
207
+ * moment a connection succeeds, and that is what `signal.spec.ts` asserts over 24
208
+ * simulated hours.
209
+ *
210
+ * The jitter is not decoration either: a fixed backoff across N clients recovering
211
+ * from the same outage is a synchronized stampede at the instant the peer comes
212
+ * back up — the worst possible moment to send it a thundering herd.
213
+ */
214
+ export function connectSyncSignal(options) {
215
+ const doFetch = options.fetchImpl ?? fetch;
216
+ const base = options.basePath.replace(/\/$/, "");
217
+ const backoffBase = options.backoffBaseMs ?? 1_000;
218
+ const backoffMax = options.backoffMaxMs ?? 5 * 60_000;
219
+ const sleep = options.sleepFn ?? ((ms) => new Promise((r) => setTimeout(r, ms)));
220
+ const jitter = options.jitterFn ?? Math.random;
221
+ const log = options.log ?? (() => undefined);
222
+ let stopped = false;
223
+ let connected = false;
224
+ let requests = 0;
225
+ let failures = 0;
226
+ let abort = null;
227
+ /** Open one stream and stay on it until it ends. Throws on any failure. */
228
+ const runOnce = async () => {
229
+ const controller = new AbortController();
230
+ abort = controller;
231
+ requests += 1;
232
+ const response = await doFetch(`${base}/signal?peer=${encodeURIComponent(options.selfId)}`, {
233
+ // 🔴 A followed redirect is how a moved hostname became eight days of
234
+ // silence on the `/pull` side — see `./http.ts`'s header. Same rule here.
235
+ redirect: "manual",
236
+ headers: {
237
+ authorization: `Bearer ${options.token}`,
238
+ "x-sync-peer": options.selfId,
239
+ accept: "text/event-stream",
240
+ },
241
+ signal: controller.signal,
242
+ });
243
+ if (response.status >= 300 && response.status < 400) {
244
+ throw new Error(`sync/signal: ${response.status} — the peer address redirects to ` +
245
+ `${response.headers.get("location") ?? "an unnamed destination"}.`);
246
+ }
247
+ if (!response.ok)
248
+ throw new Error(`sync/signal: ${response.status}`);
249
+ if (!response.body)
250
+ throw new Error("sync/signal: the peer sent no stream body");
251
+ connected = true;
252
+ failures = 0;
253
+ const reader = response.body.getReader();
254
+ const decoder = new TextDecoder();
255
+ let buffer = "";
256
+ try {
257
+ while (!stopped) {
258
+ const { done, value } = await reader.read();
259
+ if (done)
260
+ break;
261
+ buffer += decoder.decode(value, { stream: true });
262
+ // SSE frames are separated by a blank line. A partial tail stays in
263
+ // the buffer until the rest of it arrives.
264
+ let split = buffer.indexOf("\n\n");
265
+ while (split !== -1) {
266
+ const frame = buffer.slice(0, split);
267
+ buffer = buffer.slice(split + 2);
268
+ const news = parseNewsFrame(frame);
269
+ if (news)
270
+ options.onNews(news);
271
+ split = buffer.indexOf("\n\n");
272
+ }
273
+ }
274
+ }
275
+ finally {
276
+ connected = false;
277
+ try {
278
+ await reader.cancel();
279
+ }
280
+ catch {
281
+ // The socket is already gone — that is the normal end of a stream.
282
+ }
283
+ }
284
+ };
285
+ /**
286
+ * The supervisor. Every `await` here is either the open stream itself or a sleep
287
+ * between failures; there is no path that waits on a healthy peer in order to
288
+ * ask it something.
289
+ */
290
+ const supervise = async () => {
291
+ while (!stopped) {
292
+ try {
293
+ await runOnce();
294
+ }
295
+ catch (error) {
296
+ if (stopped)
297
+ return;
298
+ failures += 1;
299
+ if (failures === 1) {
300
+ log(`[sync] signal stream lost — reconnecting. This is normal (peer asleep / offline): ` +
301
+ `${error instanceof Error ? error.message : String(error)}`);
302
+ }
303
+ }
304
+ if (stopped)
305
+ return;
306
+ const step = Math.min(backoffMax, backoffBase * 2 ** Math.max(0, failures - 1));
307
+ await sleep(Math.round(step * (0.5 + jitter() * 0.5)));
308
+ }
309
+ };
310
+ void supervise();
311
+ return {
312
+ stop() {
313
+ stopped = true;
314
+ abort?.abort();
315
+ abort = null;
316
+ },
317
+ get connected() {
318
+ return connected;
319
+ },
320
+ get requestCount() {
321
+ return requests;
322
+ },
323
+ };
324
+ }
325
+ /** `event: news\ndata: {...}` → {@link SyncNews}, or null for a heartbeat/unknown. */
326
+ export function parseNewsFrame(frame) {
327
+ let isNews = false;
328
+ let data = null;
329
+ for (const line of frame.split("\n")) {
330
+ if (line.startsWith(":"))
331
+ continue; // comment frame — the heartbeat
332
+ if (line.startsWith("event:"))
333
+ isNews = line.slice(6).trim() === "news";
334
+ else if (line.startsWith("data:"))
335
+ data = line.slice(5).trim();
336
+ }
337
+ if (!isNews || data === null)
338
+ return null;
339
+ try {
340
+ const parsed = JSON.parse(data);
341
+ if (typeof parsed.head !== "number" || typeof parsed.instanceId !== "string")
342
+ return null;
343
+ return { head: parsed.head, instanceId: parsed.instanceId };
344
+ }
345
+ catch {
346
+ return null;
347
+ }
348
+ }
@@ -1,20 +1,63 @@
1
1
  /**
2
- * The in-process dialer loop — vault's `syncTimer` generalized. Runs on the dialing
3
- * half only; a failure NEVER takes the host daemon down (an unsynced instance still
4
- * serves and still accepts writes — that is the point of two independently-writable
5
- * instances). The timer is `unref`'d so it never holds the process open.
6
- *
7
- * The peer is re-resolved on EVERY tick, so linking (or revoking) a peer takes
8
- * effect without a restart. The "Sync now" fast lane polls the receiver's request
9
- * stamp every ~20s while healthy, so a button press on the far side means seconds,
10
- * not "sometime in the next interval".
2
+ * The in-process sync loop. Runs on the dialing half only; a failure NEVER takes
3
+ * the host daemon down (an unsynced instance still serves and still accepts writes
4
+ * — that is the point of two independently-writable instances). The timer is
5
+ * `unref`'d so it never holds the process open.
6
+ *
7
+ * ── 🔴 This loop does not poll, and it may never be made to again ────────────────
8
+ *
9
+ * Until 2026-09-15 this file was a dialer twice over: a `GET /requested` probe at a
10
+ * peer every 20 s while healthy (`REQUEST_POLL_MS`), and a full reconciliation every
11
+ * 5 minutes (`DEFAULT_LOOP.intervalMs`) whether or not either side had news. The
12
+ * owner ruled both out — *"There should be no polling in cb unless you can make some
13
+ * good case for it that beats the api option"* — and the case against was asked for
14
+ * and not found. Measured in `vault`: the 20 s probe alone was **68 % of that app's
15
+ * entire traffic, one request every 22 seconds, for an app with one user.**
16
+ *
17
+ * Both are gone. Every wake-up this loop books now carries a {@link WakeReason}, and
18
+ * that union is deliberately closed with no periodic member — there is no value you
19
+ * can pass to {@link schedule} that means "again in a while". That makes the poll
20
+ * unrepresentable rather than merely discouraged, which matters because
21
+ * `REQUEST_POLL_MS` did not survive as a habit; it survived as **exported public
22
+ * API**, and the next consumer to import it would have made its removal a breaking
23
+ * change instead of an edit.
24
+ *
25
+ * What replaced each half:
26
+ *
27
+ * · **Mac → peer** is a PUSH. `markDirty()` after a local write books one
28
+ * debounced run: the side that has news says so, which is what "sync now"
29
+ * should always have been.
30
+ * · **peer → Mac** is `./signal` — the dialing half holds one long-lived stream
31
+ * open and the receiver writes a frame when its log grows. The Mac is still
32
+ * purely a client: it dials OUT, listens on nothing, and no timer fires.
33
+ *
34
+ * 🔴 **`markDirty()` is now REQUIRED wiring, not an optimization.** With the steady
35
+ * interval gone it is the only thing that pushes a local write. A consumer that
36
+ * takes the handle and drops `markDirty` (which `apps/vault` did while the interval
37
+ * still covered for it) will sit clean and silent forever.
38
+ *
39
+ * The peer is re-resolved on every run, so revoking a link takes effect without a
40
+ * restart. Linking a NEW peer is a user action and is noticed at the next
41
+ * `syncNow()` / `markDirty()` rather than by a timer that was watching for it —
42
+ * call `syncNow()` after a link and it is immediate.
11
43
  */
12
44
  import { type SyncLoopConfig } from "./planner";
13
45
  import type { SyncStatusReporter } from "./status";
14
46
  import type { SyncSummary } from "./types";
15
- export declare const REQUEST_POLL_MS = 20000;
47
+ /**
48
+ * 🔴 Every reason this loop is allowed to wake up. There is deliberately no
49
+ * `"interval"` / `"poll"` member, and adding one is the change this whole file
50
+ * exists to refuse.
51
+ *
52
+ * · `boot` — one catch-up run at start.
53
+ * · `dirty` — local ops are waiting to be pushed (debounced).
54
+ * · `forced` — `syncNow()`, or a frame off `./signal` saying the peer has news.
55
+ * · `retry` — the last attempt failed. Runs only while disconnected, never
56
+ * against a healthy peer; that is a reconnect backoff, not a poll.
57
+ */
58
+ export type WakeReason = "boot" | "dirty" | "forced" | "retry";
16
59
  export interface SyncTimerDeps {
17
- /** Re-read every tick: null = not linked (quiet no-op, re-checked next tick). */
60
+ /** Re-read on every run: null = not linked (quiet no-op). */
18
61
  resolvePeer: () => {
19
62
  basePath: string;
20
63
  token: string;
@@ -26,14 +69,8 @@ export interface SyncTimerDeps {
26
69
  }) => Promise<SyncSummary>;
27
70
  /** The local op-log head — a head beyond the last synced one marks dirty. */
28
71
  head: () => number;
29
- /** Ask the peer whether its "Sync now" was pressed; null on any failure. */
30
- checkRequest?: (peer: {
31
- basePath: string;
32
- token: string;
33
- }) => Promise<number | null>;
34
72
  status?: SyncStatusReporter;
35
73
  loop?: SyncLoopConfig;
36
- requestPollMs?: number;
37
74
  now?: () => number;
38
75
  setTimer?: (fn: () => void, ms: number) => unknown;
39
76
  clearTimer?: (handle: unknown) => void;
@@ -44,9 +81,16 @@ export interface SyncTimerDeps {
44
81
  }
45
82
  export interface SyncTimerHandle {
46
83
  stop(): void;
47
- /** Run the next tick immediately (the "Sync now" button on the dialing half). */
84
+ /**
85
+ * Run now. The "Sync now" button on the dialing half, and the handler for a
86
+ * `./signal` frame — both are "something has news", which is a push.
87
+ */
48
88
  syncNow(): void;
49
- /** Mark local writes pending (call after any local op) — debounces a burst. */
89
+ /**
90
+ * Mark local writes pending — call after any local op. 🔴 Required wiring: with
91
+ * no steady interval this is what pushes a local write. A burst coalesces into
92
+ * one run via the planner's debounce.
93
+ */
50
94
  markDirty(): void;
51
95
  }
52
96
  export declare function startSyncTimer(deps: SyncTimerDeps): SyncTimerHandle;
@@ -1,16 +1,52 @@
1
1
  /**
2
- * The in-process dialer loop — vault's `syncTimer` generalized. Runs on the dialing
3
- * half only; a failure NEVER takes the host daemon down (an unsynced instance still
4
- * serves and still accepts writes — that is the point of two independently-writable
5
- * instances). The timer is `unref`'d so it never holds the process open.
2
+ * The in-process sync loop. Runs on the dialing half only; a failure NEVER takes
3
+ * the host daemon down (an unsynced instance still serves and still accepts writes
4
+ * — that is the point of two independently-writable instances). The timer is
5
+ * `unref`'d so it never holds the process open.
6
6
  *
7
- * The peer is re-resolved on EVERY tick, so linking (or revoking) a peer takes
8
- * effect without a restart. The "Sync now" fast lane polls the receiver's request
9
- * stamp every ~20s while healthy, so a button press on the far side means seconds,
10
- * not "sometime in the next interval".
7
+ * ── 🔴 This loop does not poll, and it may never be made to again ────────────────
8
+ *
9
+ * Until 2026-09-15 this file was a dialer twice over: a `GET /requested` probe at a
10
+ * peer every 20 s while healthy (`REQUEST_POLL_MS`), and a full reconciliation every
11
+ * 5 minutes (`DEFAULT_LOOP.intervalMs`) whether or not either side had news. The
12
+ * owner ruled both out — *"There should be no polling in cb unless you can make some
13
+ * good case for it that beats the api option"* — and the case against was asked for
14
+ * and not found. Measured in `vault`: the 20 s probe alone was **68 % of that app's
15
+ * entire traffic, one request every 22 seconds, for an app with one user.**
16
+ *
17
+ * Both are gone. Every wake-up this loop books now carries a {@link WakeReason}, and
18
+ * that union is deliberately closed with no periodic member — there is no value you
19
+ * can pass to {@link schedule} that means "again in a while". That makes the poll
20
+ * unrepresentable rather than merely discouraged, which matters because
21
+ * `REQUEST_POLL_MS` did not survive as a habit; it survived as **exported public
22
+ * API**, and the next consumer to import it would have made its removal a breaking
23
+ * change instead of an edit.
24
+ *
25
+ * What replaced each half:
26
+ *
27
+ * · **Mac → peer** is a PUSH. `markDirty()` after a local write books one
28
+ * debounced run: the side that has news says so, which is what "sync now"
29
+ * should always have been.
30
+ * · **peer → Mac** is `./signal` — the dialing half holds one long-lived stream
31
+ * open and the receiver writes a frame when its log grows. The Mac is still
32
+ * purely a client: it dials OUT, listens on nothing, and no timer fires.
33
+ *
34
+ * 🔴 **`markDirty()` is now REQUIRED wiring, not an optimization.** With the steady
35
+ * interval gone it is the only thing that pushes a local write. A consumer that
36
+ * takes the handle and drops `markDirty` (which `apps/vault` did while the interval
37
+ * still covered for it) will sit clean and silent forever.
38
+ *
39
+ * The peer is re-resolved on every run, so revoking a link takes effect without a
40
+ * restart. Linking a NEW peer is a user action and is noticed at the next
41
+ * `syncNow()` / `markDirty()` rather than by a timer that was watching for it —
42
+ * call `syncNow()` after a link and it is immediate.
11
43
  */
12
44
  import { isUnreachable, DEFAULT_LOOP, planNextSync, } from "./planner";
13
- export const REQUEST_POLL_MS = 20_000;
45
+ /**
46
+ * One reconciliation shortly after start: catch up on whatever happened while this
47
+ * process was down. A single shot at boot, not a cadence — the same catch-up a push
48
+ * model needs anyway.
49
+ */
14
50
  const KICKOFF_MS = 5_000;
15
51
  function defaultPeerLabel(basePath) {
16
52
  try {
@@ -29,46 +65,63 @@ export function startSyncTimer(deps) {
29
65
  const warn = deps.warn ?? ((m) => console.error(m));
30
66
  const status = deps.status;
31
67
  const peerLabel = deps.peerLabel ?? defaultPeerLabel;
32
- const requestPollMs = deps.requestPollMs ?? REQUEST_POLL_MS;
33
68
  status?.update({ enabled: true, role: "dialer" });
34
69
  const state = { lastAttempt: 0, consecutiveFailures: 0, dirtySince: null };
35
70
  let syncedHead = deps.head();
36
71
  let stopped = false;
37
72
  let handle = null;
38
73
  let reportedUnreachable = false;
39
- let forceRun = false;
40
- let handledRequestAt = 0;
41
- let probing = false;
74
+ /**
75
+ * 🔴 Starts TRUE: the boot catch-up is a forced run, not a scheduled one.
76
+ *
77
+ * It used to fall out of the planner for free — `lastAttempt: 0` plus a steady
78
+ * `intervalMs` made the first tick always due. With no interval left, a clean
79
+ * healthy state plans nothing, so the one run this loop genuinely owes at
80
+ * startup has to be asked for explicitly. It has a reason: this process may have
81
+ * missed ops while it was down, and catching up is exactly what a push model
82
+ * does on reconnect.
83
+ */
84
+ let forceRun = true;
42
85
  let inFlight = false;
43
- const schedule = (ms) => {
86
+ /** Why the currently-booked wake exists, or null when the loop is quiet. */
87
+ let bookedFor = null;
88
+ /**
89
+ * Book the next wake-up. 🔴 The `reason` is not decoration — it is the type-level
90
+ * guard: {@link WakeReason} has no periodic member, so there is no way to spell
91
+ * "wake again in `intervalMs`" through this function. It is also read back by
92
+ * {@link SyncTimerHandle.markDirty}, which must not push an imminent forced run
93
+ * out to a debounce.
94
+ */
95
+ const schedule = (reason, ms) => {
44
96
  if (stopped)
45
97
  return;
46
98
  if (handle !== null)
47
99
  clearTimer(handle);
48
100
  handle = setTimer(tick, ms);
49
101
  handle?.unref?.();
102
+ bookedFor = reason;
50
103
  };
51
- const sleepFor = (waitMs) => {
52
- const healthy = state.consecutiveFailures === 0 && requestPollMs > 0;
53
- return Math.max(1, healthy ? Math.min(waitMs, requestPollMs) : waitMs);
104
+ /** Nothing to push, nothing to retry: cancel any pending wake and go quiet. */
105
+ const idle = () => {
106
+ if (handle !== null)
107
+ clearTimer(handle);
108
+ handle = null;
109
+ bookedFor = null;
54
110
  };
55
- const probeRequest = (peer) => {
56
- if (probing || requestPollMs === 0 || !deps.checkRequest)
111
+ /** Apply the planner's verdict — the ONE place a wake-up is booked or declined. */
112
+ const bookNext = () => {
113
+ if (stopped)
57
114
  return;
58
- probing = true;
59
- void deps
60
- .checkRequest(peer)
61
- .then((requestedAt) => {
62
- if (stopped || requestedAt === null || requestedAt <= handledRequestAt)
63
- return;
64
- handledRequestAt = requestedAt;
65
- log("[sync] sync requested from the peer — running now.");
66
- forceRun = true;
67
- schedule(1);
68
- })
69
- .finally(() => {
70
- probing = false;
71
- });
115
+ if (forceRun) {
116
+ schedule("forced", 1);
117
+ return;
118
+ }
119
+ const plan = planNextSync(state, now(), loop);
120
+ if (plan.waitMs === null || plan.reason === null) {
121
+ idle();
122
+ return;
123
+ }
124
+ schedule(plan.reason, Math.max(1, plan.waitMs));
72
125
  };
73
126
  const tick = () => {
74
127
  if (stopped)
@@ -84,15 +137,16 @@ export function startSyncTimer(deps) {
84
137
  pending: Math.max(0, head - syncedHead),
85
138
  });
86
139
  if (peer === null) {
87
- schedule(loop.intervalMs);
140
+ // Not linked. Nothing to dial and nothing to wait for — `syncNow()` or the
141
+ // next local write re-checks. A re-check timer here would be a poll at a
142
+ // peer that does not exist yet.
143
+ forceRun = false;
144
+ idle();
88
145
  return;
89
146
  }
90
147
  const plan = planNextSync(state, now(), loop);
91
148
  if (!plan.runNow && !forceRun) {
92
- const linkedAndHealthy = state.consecutiveFailures === 0 && requestPollMs > 0;
93
- schedule(sleepFor(plan.waitMs));
94
- if (linkedAndHealthy)
95
- probeRequest(peer);
149
+ bookNext();
96
150
  return;
97
151
  }
98
152
  forceRun = false;
@@ -140,31 +194,36 @@ export function startSyncTimer(deps) {
140
194
  })
141
195
  .finally(() => {
142
196
  inFlight = false;
143
- if (forceRun)
144
- schedule(1);
145
- else
146
- schedule(sleepFor(Math.max(1_000, planNextSync(state, now(), loop).waitMs)));
197
+ // A local write that landed mid-run is news the planner must see.
198
+ if (deps.head() > syncedHead && state.dirtySince === null)
199
+ state.dirtySince = now();
200
+ bookNext();
147
201
  });
148
202
  };
149
- schedule(KICKOFF_MS);
203
+ schedule("boot", KICKOFF_MS);
150
204
  return {
151
205
  stop: () => {
152
206
  stopped = true;
153
207
  if (handle !== null)
154
208
  clearTimer(handle);
209
+ handle = null;
155
210
  },
156
211
  syncNow: () => {
157
212
  if (stopped)
158
213
  return;
159
214
  forceRun = true;
160
- schedule(1);
215
+ schedule("forced", 1);
161
216
  },
162
217
  markDirty: () => {
163
218
  if (stopped)
164
219
  return;
165
220
  if (state.dirtySince === null)
166
221
  state.dirtySince = now();
167
- schedule(1);
222
+ // A forced run is already due in ~1ms and will push these ops; re-booking
223
+ // would replace it with a 3s debounce and make "Sync now" slower.
224
+ if (bookedFor === "forced")
225
+ return;
226
+ bookNext();
168
227
  },
169
228
  };
170
229
  }