cursedbelt-server 1.1.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/server/sync/http.d.ts +20 -3
- package/dist/server/sync/http.js +20 -14
- package/dist/server/sync/index.d.ts +10 -2
- package/dist/server/sync/index.js +9 -1
- package/dist/server/sync/planner.d.ts +38 -8
- package/dist/server/sync/planner.js +32 -8
- package/dist/server/sync/signal.d.ts +161 -0
- package/dist/server/sync/signal.js +348 -0
- package/dist/server/sync/timer.d.ts +63 -19
- package/dist/server/sync/timer.js +104 -45
- package/dist/server/sync/types.d.ts +0 -2
- package/package.json +1 -1
- package/src/noTimerDialsAPeer.spec.ts +469 -0
- package/src/server/sync/http.ts +31 -16
- package/src/server/sync/index.ts +23 -1
- package/src/server/sync/planner.spec.ts +33 -16
- package/src/server/sync/planner.ts +48 -11
- package/src/server/sync/signal.spec.ts +306 -0
- package/src/server/sync/signal.ts +422 -0
- package/src/server/sync/timer.spec.ts +97 -16
- package/src/server/sync/timer.ts +124 -47
- package/src/server/sync/types.ts +0 -2
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* **The peer says it has news. Nobody asks.**
|
|
3
|
+
*
|
|
4
|
+
* This is the half of the replication story that deleting the dialer (`./timer`,
|
|
5
|
+
* `./http`'s old `GET /requested`) would otherwise have left open, and it is the
|
|
6
|
+
* reason that deletion costs nothing.
|
|
7
|
+
*
|
|
8
|
+
* ## The three ways, and why this is the one
|
|
9
|
+
*
|
|
10
|
+
* Push-from-the-Mac answers Mac → peer completely: the side that has news calls
|
|
11
|
+
* `markDirty()` and pushes. It does not answer **peer → Mac**, and something must:
|
|
12
|
+
*
|
|
13
|
+
* | | |
|
|
14
|
+
* |---|---|
|
|
15
|
+
* | the Mac polls an outbox | ❌ polling. Ruled out 2026-09-15 |
|
|
16
|
+
* | the peer dials the Mac | ❌ needs the Mac reachable — a tunnel, an open port, a laptop that cannot sleep. That is the thing we are trying to delete |
|
|
17
|
+
* | 🟢 **the Mac holds a stream open and the peer pushes down it** | ✅ the Mac is still purely a client: it dials **out**, listens on nothing, and no timer fires |
|
|
18
|
+
*
|
|
19
|
+
* So: one long-lived outbound connection, opened by the dialing half, carrying a
|
|
20
|
+
* `news` frame whenever the receiver's op log grows. The Mac's reaction is
|
|
21
|
+
* `syncNow()` — the push it would have made anyway, at the moment there is a reason
|
|
22
|
+
* for it instead of 4,320 times a day on the chance.
|
|
23
|
+
*
|
|
24
|
+
* ## 🔴 Two honest caveats, neither of which is a poll
|
|
25
|
+
*
|
|
26
|
+
* · **A sleeping Mac drops the connection.** On wake it reconnects and catches up
|
|
27
|
+
* from `since` — the same catch-up a push model needs anyway, so it costs
|
|
28
|
+
* nothing extra. `since` is the receiver's head as the client last saw it.
|
|
29
|
+
* · **The reconnect backoff is a timer.** It runs *only while disconnected*, never
|
|
30
|
+
* against a healthy peer, and it is written below as an `await sleep()` in a
|
|
31
|
+
* loop rather than a `setTimeout` that calls `fetch` — the shape matters,
|
|
32
|
+
* because the second one is indistinguishable from the poll this file exists to
|
|
33
|
+
* replace. A connected client issues **exactly one** outbound request for as
|
|
34
|
+
* long as it stays connected, and `signal.spec.ts` measures that over 24
|
|
35
|
+
* simulated hours.
|
|
36
|
+
*
|
|
37
|
+
* ## Why this is not `cursedbelt-cc`'s `serverPush.ts`
|
|
38
|
+
*
|
|
39
|
+
* `libs/cursedbelt-cc/src/server/stream/serverPush.ts` is the real tenant-scoped
|
|
40
|
+
* push tier — heartbeats, backpressure, SSE — and on shape it is exactly this.
|
|
41
|
+
* It is not imported here for a reason that survives the preference:
|
|
42
|
+
* `cursedbelt-server/sync` is a **leaf** whose only runtime dependency is `hono`
|
|
43
|
+
* (see `./index.ts`), and `cursedbelt-cc` is a separate published package with zero
|
|
44
|
+
* consumers today, parked whole. Importing it would make every app that wants an
|
|
45
|
+
* op log install the cc platform to get one — the precise trade `./index.ts` was
|
|
46
|
+
* split to refuse, and the one `leafSubpathsImportNothing.spec.ts` now gates.
|
|
47
|
+
*
|
|
48
|
+
* The hub below is therefore the same model at a tenth the surface: this channel
|
|
49
|
+
* carries one frame kind, to one peer, with no tenant dimension. If `cursedbelt-cc`
|
|
50
|
+
* ever gains consumers and the two packages can depend on each other, {@link
|
|
51
|
+
* SyncSignalHub} is a strict subset of `ServerPushHub` and swaps in without
|
|
52
|
+
* touching a producer.
|
|
53
|
+
*
|
|
54
|
+
* ## The Workers path, named and not required yet
|
|
55
|
+
*
|
|
56
|
+
* On Cloudflare the durable form of "the Mac holds a connection all day" is a
|
|
57
|
+
* **Durable Object with WebSocket Hibernation**: the socket stays open while the
|
|
58
|
+
* object is evicted from memory, so an idle connection is not billed for duration.
|
|
59
|
+
* That is what keeps this free at rest, and it is the shape task `195` (the Mac is
|
|
60
|
+
* a client, never an origin) lands on. Nothing here needs it today — SSE over the
|
|
61
|
+
* existing receiver works on Bun and on Workers alike — but the frame format below
|
|
62
|
+
* is deliberately transport-free so the move is a transport swap, not a protocol
|
|
63
|
+
* change.
|
|
64
|
+
*/
|
|
65
|
+
/** SSE frame headers. `x-accel-buffering` keeps nginx from holding the stream in a
|
|
66
|
+
* proxy buffer, which turns a live channel into a silent one — the same trap that
|
|
67
|
+
* makes a cached `200` sit over a dead origin elsewhere in this fleet. */
|
|
68
|
+
export const SIGNAL_SSE_HEADERS = Object.freeze({
|
|
69
|
+
"content-type": "text/event-stream",
|
|
70
|
+
"cache-control": "no-store",
|
|
71
|
+
connection: "keep-alive",
|
|
72
|
+
"x-accel-buffering": "no",
|
|
73
|
+
});
|
|
74
|
+
/**
|
|
75
|
+
* Comment-frame interval. Without it an idle stream dies silently behind a proxy
|
|
76
|
+
* with an idle timeout and the client never learns it should reconnect.
|
|
77
|
+
*
|
|
78
|
+
* 🔴 This is a heartbeat on an ALREADY-OPEN socket, not a request. It is row (c) of
|
|
79
|
+
* the 2026-09-15 ruling — a local housekeeping timer that issues no outbound
|
|
80
|
+
* request — and it is why that row had to be carved out explicitly: a rule that
|
|
81
|
+
* said "no `setInterval`" would have deleted the one mechanism that keeps a
|
|
82
|
+
* no-polling design alive.
|
|
83
|
+
*/
|
|
84
|
+
const HEARTBEAT_MS = 15_000;
|
|
85
|
+
export function createSyncSignalHub() {
|
|
86
|
+
const listeners = new Set();
|
|
87
|
+
let closed = false;
|
|
88
|
+
return {
|
|
89
|
+
announce(news) {
|
|
90
|
+
if (closed)
|
|
91
|
+
return;
|
|
92
|
+
// Snapshot: a listener that unsubscribes itself mid-fanout (the normal
|
|
93
|
+
// teardown path) must not mutate the set being iterated.
|
|
94
|
+
for (const listener of [...listeners]) {
|
|
95
|
+
try {
|
|
96
|
+
listener(news);
|
|
97
|
+
}
|
|
98
|
+
catch {
|
|
99
|
+
// A broken transport is that subscriber's problem, never the
|
|
100
|
+
// producer's: drop it and keep fanning out. An `announce` may be
|
|
101
|
+
// called from inside an op-log write, so it can never throw upward.
|
|
102
|
+
listeners.delete(listener);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
},
|
|
106
|
+
subscribe(listener) {
|
|
107
|
+
if (closed)
|
|
108
|
+
return () => undefined;
|
|
109
|
+
listeners.add(listener);
|
|
110
|
+
return () => {
|
|
111
|
+
listeners.delete(listener);
|
|
112
|
+
};
|
|
113
|
+
},
|
|
114
|
+
get subscriberCount() {
|
|
115
|
+
return listeners.size;
|
|
116
|
+
},
|
|
117
|
+
close() {
|
|
118
|
+
if (closed)
|
|
119
|
+
return;
|
|
120
|
+
closed = true;
|
|
121
|
+
listeners.clear();
|
|
122
|
+
},
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
/**
|
|
126
|
+
* Bridge a hub subscription onto a `ReadableStream` of SSE bytes.
|
|
127
|
+
*
|
|
128
|
+
* The stream ends when the client disconnects (`signal`) or the hub closes; both
|
|
129
|
+
* paths run the same teardown, so a dropped connection cannot leave a subscription
|
|
130
|
+
* — or an interval — behind.
|
|
131
|
+
*/
|
|
132
|
+
export function signalStream(hub, options = {}) {
|
|
133
|
+
const heartbeatMs = options.heartbeatMs ?? HEARTBEAT_MS;
|
|
134
|
+
const setIntervalFn = options.setIntervalFn ?? ((fn, ms) => setInterval(fn, ms));
|
|
135
|
+
const clearIntervalFn = options.clearIntervalFn ??
|
|
136
|
+
((handle) => clearInterval(handle));
|
|
137
|
+
const encoder = new TextEncoder();
|
|
138
|
+
let unsubscribe = () => undefined;
|
|
139
|
+
let heartbeat = null;
|
|
140
|
+
let done = false;
|
|
141
|
+
return new ReadableStream({
|
|
142
|
+
start(controller) {
|
|
143
|
+
const teardown = () => {
|
|
144
|
+
if (done)
|
|
145
|
+
return;
|
|
146
|
+
done = true;
|
|
147
|
+
unsubscribe();
|
|
148
|
+
if (heartbeat !== null)
|
|
149
|
+
clearIntervalFn(heartbeat);
|
|
150
|
+
heartbeat = null;
|
|
151
|
+
try {
|
|
152
|
+
controller.close();
|
|
153
|
+
}
|
|
154
|
+
catch {
|
|
155
|
+
// Already closed by the runtime when the socket died — expected.
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
const write = (chunk) => {
|
|
159
|
+
if (done)
|
|
160
|
+
return;
|
|
161
|
+
try {
|
|
162
|
+
controller.enqueue(encoder.encode(chunk));
|
|
163
|
+
}
|
|
164
|
+
catch {
|
|
165
|
+
teardown();
|
|
166
|
+
}
|
|
167
|
+
};
|
|
168
|
+
if (options.initial)
|
|
169
|
+
write(encodeNewsFrame(options.initial));
|
|
170
|
+
unsubscribe = hub.subscribe((news) => {
|
|
171
|
+
write(encodeNewsFrame(news));
|
|
172
|
+
});
|
|
173
|
+
heartbeat = setIntervalFn(() => {
|
|
174
|
+
// A comment frame. It wakes no client-side handler and issues no
|
|
175
|
+
// request — it keeps proxies from calling the connection idle.
|
|
176
|
+
write(": keep-alive\n\n");
|
|
177
|
+
}, heartbeatMs);
|
|
178
|
+
if (options.signal) {
|
|
179
|
+
if (options.signal.aborted)
|
|
180
|
+
teardown();
|
|
181
|
+
else
|
|
182
|
+
options.signal.addEventListener("abort", teardown, { once: true });
|
|
183
|
+
}
|
|
184
|
+
},
|
|
185
|
+
cancel() {
|
|
186
|
+
done = true;
|
|
187
|
+
unsubscribe();
|
|
188
|
+
if (heartbeat !== null)
|
|
189
|
+
clearIntervalFn(heartbeat);
|
|
190
|
+
heartbeat = null;
|
|
191
|
+
},
|
|
192
|
+
});
|
|
193
|
+
}
|
|
194
|
+
/** Serialize one announcement as an SSE frame. */
|
|
195
|
+
export const encodeNewsFrame = (news) => `event: news\ndata: ${JSON.stringify(news)}\n\n`;
|
|
196
|
+
/** A ready-to-return SSE `Response` — what the receiver's `GET /signal` answers. */
|
|
197
|
+
export const signalResponse = (hub, options = {}) => new Response(signalStream(hub, options), { headers: { ...SIGNAL_SSE_HEADERS } });
|
|
198
|
+
/**
|
|
199
|
+
* Hold one stream open against the peer, calling {@link SignalClientOptions.onNews}
|
|
200
|
+
* on every `news` frame, and reconnect with jittered backoff **while disconnected**.
|
|
201
|
+
*
|
|
202
|
+
* 🔴 Read the loop shape as load-bearing. The reconnect is `await sleep(ms)` inside
|
|
203
|
+
* an async loop, NOT a `setTimeout` whose callback calls `fetch`. The second shape
|
|
204
|
+
* is byte-for-byte what a poll looks like, and a reviewer — or a checker — cannot
|
|
205
|
+
* tell the two apart once it is written that way. Here, a connected client is
|
|
206
|
+
* parked on `reader.read()` and issues nothing; `requestCount` stops moving the
|
|
207
|
+
* moment a connection succeeds, and that is what `signal.spec.ts` asserts over 24
|
|
208
|
+
* simulated hours.
|
|
209
|
+
*
|
|
210
|
+
* The jitter is not decoration either: a fixed backoff across N clients recovering
|
|
211
|
+
* from the same outage is a synchronized stampede at the instant the peer comes
|
|
212
|
+
* back up — the worst possible moment to send it a thundering herd.
|
|
213
|
+
*/
|
|
214
|
+
export function connectSyncSignal(options) {
|
|
215
|
+
const doFetch = options.fetchImpl ?? fetch;
|
|
216
|
+
const base = options.basePath.replace(/\/$/, "");
|
|
217
|
+
const backoffBase = options.backoffBaseMs ?? 1_000;
|
|
218
|
+
const backoffMax = options.backoffMaxMs ?? 5 * 60_000;
|
|
219
|
+
const sleep = options.sleepFn ?? ((ms) => new Promise((r) => setTimeout(r, ms)));
|
|
220
|
+
const jitter = options.jitterFn ?? Math.random;
|
|
221
|
+
const log = options.log ?? (() => undefined);
|
|
222
|
+
let stopped = false;
|
|
223
|
+
let connected = false;
|
|
224
|
+
let requests = 0;
|
|
225
|
+
let failures = 0;
|
|
226
|
+
let abort = null;
|
|
227
|
+
/** Open one stream and stay on it until it ends. Throws on any failure. */
|
|
228
|
+
const runOnce = async () => {
|
|
229
|
+
const controller = new AbortController();
|
|
230
|
+
abort = controller;
|
|
231
|
+
requests += 1;
|
|
232
|
+
const response = await doFetch(`${base}/signal?peer=${encodeURIComponent(options.selfId)}`, {
|
|
233
|
+
// 🔴 A followed redirect is how a moved hostname became eight days of
|
|
234
|
+
// silence on the `/pull` side — see `./http.ts`'s header. Same rule here.
|
|
235
|
+
redirect: "manual",
|
|
236
|
+
headers: {
|
|
237
|
+
authorization: `Bearer ${options.token}`,
|
|
238
|
+
"x-sync-peer": options.selfId,
|
|
239
|
+
accept: "text/event-stream",
|
|
240
|
+
},
|
|
241
|
+
signal: controller.signal,
|
|
242
|
+
});
|
|
243
|
+
if (response.status >= 300 && response.status < 400) {
|
|
244
|
+
throw new Error(`sync/signal: ${response.status} — the peer address redirects to ` +
|
|
245
|
+
`${response.headers.get("location") ?? "an unnamed destination"}.`);
|
|
246
|
+
}
|
|
247
|
+
if (!response.ok)
|
|
248
|
+
throw new Error(`sync/signal: ${response.status}`);
|
|
249
|
+
if (!response.body)
|
|
250
|
+
throw new Error("sync/signal: the peer sent no stream body");
|
|
251
|
+
connected = true;
|
|
252
|
+
failures = 0;
|
|
253
|
+
const reader = response.body.getReader();
|
|
254
|
+
const decoder = new TextDecoder();
|
|
255
|
+
let buffer = "";
|
|
256
|
+
try {
|
|
257
|
+
while (!stopped) {
|
|
258
|
+
const { done, value } = await reader.read();
|
|
259
|
+
if (done)
|
|
260
|
+
break;
|
|
261
|
+
buffer += decoder.decode(value, { stream: true });
|
|
262
|
+
// SSE frames are separated by a blank line. A partial tail stays in
|
|
263
|
+
// the buffer until the rest of it arrives.
|
|
264
|
+
let split = buffer.indexOf("\n\n");
|
|
265
|
+
while (split !== -1) {
|
|
266
|
+
const frame = buffer.slice(0, split);
|
|
267
|
+
buffer = buffer.slice(split + 2);
|
|
268
|
+
const news = parseNewsFrame(frame);
|
|
269
|
+
if (news)
|
|
270
|
+
options.onNews(news);
|
|
271
|
+
split = buffer.indexOf("\n\n");
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
finally {
|
|
276
|
+
connected = false;
|
|
277
|
+
try {
|
|
278
|
+
await reader.cancel();
|
|
279
|
+
}
|
|
280
|
+
catch {
|
|
281
|
+
// The socket is already gone — that is the normal end of a stream.
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
};
|
|
285
|
+
/**
|
|
286
|
+
* The supervisor. Every `await` here is either the open stream itself or a sleep
|
|
287
|
+
* between failures; there is no path that waits on a healthy peer in order to
|
|
288
|
+
* ask it something.
|
|
289
|
+
*/
|
|
290
|
+
const supervise = async () => {
|
|
291
|
+
while (!stopped) {
|
|
292
|
+
try {
|
|
293
|
+
await runOnce();
|
|
294
|
+
}
|
|
295
|
+
catch (error) {
|
|
296
|
+
if (stopped)
|
|
297
|
+
return;
|
|
298
|
+
failures += 1;
|
|
299
|
+
if (failures === 1) {
|
|
300
|
+
log(`[sync] signal stream lost — reconnecting. This is normal (peer asleep / offline): ` +
|
|
301
|
+
`${error instanceof Error ? error.message : String(error)}`);
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
if (stopped)
|
|
305
|
+
return;
|
|
306
|
+
const step = Math.min(backoffMax, backoffBase * 2 ** Math.max(0, failures - 1));
|
|
307
|
+
await sleep(Math.round(step * (0.5 + jitter() * 0.5)));
|
|
308
|
+
}
|
|
309
|
+
};
|
|
310
|
+
void supervise();
|
|
311
|
+
return {
|
|
312
|
+
stop() {
|
|
313
|
+
stopped = true;
|
|
314
|
+
abort?.abort();
|
|
315
|
+
abort = null;
|
|
316
|
+
},
|
|
317
|
+
get connected() {
|
|
318
|
+
return connected;
|
|
319
|
+
},
|
|
320
|
+
get requestCount() {
|
|
321
|
+
return requests;
|
|
322
|
+
},
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
/** `event: news\ndata: {...}` → {@link SyncNews}, or null for a heartbeat/unknown. */
|
|
326
|
+
export function parseNewsFrame(frame) {
|
|
327
|
+
let isNews = false;
|
|
328
|
+
let data = null;
|
|
329
|
+
for (const line of frame.split("\n")) {
|
|
330
|
+
if (line.startsWith(":"))
|
|
331
|
+
continue; // comment frame — the heartbeat
|
|
332
|
+
if (line.startsWith("event:"))
|
|
333
|
+
isNews = line.slice(6).trim() === "news";
|
|
334
|
+
else if (line.startsWith("data:"))
|
|
335
|
+
data = line.slice(5).trim();
|
|
336
|
+
}
|
|
337
|
+
if (!isNews || data === null)
|
|
338
|
+
return null;
|
|
339
|
+
try {
|
|
340
|
+
const parsed = JSON.parse(data);
|
|
341
|
+
if (typeof parsed.head !== "number" || typeof parsed.instanceId !== "string")
|
|
342
|
+
return null;
|
|
343
|
+
return { head: parsed.head, instanceId: parsed.instanceId };
|
|
344
|
+
}
|
|
345
|
+
catch {
|
|
346
|
+
return null;
|
|
347
|
+
}
|
|
348
|
+
}
|
|
@@ -1,20 +1,63 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The in-process
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
2
|
+
* The in-process sync loop. Runs on the dialing half only; a failure NEVER takes
|
|
3
|
+
* the host daemon down (an unsynced instance still serves and still accepts writes
|
|
4
|
+
* — that is the point of two independently-writable instances). The timer is
|
|
5
|
+
* `unref`'d so it never holds the process open.
|
|
6
|
+
*
|
|
7
|
+
* ── 🔴 This loop does not poll, and it may never be made to again ────────────────
|
|
8
|
+
*
|
|
9
|
+
* Until 2026-09-15 this file was a dialer twice over: a `GET /requested` probe at a
|
|
10
|
+
* peer every 20 s while healthy (`REQUEST_POLL_MS`), and a full reconciliation every
|
|
11
|
+
* 5 minutes (`DEFAULT_LOOP.intervalMs`) whether or not either side had news. The
|
|
12
|
+
* owner ruled both out — *"There should be no polling in cb unless you can make some
|
|
13
|
+
* good case for it that beats the api option"* — and the case against was asked for
|
|
14
|
+
* and not found. Measured in `vault`: the 20 s probe alone was **68 % of that app's
|
|
15
|
+
* entire traffic, one request every 22 seconds, for an app with one user.**
|
|
16
|
+
*
|
|
17
|
+
* Both are gone. Every wake-up this loop books now carries a {@link WakeReason}, and
|
|
18
|
+
* that union is deliberately closed with no periodic member — there is no value you
|
|
19
|
+
* can pass to {@link schedule} that means "again in a while". That makes the poll
|
|
20
|
+
* unrepresentable rather than merely discouraged, which matters because
|
|
21
|
+
* `REQUEST_POLL_MS` did not survive as a habit; it survived as **exported public
|
|
22
|
+
* API**, and the next consumer to import it would have made its removal a breaking
|
|
23
|
+
* change instead of an edit.
|
|
24
|
+
*
|
|
25
|
+
* What replaced each half:
|
|
26
|
+
*
|
|
27
|
+
* · **Mac → peer** is a PUSH. `markDirty()` after a local write books one
|
|
28
|
+
* debounced run: the side that has news says so, which is what "sync now"
|
|
29
|
+
* should always have been.
|
|
30
|
+
* · **peer → Mac** is `./signal` — the dialing half holds one long-lived stream
|
|
31
|
+
* open and the receiver writes a frame when its log grows. The Mac is still
|
|
32
|
+
* purely a client: it dials OUT, listens on nothing, and no timer fires.
|
|
33
|
+
*
|
|
34
|
+
* 🔴 **`markDirty()` is now REQUIRED wiring, not an optimization.** With the steady
|
|
35
|
+
* interval gone it is the only thing that pushes a local write. A consumer that
|
|
36
|
+
* takes the handle and drops `markDirty` (which `apps/vault` did while the interval
|
|
37
|
+
* still covered for it) will sit clean and silent forever.
|
|
38
|
+
*
|
|
39
|
+
* The peer is re-resolved on every run, so revoking a link takes effect without a
|
|
40
|
+
* restart. Linking a NEW peer is a user action and is noticed at the next
|
|
41
|
+
* `syncNow()` / `markDirty()` rather than by a timer that was watching for it —
|
|
42
|
+
* call `syncNow()` after a link and it is immediate.
|
|
11
43
|
*/
|
|
12
44
|
import { type SyncLoopConfig } from "./planner";
|
|
13
45
|
import type { SyncStatusReporter } from "./status";
|
|
14
46
|
import type { SyncSummary } from "./types";
|
|
15
|
-
|
|
47
|
+
/**
|
|
48
|
+
* 🔴 Every reason this loop is allowed to wake up. There is deliberately no
|
|
49
|
+
* `"interval"` / `"poll"` member, and adding one is the change this whole file
|
|
50
|
+
* exists to refuse.
|
|
51
|
+
*
|
|
52
|
+
* · `boot` — one catch-up run at start.
|
|
53
|
+
* · `dirty` — local ops are waiting to be pushed (debounced).
|
|
54
|
+
* · `forced` — `syncNow()`, or a frame off `./signal` saying the peer has news.
|
|
55
|
+
* · `retry` — the last attempt failed. Runs only while disconnected, never
|
|
56
|
+
* against a healthy peer; that is a reconnect backoff, not a poll.
|
|
57
|
+
*/
|
|
58
|
+
export type WakeReason = "boot" | "dirty" | "forced" | "retry";
|
|
16
59
|
export interface SyncTimerDeps {
|
|
17
|
-
/** Re-read every
|
|
60
|
+
/** Re-read on every run: null = not linked (quiet no-op). */
|
|
18
61
|
resolvePeer: () => {
|
|
19
62
|
basePath: string;
|
|
20
63
|
token: string;
|
|
@@ -26,14 +69,8 @@ export interface SyncTimerDeps {
|
|
|
26
69
|
}) => Promise<SyncSummary>;
|
|
27
70
|
/** The local op-log head — a head beyond the last synced one marks dirty. */
|
|
28
71
|
head: () => number;
|
|
29
|
-
/** Ask the peer whether its "Sync now" was pressed; null on any failure. */
|
|
30
|
-
checkRequest?: (peer: {
|
|
31
|
-
basePath: string;
|
|
32
|
-
token: string;
|
|
33
|
-
}) => Promise<number | null>;
|
|
34
72
|
status?: SyncStatusReporter;
|
|
35
73
|
loop?: SyncLoopConfig;
|
|
36
|
-
requestPollMs?: number;
|
|
37
74
|
now?: () => number;
|
|
38
75
|
setTimer?: (fn: () => void, ms: number) => unknown;
|
|
39
76
|
clearTimer?: (handle: unknown) => void;
|
|
@@ -44,9 +81,16 @@ export interface SyncTimerDeps {
|
|
|
44
81
|
}
|
|
45
82
|
export interface SyncTimerHandle {
|
|
46
83
|
stop(): void;
|
|
47
|
-
/**
|
|
84
|
+
/**
|
|
85
|
+
* Run now. The "Sync now" button on the dialing half, and the handler for a
|
|
86
|
+
* `./signal` frame — both are "something has news", which is a push.
|
|
87
|
+
*/
|
|
48
88
|
syncNow(): void;
|
|
49
|
-
/**
|
|
89
|
+
/**
|
|
90
|
+
* Mark local writes pending — call after any local op. 🔴 Required wiring: with
|
|
91
|
+
* no steady interval this is what pushes a local write. A burst coalesces into
|
|
92
|
+
* one run via the planner's debounce.
|
|
93
|
+
*/
|
|
50
94
|
markDirty(): void;
|
|
51
95
|
}
|
|
52
96
|
export declare function startSyncTimer(deps: SyncTimerDeps): SyncTimerHandle;
|
|
@@ -1,16 +1,52 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The in-process
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
2
|
+
* The in-process sync loop. Runs on the dialing half only; a failure NEVER takes
|
|
3
|
+
* the host daemon down (an unsynced instance still serves and still accepts writes
|
|
4
|
+
* — that is the point of two independently-writable instances). The timer is
|
|
5
|
+
* `unref`'d so it never holds the process open.
|
|
6
6
|
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
7
|
+
* ── 🔴 This loop does not poll, and it may never be made to again ────────────────
|
|
8
|
+
*
|
|
9
|
+
* Until 2026-09-15 this file was a dialer twice over: a `GET /requested` probe at a
|
|
10
|
+
* peer every 20 s while healthy (`REQUEST_POLL_MS`), and a full reconciliation every
|
|
11
|
+
* 5 minutes (`DEFAULT_LOOP.intervalMs`) whether or not either side had news. The
|
|
12
|
+
* owner ruled both out — *"There should be no polling in cb unless you can make some
|
|
13
|
+
* good case for it that beats the api option"* — and the case against was asked for
|
|
14
|
+
* and not found. Measured in `vault`: the 20 s probe alone was **68 % of that app's
|
|
15
|
+
* entire traffic, one request every 22 seconds, for an app with one user.**
|
|
16
|
+
*
|
|
17
|
+
* Both are gone. Every wake-up this loop books now carries a {@link WakeReason}, and
|
|
18
|
+
* that union is deliberately closed with no periodic member — there is no value you
|
|
19
|
+
* can pass to {@link schedule} that means "again in a while". That makes the poll
|
|
20
|
+
* unrepresentable rather than merely discouraged, which matters because
|
|
21
|
+
* `REQUEST_POLL_MS` did not survive as a habit; it survived as **exported public
|
|
22
|
+
* API**, and the next consumer to import it would have made its removal a breaking
|
|
23
|
+
* change instead of an edit.
|
|
24
|
+
*
|
|
25
|
+
* What replaced each half:
|
|
26
|
+
*
|
|
27
|
+
* · **Mac → peer** is a PUSH. `markDirty()` after a local write books one
|
|
28
|
+
* debounced run: the side that has news says so, which is what "sync now"
|
|
29
|
+
* should always have been.
|
|
30
|
+
* · **peer → Mac** is `./signal` — the dialing half holds one long-lived stream
|
|
31
|
+
* open and the receiver writes a frame when its log grows. The Mac is still
|
|
32
|
+
* purely a client: it dials OUT, listens on nothing, and no timer fires.
|
|
33
|
+
*
|
|
34
|
+
* 🔴 **`markDirty()` is now REQUIRED wiring, not an optimization.** With the steady
|
|
35
|
+
* interval gone it is the only thing that pushes a local write. A consumer that
|
|
36
|
+
* takes the handle and drops `markDirty` (which `apps/vault` did while the interval
|
|
37
|
+
* still covered for it) will sit clean and silent forever.
|
|
38
|
+
*
|
|
39
|
+
* The peer is re-resolved on every run, so revoking a link takes effect without a
|
|
40
|
+
* restart. Linking a NEW peer is a user action and is noticed at the next
|
|
41
|
+
* `syncNow()` / `markDirty()` rather than by a timer that was watching for it —
|
|
42
|
+
* call `syncNow()` after a link and it is immediate.
|
|
11
43
|
*/
|
|
12
44
|
import { isUnreachable, DEFAULT_LOOP, planNextSync, } from "./planner";
|
|
13
|
-
|
|
45
|
+
/**
|
|
46
|
+
* One reconciliation shortly after start: catch up on whatever happened while this
|
|
47
|
+
* process was down. A single shot at boot, not a cadence — the same catch-up a push
|
|
48
|
+
* model needs anyway.
|
|
49
|
+
*/
|
|
14
50
|
const KICKOFF_MS = 5_000;
|
|
15
51
|
function defaultPeerLabel(basePath) {
|
|
16
52
|
try {
|
|
@@ -29,46 +65,63 @@ export function startSyncTimer(deps) {
|
|
|
29
65
|
const warn = deps.warn ?? ((m) => console.error(m));
|
|
30
66
|
const status = deps.status;
|
|
31
67
|
const peerLabel = deps.peerLabel ?? defaultPeerLabel;
|
|
32
|
-
const requestPollMs = deps.requestPollMs ?? REQUEST_POLL_MS;
|
|
33
68
|
status?.update({ enabled: true, role: "dialer" });
|
|
34
69
|
const state = { lastAttempt: 0, consecutiveFailures: 0, dirtySince: null };
|
|
35
70
|
let syncedHead = deps.head();
|
|
36
71
|
let stopped = false;
|
|
37
72
|
let handle = null;
|
|
38
73
|
let reportedUnreachable = false;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
74
|
+
/**
|
|
75
|
+
* 🔴 Starts TRUE: the boot catch-up is a forced run, not a scheduled one.
|
|
76
|
+
*
|
|
77
|
+
* It used to fall out of the planner for free — `lastAttempt: 0` plus a steady
|
|
78
|
+
* `intervalMs` made the first tick always due. With no interval left, a clean
|
|
79
|
+
* healthy state plans nothing, so the one run this loop genuinely owes at
|
|
80
|
+
* startup has to be asked for explicitly. It has a reason: this process may have
|
|
81
|
+
* missed ops while it was down, and catching up is exactly what a push model
|
|
82
|
+
* does on reconnect.
|
|
83
|
+
*/
|
|
84
|
+
let forceRun = true;
|
|
42
85
|
let inFlight = false;
|
|
43
|
-
|
|
86
|
+
/** Why the currently-booked wake exists, or null when the loop is quiet. */
|
|
87
|
+
let bookedFor = null;
|
|
88
|
+
/**
|
|
89
|
+
* Book the next wake-up. 🔴 The `reason` is not decoration — it is the type-level
|
|
90
|
+
* guard: {@link WakeReason} has no periodic member, so there is no way to spell
|
|
91
|
+
* "wake again in `intervalMs`" through this function. It is also read back by
|
|
92
|
+
* {@link SyncTimerHandle.markDirty}, which must not push an imminent forced run
|
|
93
|
+
* out to a debounce.
|
|
94
|
+
*/
|
|
95
|
+
const schedule = (reason, ms) => {
|
|
44
96
|
if (stopped)
|
|
45
97
|
return;
|
|
46
98
|
if (handle !== null)
|
|
47
99
|
clearTimer(handle);
|
|
48
100
|
handle = setTimer(tick, ms);
|
|
49
101
|
handle?.unref?.();
|
|
102
|
+
bookedFor = reason;
|
|
50
103
|
};
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
104
|
+
/** Nothing to push, nothing to retry: cancel any pending wake and go quiet. */
|
|
105
|
+
const idle = () => {
|
|
106
|
+
if (handle !== null)
|
|
107
|
+
clearTimer(handle);
|
|
108
|
+
handle = null;
|
|
109
|
+
bookedFor = null;
|
|
54
110
|
};
|
|
55
|
-
|
|
56
|
-
|
|
111
|
+
/** Apply the planner's verdict — the ONE place a wake-up is booked or declined. */
|
|
112
|
+
const bookNext = () => {
|
|
113
|
+
if (stopped)
|
|
57
114
|
return;
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
})
|
|
69
|
-
.finally(() => {
|
|
70
|
-
probing = false;
|
|
71
|
-
});
|
|
115
|
+
if (forceRun) {
|
|
116
|
+
schedule("forced", 1);
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
const plan = planNextSync(state, now(), loop);
|
|
120
|
+
if (plan.waitMs === null || plan.reason === null) {
|
|
121
|
+
idle();
|
|
122
|
+
return;
|
|
123
|
+
}
|
|
124
|
+
schedule(plan.reason, Math.max(1, plan.waitMs));
|
|
72
125
|
};
|
|
73
126
|
const tick = () => {
|
|
74
127
|
if (stopped)
|
|
@@ -84,15 +137,16 @@ export function startSyncTimer(deps) {
|
|
|
84
137
|
pending: Math.max(0, head - syncedHead),
|
|
85
138
|
});
|
|
86
139
|
if (peer === null) {
|
|
87
|
-
|
|
140
|
+
// Not linked. Nothing to dial and nothing to wait for — `syncNow()` or the
|
|
141
|
+
// next local write re-checks. A re-check timer here would be a poll at a
|
|
142
|
+
// peer that does not exist yet.
|
|
143
|
+
forceRun = false;
|
|
144
|
+
idle();
|
|
88
145
|
return;
|
|
89
146
|
}
|
|
90
147
|
const plan = planNextSync(state, now(), loop);
|
|
91
148
|
if (!plan.runNow && !forceRun) {
|
|
92
|
-
|
|
93
|
-
schedule(sleepFor(plan.waitMs));
|
|
94
|
-
if (linkedAndHealthy)
|
|
95
|
-
probeRequest(peer);
|
|
149
|
+
bookNext();
|
|
96
150
|
return;
|
|
97
151
|
}
|
|
98
152
|
forceRun = false;
|
|
@@ -140,31 +194,36 @@ export function startSyncTimer(deps) {
|
|
|
140
194
|
})
|
|
141
195
|
.finally(() => {
|
|
142
196
|
inFlight = false;
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
197
|
+
// A local write that landed mid-run is news the planner must see.
|
|
198
|
+
if (deps.head() > syncedHead && state.dirtySince === null)
|
|
199
|
+
state.dirtySince = now();
|
|
200
|
+
bookNext();
|
|
147
201
|
});
|
|
148
202
|
};
|
|
149
|
-
schedule(KICKOFF_MS);
|
|
203
|
+
schedule("boot", KICKOFF_MS);
|
|
150
204
|
return {
|
|
151
205
|
stop: () => {
|
|
152
206
|
stopped = true;
|
|
153
207
|
if (handle !== null)
|
|
154
208
|
clearTimer(handle);
|
|
209
|
+
handle = null;
|
|
155
210
|
},
|
|
156
211
|
syncNow: () => {
|
|
157
212
|
if (stopped)
|
|
158
213
|
return;
|
|
159
214
|
forceRun = true;
|
|
160
|
-
schedule(1);
|
|
215
|
+
schedule("forced", 1);
|
|
161
216
|
},
|
|
162
217
|
markDirty: () => {
|
|
163
218
|
if (stopped)
|
|
164
219
|
return;
|
|
165
220
|
if (state.dirtySince === null)
|
|
166
221
|
state.dirtySince = now();
|
|
167
|
-
|
|
222
|
+
// A forced run is already due in ~1ms and will push these ops; re-booking
|
|
223
|
+
// would replace it with a 3s debounce and make "Sync now" slower.
|
|
224
|
+
if (bookedFor === "forced")
|
|
225
|
+
return;
|
|
226
|
+
bookNext();
|
|
168
227
|
},
|
|
169
228
|
};
|
|
170
229
|
}
|