cursedbelt-server 2.0.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/server/bench/assert.d.ts +61 -0
  2. package/dist/server/bench/assert.js +117 -0
  3. package/dist/server/bench/budget.d.ts +130 -0
  4. package/dist/server/bench/budget.js +131 -0
  5. package/dist/server/bench/cpuBudget.d.ts +45 -0
  6. package/dist/server/bench/cpuBudget.js +34 -0
  7. package/dist/server/bench/cpuClock.d.ts +65 -0
  8. package/dist/server/bench/cpuClock.js +100 -0
  9. package/dist/server/bench/index.d.ts +40 -0
  10. package/dist/server/bench/index.js +40 -0
  11. package/dist/server/bench/recorder.d.ts +70 -0
  12. package/dist/server/bench/recorder.js +95 -0
  13. package/dist/server/bench/runBench.d.ts +61 -0
  14. package/dist/server/bench/runBench.js +61 -0
  15. package/dist/server/d1/backup.d.ts +110 -0
  16. package/dist/server/d1/backup.js +128 -0
  17. package/dist/server/d1/fakeD1.d.ts +41 -0
  18. package/dist/server/d1/fakeD1.js +185 -0
  19. package/dist/server/d1/index.d.ts +24 -0
  20. package/dist/server/d1/index.js +24 -0
  21. package/dist/server/d1/kysely.d.ts +56 -0
  22. package/dist/server/d1/kysely.js +138 -0
  23. package/dist/server/d1/limits.d.ts +56 -0
  24. package/dist/server/d1/limits.js +96 -0
  25. package/dist/server/d1/local.d.ts +31 -0
  26. package/dist/server/d1/local.js +135 -0
  27. package/dist/server/d1/remote.d.ts +59 -0
  28. package/dist/server/d1/remote.js +124 -0
  29. package/dist/server/d1/scheduling.d.ts +113 -0
  30. package/dist/server/d1/scheduling.js +164 -0
  31. package/dist/server/d1/types.d.ts +143 -0
  32. package/dist/server/d1/types.js +80 -0
  33. package/dist/server/d1/values.d.ts +50 -0
  34. package/dist/server/d1/values.js +124 -0
  35. package/dist/server/master-lock/guard.d.ts +10 -0
  36. package/dist/server/master-lock/guard.js +70 -19
  37. package/dist/server/master-lock/index.d.ts +1 -1
  38. package/dist/server/master-lock/index.js +1 -1
  39. package/dist/server/master-lock/lockPage.d.ts +1 -1
  40. package/dist/server/master-lock/lockPage.js +68 -3
  41. package/dist/server/master-lock/masterLock.d.ts +250 -76
  42. package/dist/server/master-lock/masterLock.js +426 -114
  43. package/dist/server/master-lock/principals.js +6 -1
  44. package/dist/server/master-lock/seed.d.ts +5 -1
  45. package/dist/server/master-lock/seed.js +18 -1
  46. package/package.json +21 -3
  47. package/src/leafSubpathsImportNothing.spec.ts +15 -3
  48. package/src/server/bench/assert.ts +192 -0
  49. package/src/server/bench/budget.spec.ts +126 -0
  50. package/src/server/bench/budget.ts +207 -0
  51. package/src/server/bench/cpuBudget.spec.ts +302 -0
  52. package/src/server/bench/cpuBudget.ts +81 -0
  53. package/src/server/bench/cpuClock.ts +119 -0
  54. package/src/server/bench/index.ts +81 -0
  55. package/src/server/bench/recorder.ts +163 -0
  56. package/src/server/bench/runBench.ts +110 -0
  57. package/src/server/d1/backup.spec.ts +121 -0
  58. package/src/server/d1/backup.ts +186 -0
  59. package/src/server/d1/fakeD1.ts +193 -0
  60. package/src/server/d1/index.ts +62 -0
  61. package/src/server/d1/kysely.spec.ts +145 -0
  62. package/src/server/d1/kysely.ts +169 -0
  63. package/src/server/d1/limits.spec.ts +90 -0
  64. package/src/server/d1/limits.ts +123 -0
  65. package/src/server/d1/local.ts +173 -0
  66. package/src/server/d1/remote.ts +182 -0
  67. package/src/server/d1/sameShape.spec.ts +279 -0
  68. package/src/server/d1/scheduling.spec.ts +120 -0
  69. package/src/server/d1/scheduling.ts +210 -0
  70. package/src/server/d1/types.ts +163 -0
  71. package/src/server/d1/values.ts +138 -0
  72. package/src/server/master-lock/accounts.spec.ts +308 -0
  73. package/src/server/master-lock/guard.spec.ts +69 -7
  74. package/src/server/master-lock/guard.ts +78 -20
  75. package/src/server/master-lock/index.ts +3 -0
  76. package/src/server/master-lock/lockPage.ts +70 -3
  77. package/src/server/master-lock/masterLock.spec.ts +56 -23
  78. package/src/server/master-lock/masterLock.ts +529 -151
  79. package/src/server/master-lock/principals.spec.ts +45 -15
  80. package/src/server/master-lock/principals.ts +6 -1
  81. package/src/server/master-lock/seed.spec.ts +7 -2
  82. package/src/server/master-lock/seed.ts +22 -2
@@ -0,0 +1,163 @@
1
+ /**
2
+ * Per-route CPU samples, and the percentiles read off them.
3
+ *
4
+ * 🔴 **p99, not a mean.** A mean over a route that is fast 99 times and 400 ms once
5
+ * reports ~4 ms and looks healthy. The number that decides whether a route is safe to
6
+ * make common is its tail, so the mean is reported beside p99 and never instead of it.
7
+ */
8
+
9
+ import {
10
+ type CpuBudgetConfig,
11
+ type ResolvedBudget,
12
+ resolveBudget,
13
+ validateCpuBudgetConfig,
14
+ } from './budget';
15
+ import type { CpuClock } from './cpuClock';
16
+
17
+ /**
18
+ * Nearest-rank percentile over an ASCENDING-sorted array: the smallest value at or below
19
+ * which at least `p`% of samples fall. No interpolation — an interpolated p99 reports a
20
+ * CPU cost no request ever actually paid, which is the wrong direction to be wrong in
21
+ * when the number is a ceiling.
22
+ */
23
+ export function percentile(sortedAsc: readonly number[], p: number): number {
24
+ if (sortedAsc.length === 0) return Number.NaN;
25
+ if (p <= 0) return sortedAsc[0];
26
+ if (p >= 100) return sortedAsc[sortedAsc.length - 1];
27
+ const rank = Math.ceil((p / 100) * sortedAsc.length);
28
+ return sortedAsc[Math.min(sortedAsc.length, Math.max(1, rank)) - 1];
29
+ }
30
+
31
+ export interface RouteCpuStats {
32
+ method: string;
33
+ route: string;
34
+ /** `'GET /api/notes/:id'` — how the route is named in a violation message. */
35
+ key: string;
36
+ samples: number;
37
+ /** Samples actually retained (≤ `samples` once the reservoir is full). */
38
+ retained: number;
39
+ mean: number;
40
+ p50: number;
41
+ p95: number;
42
+ p99: number;
43
+ max: number;
44
+ /** Total CPU-ms observed across every sample — what the route costs in aggregate. */
45
+ totalCpuMs: number;
46
+ /** The ceiling in CPU-ms, or `null` when exempt. */
47
+ budgetMs: number | null;
48
+ exemptReason?: string;
49
+ noticeAboveMs?: number;
50
+ /** False when no explicit entry matched and the default was applied. */
51
+ declared: boolean;
52
+ }
53
+
54
+ export interface CpuBudgetReport {
55
+ /** Provenance of every number below — e.g. `'process.cpuUsage'`. */
56
+ source: string;
57
+ /** 🔴 True ⇒ these are a PROXY for Worker CPU-ms, not a reading of them. */
58
+ proxy: boolean;
59
+ /** False ⇒ nothing could be measured; every stat is NaN and nothing may be concluded. */
60
+ available: boolean;
61
+ routes: RouteCpuStats[];
62
+ }
63
+
64
+ export interface CpuRecorderOpts {
65
+ clock: CpuClock;
66
+ config?: CpuBudgetConfig;
67
+ /**
68
+ * Samples retained per route. Beyond this, reservoir sampling keeps the retained set
69
+ * representative of the WHOLE run rather than of its first N requests — truncating
70
+ * would quietly turn a long-lived mount into a measurement of its own warm-up.
71
+ */
72
+ maxSamplesPerRoute?: number;
73
+ /**
74
+ * Deterministic source of randomness for the reservoir, for tests. Defaults to
75
+ * `Math.random`.
76
+ */
77
+ random?: () => number;
78
+ }
79
+
80
+ export interface CpuRecorder {
81
+ readonly clock: CpuClock;
82
+ readonly config: CpuBudgetConfig;
83
+ record(method: string, route: string, cpuMs: number): void;
84
+ report(): CpuBudgetReport;
85
+ reset(): void;
86
+ }
87
+
88
+ interface Bucket {
89
+ method: string;
90
+ route: string;
91
+ seen: number;
92
+ samples: number[];
93
+ total: number;
94
+ }
95
+
96
+ export function createCpuRecorder(opts: CpuRecorderOpts): CpuRecorder {
97
+ const config = opts.config ?? {};
98
+ validateCpuBudgetConfig(config);
99
+ const cap = opts.maxSamplesPerRoute ?? 10_000;
100
+ if (!(cap > 0)) throw new Error('createCpuRecorder: maxSamplesPerRoute must be > 0');
101
+ const random = opts.random ?? Math.random;
102
+ let buckets = new Map<string, Bucket>();
103
+
104
+ return {
105
+ clock: opts.clock,
106
+ config,
107
+ record(method, route, cpuMs) {
108
+ // A NaN reading means the clock could not measure. Recording it would poison every
109
+ // percentile on the route, so an unmeasurable request is not a sample.
110
+ if (!Number.isFinite(cpuMs)) return;
111
+ const key = `${method} ${route}`;
112
+ let b = buckets.get(key);
113
+ if (!b) {
114
+ b = { method, route, seen: 0, samples: [], total: 0 };
115
+ buckets.set(key, b);
116
+ }
117
+ b.seen += 1;
118
+ b.total += cpuMs;
119
+ if (b.samples.length < cap) {
120
+ b.samples.push(cpuMs);
121
+ } else {
122
+ // Algorithm R: keep each observed sample with equal probability.
123
+ const j = Math.floor(random() * b.seen);
124
+ if (j < cap) b.samples[j] = cpuMs;
125
+ }
126
+ },
127
+ report(): CpuBudgetReport {
128
+ const routes: RouteCpuStats[] = [];
129
+ for (const [key, b] of buckets) {
130
+ const sorted = [...b.samples].sort((x, y) => x - y);
131
+ const resolved: ResolvedBudget = resolveBudget(config, b.method, b.route);
132
+ routes.push({
133
+ method: b.method,
134
+ route: b.route,
135
+ key,
136
+ samples: b.seen,
137
+ retained: sorted.length,
138
+ mean: b.seen > 0 ? b.total / b.seen : Number.NaN,
139
+ p50: percentile(sorted, 50),
140
+ p95: percentile(sorted, 95),
141
+ p99: percentile(sorted, 99),
142
+ max: percentile(sorted, 100),
143
+ totalCpuMs: b.total,
144
+ budgetMs: resolved.cpuMs,
145
+ exemptReason: resolved.exemptReason,
146
+ noticeAboveMs: resolved.noticeAboveMs,
147
+ declared: resolved.declared,
148
+ });
149
+ }
150
+ // Worst offender first — a report is read from the top.
151
+ routes.sort((a, z) => (z.p99 || 0) - (a.p99 || 0));
152
+ return {
153
+ source: opts.clock.source,
154
+ proxy: opts.clock.proxy,
155
+ available: opts.clock.available,
156
+ routes,
157
+ };
158
+ },
159
+ reset() {
160
+ buckets = new Map();
161
+ },
162
+ };
163
+ }
@@ -0,0 +1,110 @@
1
+ import type { CpuRecorder } from './recorder';
2
+ import type { CpuBudgetReport } from './recorder';
3
+
4
+ /**
5
+ * Drive a Hono app's routes and measure what each costs in CPU.
6
+ *
7
+ * 🔴 **Serial on purpose, and this is the load-bearing decision.** `process.cpuUsage()`
8
+ * deltas are PROCESS-wide (see `cpuClock.ts`), so two requests in flight each absorb the
9
+ * other's CPU and every number is inflated by an amount nobody can subtract afterwards.
10
+ * Running one request at a time is what makes the local proxy sound enough to gate on.
11
+ * There is deliberately no `concurrency` option — it would produce numbers that look
12
+ * finer-grained and are strictly less true.
13
+ *
14
+ * Wall-clock cost of the bench itself is irrelevant; it is a gate, not a load test.
15
+ */
16
+
17
+ /** RFC 2606 reserved TLD — a bench request must never be able to leave the process. */
18
+ export const BENCH_ORIGIN = 'http://cpu-bench.invalid';
19
+
20
+ export interface BenchCase {
21
+ /** Default `'GET'`. */
22
+ method?: string;
23
+ /** Path with a leading slash, e.g. `'/api/notes/abc'`. */
24
+ path: string;
25
+ /** Extra request init — headers, body. `method` above wins over `init.method`. */
26
+ init?: RequestInit;
27
+ /** Overrides the run-wide `iterations` for this case. */
28
+ iterations?: number;
29
+ /**
30
+ * Statuses this case is expected to return. Default: anything `< 400`.
31
+ *
32
+ * 🔴 This guard is the point. A route that 404s or 500s costs almost no CPU, so a
33
+ * bench that does not check the response reports a beautiful number for a handler
34
+ * that never ran — a green budget over a broken route.
35
+ */
36
+ expectStatus?: number | number[] | ((status: number) => boolean);
37
+ }
38
+
39
+ export interface RunCpuBenchOpts {
40
+ /** Anything with Hono's `fetch` shape. */
41
+ app: { fetch: (req: Request, env?: unknown, ctx?: unknown) => Response | Promise<Response> };
42
+ /** The recorder the app's `cpuBudget()` middleware writes through. */
43
+ recorder: CpuRecorder;
44
+ cases: BenchCase[];
45
+ /** Measured iterations per case. Default 30 — above `assertCpuBudgets`' 20-sample floor. */
46
+ iterations?: number;
47
+ /**
48
+ * Unmeasured iterations per case, run first and then DISCARDED.
49
+ *
50
+ * 🔴 Without this the bench measures JIT warm-up. A first call through a cold handler
51
+ * can cost an order of magnitude more than its steady state, and with 30 samples one
52
+ * cold call lands squarely in the p99 — so the route that fails is whichever one the
53
+ * bench happened to touch first. Default 5.
54
+ */
55
+ warmup?: number;
56
+ /** Origin for the synthesized requests. Default {@link BENCH_ORIGIN}. */
57
+ origin?: string;
58
+ /** Passed through to `app.fetch` as the Worker `env`. */
59
+ env?: unknown;
60
+ /** Passed through to `app.fetch` as the Worker execution context. */
61
+ ctx?: unknown;
62
+ }
63
+
64
+ function statusAllowed(c: BenchCase, status: number): boolean {
65
+ const expect = c.expectStatus;
66
+ if (expect === undefined) return status < 400;
67
+ if (typeof expect === 'function') return expect(status);
68
+ if (Array.isArray(expect)) return expect.includes(status);
69
+ return expect === status;
70
+ }
71
+
72
+ function describe(c: BenchCase): string {
73
+ return `${(c.method ?? 'GET').toUpperCase()} ${c.path}`;
74
+ }
75
+
76
+ async function fire(opts: RunCpuBenchOpts, c: BenchCase): Promise<void> {
77
+ const method = (c.method ?? c.init?.method ?? 'GET').toUpperCase();
78
+ const req = new Request(`${opts.origin ?? BENCH_ORIGIN}${c.path}`, { ...c.init, method });
79
+ const res = await opts.app.fetch(req, opts.env, opts.ctx);
80
+ if (!statusAllowed(c, res.status)) {
81
+ const body = await res.text().catch(() => '');
82
+ throw new Error(
83
+ `cpu-bench: ${describe(c)} returned ${res.status}, which this case does not expect. ` +
84
+ 'A route that errors costs no CPU, so measuring it would report a budget it never ' +
85
+ `met. Fix the case or set expectStatus.${body ? ` Body: ${body.slice(0, 200)}` : ''}`,
86
+ );
87
+ }
88
+ // Drain the body so a streamed response's work is actually done before the span ends.
89
+ if (res.body && !res.bodyUsed) await res.arrayBuffer().catch(() => undefined);
90
+ }
91
+
92
+ export async function runCpuBench(opts: RunCpuBenchOpts): Promise<CpuBudgetReport> {
93
+ if (opts.cases.length === 0) throw new Error('runCpuBench: no cases given');
94
+ const iterations = opts.iterations ?? 30;
95
+ const warmup = opts.warmup ?? 5;
96
+ if (!(iterations > 0)) throw new Error('runCpuBench: iterations must be > 0');
97
+
98
+ for (const c of opts.cases) {
99
+ for (let i = 0; i < warmup; i += 1) await fire(opts, c);
100
+ }
101
+ // 🔴 Everything above was warm-up. Discard it — measuring it is the bug this guards.
102
+ opts.recorder.reset();
103
+
104
+ for (const c of opts.cases) {
105
+ const n = c.iterations ?? iterations;
106
+ for (let i = 0; i < n; i += 1) await fire(opts, c);
107
+ }
108
+
109
+ return opts.recorder.report();
110
+ }
@@ -0,0 +1,121 @@
1
+ import { Database } from 'bun:sqlite';
2
+ import { afterEach, describe, expect, test } from 'bun:test';
3
+ import { existsSync, mkdtempSync, rmSync, statSync } from 'node:fs';
4
+ import { tmpdir } from 'node:os';
5
+ import { join } from 'node:path';
6
+ import { backupFor, checkpointWal, createLocalBackup, createTimeTravelBackup } from './backup';
7
+
8
+ const dirs: string[] = [];
9
+ const scratch = (): string => {
10
+ const d = mkdtempSync(join(tmpdir(), 'd1-backup-'));
11
+ dirs.push(d);
12
+ return d;
13
+ };
14
+ afterEach(() => {
15
+ for (const d of dirs.splice(0)) rmSync(d, { recursive: true, force: true });
16
+ });
17
+
18
+ /** A WAL-mode database on disk with enough rows to produce a real WAL file. */
19
+ function seeded(dir: string): { path: string; db: Database } {
20
+ const path = join(dir, 'app.sqlite');
21
+ const db = new Database(path, { create: true });
22
+ db.exec('PRAGMA journal_mode = WAL');
23
+ db.run('CREATE TABLE t (id INTEGER PRIMARY KEY, v TEXT)');
24
+ const insert = db.prepare('INSERT INTO t (v) VALUES (?)');
25
+ for (let i = 0; i < 2_000; i++) insert.run(`row-${i}`);
26
+ return { path, db };
27
+ }
28
+
29
+ describe('the local half keeps checkpointing the WAL', () => {
30
+ test('🔴 wal_checkpoint(TRUNCATE) actually drains the WAL file', () => {
31
+ // The measurement this exists for: `family.sqlite` was once 2.2 MB of principal
32
+ // against a 2.1 MB WAL. An un-checkpointed copy is half a database.
33
+ const dir = scratch();
34
+ const { path, db } = seeded(dir);
35
+ const walPath = `${path}-wal`;
36
+
37
+ expect(existsSync(walPath)).toBe(true);
38
+ expect(statSync(walPath).size).toBeGreaterThan(0);
39
+
40
+ const result = checkpointWal(db);
41
+ expect(result.busy).toBe(false);
42
+ // 🔴 The post-condition is the FILE SIZE, not the pragma's counters. Measured
43
+ // 2026-09-16: under TRUNCATE, SQLite reports the counters from after the
44
+ // truncation, so a fully successful checkpoint returns `checkpointed: 0`. A test
45
+ // asserting `checkpointed > 0` would be asserting a structurally impossible value
46
+ // — it fails on a working checkpoint, which is how a real check gets deleted.
47
+ expect(result.checkpointed).toBe(0);
48
+ // TRUNCATE, not PASSIVE: the file is emptied, not merely merged.
49
+ expect(statSync(walPath).size).toBe(0);
50
+ db.close();
51
+ });
52
+
53
+ test('capture() checkpoints and writes a readable snapshot containing every row', async () => {
54
+ const dir = scratch();
55
+ const { path, db } = seeded(dir);
56
+ const destDir = join(dir, 'backups');
57
+
58
+ const point = await createLocalBackup({ db, sourcePath: path, destDir }).capture();
59
+
60
+ expect(point.kind).toBe('file');
61
+ expect(existsSync(point.ref)).toBe(true);
62
+ expect(point.bytes).toBeGreaterThan(0);
63
+ expect(point.exportCommand).toBeNull();
64
+
65
+ // The snapshot is a real database with the full row count — which is what would
66
+ // have failed had the WAL not been merged in.
67
+ const restored = new Database(point.ref, { readonly: true });
68
+ const count = restored.query('SELECT COUNT(*) AS n FROM t').get() as { n: number };
69
+ expect(count.n).toBe(2_000);
70
+ restored.close();
71
+ db.close();
72
+ });
73
+
74
+ test('the local half reports that it DOES need a periodic capture', () => {
75
+ const dir = scratch();
76
+ const { path, db } = seeded(dir);
77
+ expect(createLocalBackup({ db, sourcePath: path, destDir: dir }).needsPeriodicCapture).toBe(true);
78
+ db.close();
79
+ });
80
+ });
81
+
82
+ describe('the D1 half records a coordinate instead of copying bytes', () => {
83
+ const at = new Date('2026-09-16T04:30:00.000Z');
84
+
85
+ test('capture() returns a restorable timestamp and the two real commands', async () => {
86
+ const point = await createTimeTravelBackup({ databaseName: 'patterns', now: () => at }).capture();
87
+
88
+ expect(point.kind).toBe('time-travel');
89
+ expect(point.ref).toBe('2026-09-16T04:30:00.000Z');
90
+ expect(point.restoreCommand).toBe(
91
+ 'wrangler d1 time-travel restore patterns --timestamp=2026-09-16T04:30:00.000Z',
92
+ );
93
+ expect(point.exportCommand).toContain('wrangler d1 export patterns --remote');
94
+ });
95
+
96
+ test('🔴 it reports needsPeriodicCapture: false — Time Travel is automatic', () => {
97
+ // A nightly no-op job logging "backup complete" is worse than no job, because it
98
+ // reads as evidence. A scheduler must honour this flag.
99
+ expect(createTimeTravelBackup({ databaseName: 'patterns' }).needsPeriodicCapture).toBe(false);
100
+ });
101
+
102
+ test('bytes is null rather than a fabricated zero', () => {
103
+ // Cloudflare reports no size for a restore point; a 0 would read as a measurement.
104
+ return createTimeTravelBackup({ databaseName: 'x', now: () => at })
105
+ .capture()
106
+ .then((p) => expect(p.bytes).toBeNull());
107
+ });
108
+ });
109
+
110
+ describe('backupFor picks the implementation from the driver flavor', () => {
111
+ test('local flavor checkpoints; d1 flavor does not need to', async () => {
112
+ const dir = scratch();
113
+ const { path, db } = seeded(dir);
114
+ const local = () => ({ db, sourcePath: path, destDir: join(dir, 'b') });
115
+ const remote = () => ({ databaseName: 'patterns' });
116
+
117
+ expect(backupFor('local', local, remote).needsPeriodicCapture).toBe(true);
118
+ expect(backupFor('d1', local, remote).needsPeriodicCapture).toBe(false);
119
+ db.close();
120
+ });
121
+ });
@@ -0,0 +1,186 @@
1
+ /**
2
+ * The backup seam — `PRAGMA wal_checkpoint(TRUNCATE)` locally, Time Travel remotely.
3
+ *
4
+ * ## 🔴 BOTH, not one
5
+ *
6
+ * D1 has no WAL a caller can checkpoint, and 30 days of free point-in-time restore is
7
+ * strictly better than the file copy it replaces. It is tempting to read that as "the
8
+ * checkpoint goes away". It does not: `PRAGMA wal_checkpoint(TRUNCATE)` is in the backup
9
+ * path of every app in this fleet and it is load-bearing while any of them is still
10
+ * Mac-hosted. Measured previously, `family.sqlite` was **2.2 MB of principal against a
11
+ * 2.1 MB WAL** — an un-checkpointed copy is half a database, and `VACUUM INTO` merging
12
+ * the WAL is the only reason the existing `../sqlite/backup.ts` is safe.
13
+ *
14
+ * So the local implementation keeps checkpointing, the remote one records a restore
15
+ * coordinate, and both answer the same interface. An app's backup job stops caring which
16
+ * side it is on — which is the property that lets the job move before the database does.
17
+ *
18
+ * ## The remote side does not "take" a backup, and that is not a gap
19
+ *
20
+ * Time Travel is automatic and continuous: D1 retains 30 days and restores to any
21
+ * timestamp or bookmark within it. There is nothing to trigger, so
22
+ * {@link createTimeTravelBackup} records the coordinate rather than inventing an API call
23
+ * that does not exist. The coordinate IS the backup — `wrangler d1 time-travel restore`
24
+ * takes a timestamp.
25
+ *
26
+ * 🔴 **`wrangler d1 export` is the off-Cloudflare leg, and a Worker cannot run it.** A
27
+ * Worker has no shell. That leg belongs to a Cron Trigger on this Mac or in CI, which is
28
+ * why {@link BackupPoint.exportCommand} hands back the command rather than running it —
29
+ * the same split as the binary server, where the bytes and the thing that copies them are
30
+ * deliberately not the same process.
31
+ */
32
+
33
+ import type { Database } from 'bun:sqlite';
34
+ import { existsSync, mkdirSync } from 'node:fs';
35
+ import { join } from 'node:path';
36
+ import { snapshotSqlite } from '../sqlite/backup';
37
+
38
+ /** A restore coordinate — a file on this Mac, or a point in D1's retention window. */
39
+ export interface BackupPoint {
40
+ kind: 'file' | 'time-travel';
41
+ /** A filesystem path for `file`; an ISO-8601 timestamp for `time-travel`. */
42
+ ref: string;
43
+ createdAt: string;
44
+ /** Bytes on disk, or `null` where the platform does not tell us. */
45
+ bytes: number | null;
46
+ /** The exact command that restores this point. Printed in the job log, not executed. */
47
+ restoreCommand: string;
48
+ /** For `time-travel`, the command that pulls a copy OFF Cloudflare. `null` locally. */
49
+ exportCommand: string | null;
50
+ }
51
+
52
+ export interface DatabaseBackup {
53
+ /** Capture a restore point now, and describe it. */
54
+ capture(): Promise<BackupPoint>;
55
+ /** Whether this side needs a periodic capture at all. Time Travel does not. */
56
+ readonly needsPeriodicCapture: boolean;
57
+ }
58
+
59
+ export interface LocalBackupOpts {
60
+ /** The live handle — checkpointed before the snapshot is taken. */
61
+ db: Database;
62
+ /** Path of the database file being backed up. */
63
+ sourcePath: string;
64
+ /** Directory the snapshot is written into. Created if missing. */
65
+ destDir: string;
66
+ /** Override the snapshot's file name. Defaults to `<name>-<ISO>.sqlite`. */
67
+ nameFor?: (now: Date) => string;
68
+ }
69
+
70
+ /**
71
+ * Force the WAL back into the principal database and truncate the log file.
72
+ *
73
+ * `TRUNCATE` — not `PASSIVE` — because `PASSIVE` gives up silently when a reader holds the
74
+ * WAL, leaving a WAL that never drains while every log line says the backup succeeded.
75
+ *
76
+ * 🔴 **`busy` is the only field worth reading, and that is a measured correction.** Under
77
+ * `TRUNCATE` SQLite reports the counters from AFTER the truncation, so a completely
78
+ * successful checkpoint returns `(busy 0, log 0, checkpointed 0)` — measured 2026-09-16,
79
+ * draining a 70,072-byte WAL to zero reported `checkpointed: 0`. Anything asserting
80
+ * `checkpointed > 0` as proof of work is asserting a value that is structurally always
81
+ * zero here. The honest post-condition is the WAL file's SIZE, which is what
82
+ * `backup.spec.ts` checks; the honest error signal is `busy`.
83
+ */
84
+ export function checkpointWal(db: Database): { busy: boolean; logPages: number; checkpointed: number } {
85
+ // One row: (busy, log, checkpointed). See the note above on what they mean here.
86
+ const row = db.query('PRAGMA wal_checkpoint(TRUNCATE)').get() as
87
+ | { busy?: number; log?: number; checkpointed?: number }
88
+ | null;
89
+ return {
90
+ busy: (row?.busy ?? 0) === 1,
91
+ logPages: row?.log ?? 0,
92
+ checkpointed: row?.checkpointed ?? 0,
93
+ };
94
+ }
95
+
96
+ /**
97
+ * The Mac-hosted implementation: checkpoint, then `VACUUM INTO` a fresh file.
98
+ *
99
+ * `VACUUM INTO` is already transactionally consistent under WAL, so the checkpoint is not
100
+ * what makes the copy correct — it is what keeps the WAL from growing without bound on a
101
+ * database that is written far more often than it is read, which is how a 2.2 MB database
102
+ * came to carry a 2.1 MB WAL.
103
+ */
104
+ export function createLocalBackup(opts: LocalBackupOpts): DatabaseBackup {
105
+ return {
106
+ needsPeriodicCapture: true,
107
+
108
+ async capture(): Promise<BackupPoint> {
109
+ const checkpoint = checkpointWal(opts.db);
110
+ if (checkpoint.busy) {
111
+ // Not fatal — `VACUUM INTO` still produces a consistent copy — but it is the
112
+ // signal that the WAL is not draining, and silence here is how it grows.
113
+ console.warn(
114
+ `[d1/backup] wal_checkpoint(TRUNCATE) reported BUSY for ${opts.sourcePath} — ` +
115
+ `${checkpoint.logPages} page(s) still in the WAL. A reader is holding it open.`,
116
+ );
117
+ }
118
+ if (!existsSync(opts.destDir)) mkdirSync(opts.destDir, { recursive: true });
119
+
120
+ const now = new Date();
121
+ const stamp = now.toISOString().replace(/[:.]/g, '-');
122
+ const name = opts.nameFor?.(now) ?? `backup-${stamp}.sqlite`;
123
+ const dest = join(opts.destDir, name);
124
+
125
+ const bytes = snapshotSqlite(opts.sourcePath, dest);
126
+ return {
127
+ kind: 'file',
128
+ ref: dest,
129
+ createdAt: now.toISOString(),
130
+ bytes,
131
+ restoreCommand: `cp '${dest}' '${opts.sourcePath}' # with the app stopped`,
132
+ exportCommand: null,
133
+ };
134
+ },
135
+ };
136
+ }
137
+
138
+ export interface TimeTravelOpts {
139
+ /** The database name as `wrangler` knows it — what the restore command needs. */
140
+ databaseName: string;
141
+ /** Where an export should be written, for the off-Cloudflare leg. */
142
+ exportPath?: string;
143
+ /** Injected for the test; defaults to the real clock. */
144
+ now?: () => Date;
145
+ }
146
+
147
+ /**
148
+ * The D1 implementation: record the coordinate, because retention is automatic.
149
+ *
150
+ * 🔴 It reports `needsPeriodicCapture: false`, and a scheduler must honour that rather
151
+ * than calling `capture()` on a timer — a nightly no-op job that logs "backup complete"
152
+ * is worse than no job, because it reads as evidence.
153
+ */
154
+ export function createTimeTravelBackup(opts: TimeTravelOpts): DatabaseBackup {
155
+ const clock = opts.now ?? (() => new Date());
156
+ const exportPath = opts.exportPath ?? `./${opts.databaseName}-export.sql`;
157
+ return {
158
+ needsPeriodicCapture: false,
159
+
160
+ async capture(): Promise<BackupPoint> {
161
+ const at = clock().toISOString();
162
+ return {
163
+ kind: 'time-travel',
164
+ ref: at,
165
+ createdAt: at,
166
+ // Cloudflare does not report a size for a restore point, and a fabricated
167
+ // number would read as a measurement.
168
+ bytes: null,
169
+ restoreCommand: `wrangler d1 time-travel restore ${opts.databaseName} --timestamp=${at}`,
170
+ exportCommand: `wrangler d1 export ${opts.databaseName} --remote --output=${exportPath}`,
171
+ };
172
+ },
173
+ };
174
+ }
175
+
176
+ /**
177
+ * Pick the backup implementation from the driver flavor, so an app's job definition reads
178
+ * the same on both sides.
179
+ */
180
+ export function backupFor(
181
+ flavor: 'local' | 'd1',
182
+ local: () => LocalBackupOpts,
183
+ remote: () => TimeTravelOpts,
184
+ ): DatabaseBackup {
185
+ return flavor === 'local' ? createLocalBackup(local()) : createTimeTravelBackup(remote());
186
+ }