cursedbelt-server 4.16.0 → 4.17.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -82,3 +82,4 @@ export declare function createLocalBackup(opts: LocalBackupOpts): DatabaseBackup
82
82
  * a host that genuinely has both.
83
83
  */
84
84
  export declare function backupFor(flavor: 'local' | 'd1', local: () => LocalBackupOpts, remote: () => TimeTravelOpts): DatabaseBackup;
85
+ export { type BackupSourceChoice, type BackupSourceInput, backupTables, chooseBackupSource, type PullD1Options, pullD1ToSqlite, servedRuntime, type Wrangler, } from './pullD1.js';
@@ -108,3 +108,7 @@ export function createLocalBackup(opts) {
108
108
  export function backupFor(flavor, local, remote) {
109
109
  return flavor === 'local' ? createLocalBackup(local()) : createTimeTravelBackup(remote());
110
110
  }
111
+ // After a Worker cutover the live rows are D1's, and a Mac-hosted backup must pull them down
112
+ // rather than snapshot the frozen file — see `./pullD1.ts`. Here, not in the `./d1` barrel,
113
+ // for the reason this file's header gives: it reaches `bun:sqlite`.
114
+ export { backupTables, chooseBackupSource, pullD1ToSqlite, servedRuntime, } from './pullD1.js';
@@ -0,0 +1,90 @@
1
+ /**
2
+ * After a Worker cutover, the Mac's nightly backup must snapshot the LIVE rows — D1 — and not
3
+ * the Mac's frozen file. The three pieces of that, which `apps/collections` wrote first
4
+ * (2026-09-19/22) and `apps/music` needed next, so they live here once:
5
+ *
6
+ * · {@link chooseBackupSource} — which side holds the live rows, asked of the production
7
+ * hostname's `/healthz` rather than of a config flag, and REFUSING when it cannot tell;
8
+ * · {@link servedRuntime} — that probe;
9
+ * · {@link pullD1ToSqlite} — the live database pulled down into a real SQLite file, table by
10
+ * table ({@link backupTables}), so the app's existing whole-file snapshot path is unchanged.
11
+ *
12
+ * ## 🔴 Why the production hostname and not `cursed.app.runtime`
13
+ *
14
+ * An app declares `runtime: "worker"` when it is PORTED, which can be months before a single
15
+ * row moves. Keying on the marker points the nightly backup at an empty D1 the day the port
16
+ * lands. `/healthz` publishes `runtime` for exactly this question — ask the hostname who
17
+ * answered. And an unreadable answer REFUSES: guessing the Mac snapshots a frozen file after
18
+ * the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
19
+ * D1 snapshots a stale copy while the Mac is live.
20
+ *
21
+ * ## 🔴 Why table by table, and through `wrangler`
22
+ *
23
+ * `wrangler d1 export` REFUSES a whole database holding a virtual table — *"cannot export
24
+ * databases with Virtual Tables (fts5)"*, measured on collections' first post-cutover night —
25
+ * and a per-table export is allowed. `wrangler` rather than the REST API because the export
26
+ * is typed SQL: the REST API's JSON cannot tell the REAL `5.0` from the INTEGER `5`
27
+ * (`./http.ts` has the measurement), and a restore must put back what was there.
28
+ */
29
+ export type BackupSourceChoice = {
30
+ kind: 'file' | 'd1' | 'refuse';
31
+ why: string;
32
+ };
33
+ export interface BackupSourceInput {
34
+ /** `--from-file`: the operator overriding all of this on purpose. */
35
+ fromFileFlag: boolean;
36
+ /** `package.json`'s `cursed.app.runtime`. NOT sufficient on its own — see the header. */
37
+ runtimeMarker: string | undefined;
38
+ /** What the PRODUCTION hostname's `/healthz` said its `runtime` was; `null` = no usable answer. */
39
+ servedRuntime: string | null;
40
+ }
41
+ /** Which side a backup must read. Pure, so every branch is testable without a network. */
42
+ export declare function chooseBackupSource(input: BackupSourceInput): BackupSourceChoice;
43
+ /**
44
+ * What `/healthz` says `runtime` is at `url`, or `null` if no usable answer came back — which
45
+ * means UNKNOWN and is never conflated with "not the Worker". Three attempts, because one blip
46
+ * must not cost a night's backup; a short timeout, because a hung fetch leaves launchd with a
47
+ * job that never exits.
48
+ */
49
+ export declare function servedRuntime(url: string, { attempts, sleep }?: {
50
+ attempts?: number | undefined;
51
+ sleep?: ((ms: number) => Promise<void>) | undefined;
52
+ }): Promise<string | null>;
53
+ /** One `wrangler` invocation, as {@link pullD1ToSqlite} needs it. Injected so a spec runs none. */
54
+ export type Wrangler = (args: string[]) => {
55
+ code: number;
56
+ stdout: string;
57
+ stderr: string;
58
+ };
59
+ /**
60
+ * The tables a snapshot carries, from D1's own `sqlite_master` rows — every real table, and
61
+ * none of what cannot or must not be re-inserted: a virtual table and its shadows (`_config`,
62
+ * `_data`, `_docsize`, `_idx`, `_content` — an inverted index, not rows), `_cf_*`
63
+ * (Cloudflare's bookkeeping), and every `sqlite_*` table except `sqlite_sequence`, whose
64
+ * AUTOINCREMENT high-water marks belong in a restore.
65
+ */
66
+ export declare function backupTables(rows: ReadonlyArray<{
67
+ name: string;
68
+ sql: string | null;
69
+ }>): string[];
70
+ export interface PullD1Options {
71
+ /** The SQLite file to create. */
72
+ destination: string;
73
+ /** The D1 database's NAME as `wrangler.jsonc` declares it at the top level. */
74
+ database: string;
75
+ /** The whole schema — normally the app's `db/schema.sql`. */
76
+ schemaSql: string;
77
+ /** A table that must be listed, or the pull is refused as the wrong database. */
78
+ requiredTable: string;
79
+ /** Run after the rows are loaded — where an app REBUILDS its search index (never copied). */
80
+ afterLoadSql?: string;
81
+ wrangler: Wrangler;
82
+ }
83
+ /**
84
+ * Pull the live D1 database into a real SQLite file.
85
+ *
86
+ * 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
87
+ * SQLite file as success, so an export that produced nothing would sail through the
88
+ * magic-byte check and `integrity_check` and land on the shelf with today's date on it.
89
+ */
90
+ export declare function pullD1ToSqlite(options: PullD1Options): Promise<void>;
@@ -0,0 +1,159 @@
1
+ /**
2
+ * After a Worker cutover, the Mac's nightly backup must snapshot the LIVE rows — D1 — and not
3
+ * the Mac's frozen file. The three pieces of that, which `apps/collections` wrote first
4
+ * (2026-09-19/22) and `apps/music` needed next, so they live here once:
5
+ *
6
+ * · {@link chooseBackupSource} — which side holds the live rows, asked of the production
7
+ * hostname's `/healthz` rather than of a config flag, and REFUSING when it cannot tell;
8
+ * · {@link servedRuntime} — that probe;
9
+ * · {@link pullD1ToSqlite} — the live database pulled down into a real SQLite file, table by
10
+ * table ({@link backupTables}), so the app's existing whole-file snapshot path is unchanged.
11
+ *
12
+ * ## 🔴 Why the production hostname and not `cursed.app.runtime`
13
+ *
14
+ * An app declares `runtime: "worker"` when it is PORTED, which can be months before a single
15
+ * row moves. Keying on the marker points the nightly backup at an empty D1 the day the port
16
+ * lands. `/healthz` publishes `runtime` for exactly this question — ask the hostname who
17
+ * answered. And an unreadable answer REFUSES: guessing the Mac snapshots a frozen file after
18
+ * the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
19
+ * D1 snapshots a stale copy while the Mac is live.
20
+ *
21
+ * ## 🔴 Why table by table, and through `wrangler`
22
+ *
23
+ * `wrangler d1 export` REFUSES a whole database holding a virtual table — *"cannot export
24
+ * databases with Virtual Tables (fts5)"*, measured on collections' first post-cutover night —
25
+ * and a per-table export is allowed. `wrangler` rather than the REST API because the export
26
+ * is typed SQL: the REST API's JSON cannot tell the REAL `5.0` from the INTEGER `5`
27
+ * (`./http.ts` has the measurement), and a restore must put back what was there.
28
+ */
29
+ import { Database } from 'bun:sqlite';
30
+ import { existsSync, readFileSync } from 'node:fs';
31
+ /** Which side a backup must read. Pure, so every branch is testable without a network. */
32
+ export function chooseBackupSource(input) {
33
+ if (input.fromFileFlag) {
34
+ return { kind: 'file', why: "--from-file was given: snapshotting this Mac's sqlite on purpose" };
35
+ }
36
+ if (input.runtimeMarker !== 'worker') {
37
+ return { kind: 'file', why: 'this app does not declare a worker runtime, so the Mac holds the live rows' };
38
+ }
39
+ if (input.servedRuntime === null) {
40
+ return {
41
+ kind: 'refuse',
42
+ why: "could not read `runtime` from the production hostname's /healthz, so which side holds the live " +
43
+ 'rows is unknown. Guessing the Mac would snapshot a database that may have been frozen at the ' +
44
+ 'cutover; guessing D1 would snapshot a stale copy while the Mac is still serving. Neither is a ' +
45
+ 'backup. Fix the probe, or run `bun run backup -- --from-file` if you know the Mac is live.',
46
+ };
47
+ }
48
+ if (input.servedRuntime === 'worker') {
49
+ return { kind: 'd1', why: 'the production hostname is served by the Worker, so the live rows are in D1' };
50
+ }
51
+ return {
52
+ kind: 'file',
53
+ why: `the production hostname is still served by '${input.servedRuntime}', so the Mac holds the live rows`,
54
+ };
55
+ }
56
+ /**
57
+ * What `/healthz` says `runtime` is at `url`, or `null` if no usable answer came back — which
58
+ * means UNKNOWN and is never conflated with "not the Worker". Three attempts, because one blip
59
+ * must not cost a night's backup; a short timeout, because a hung fetch leaves launchd with a
60
+ * job that never exits.
61
+ */
62
+ export async function servedRuntime(url, { attempts = 3, sleep = (ms) => new Promise((r) => setTimeout(r, ms)) } = {}) {
63
+ for (let attempt = 1; attempt <= attempts; attempt++) {
64
+ try {
65
+ const response = await fetch(`${url.replace(/\/+$/, '')}/healthz`, {
66
+ signal: AbortSignal.timeout(10_000),
67
+ headers: { 'cache-control': 'no-cache' },
68
+ });
69
+ if (!response.ok)
70
+ throw new Error(`HTTP ${response.status}`);
71
+ const body = (await response.json());
72
+ // A body with no `runtime` is not an answer.
73
+ if (typeof body.runtime === 'string' && body.runtime)
74
+ return body.runtime;
75
+ return null;
76
+ }
77
+ catch {
78
+ if (attempt === attempts)
79
+ return null;
80
+ await sleep(2000 * attempt);
81
+ }
82
+ }
83
+ return null;
84
+ }
85
+ /**
86
+ * The tables a snapshot carries, from D1's own `sqlite_master` rows — every real table, and
87
+ * none of what cannot or must not be re-inserted: a virtual table and its shadows (`_config`,
88
+ * `_data`, `_docsize`, `_idx`, `_content` — an inverted index, not rows), `_cf_*`
89
+ * (Cloudflare's bookkeeping), and every `sqlite_*` table except `sqlite_sequence`, whose
90
+ * AUTOINCREMENT high-water marks belong in a restore.
91
+ */
92
+ export function backupTables(rows) {
93
+ const virtual = rows.filter((r) => /^\s*CREATE\s+VIRTUAL\s+TABLE/i.test(r.sql ?? '')).map((r) => r.name);
94
+ const shadow = new Set(virtual.flatMap((vt) => ['config', 'data', 'docsize', 'idx', 'content'].map((suffix) => `${vt}_${suffix}`)));
95
+ return rows
96
+ .map((r) => r.name)
97
+ .filter((name) => !virtual.includes(name) && !shadow.has(name))
98
+ .filter((name) => !name.startsWith('_cf_'))
99
+ .filter((name) => name === 'sqlite_sequence' || !name.startsWith('sqlite_'))
100
+ .sort();
101
+ }
102
+ /**
103
+ * Pull the live D1 database into a real SQLite file.
104
+ *
105
+ * 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
106
+ * SQLite file as success, so an export that produced nothing would sail through the
107
+ * magic-byte check and `integrity_check` and land on the shelf with today's date on it.
108
+ */
109
+ export async function pullD1ToSqlite(options) {
110
+ const { destination, database, wrangler } = options;
111
+ const listed = wrangler([
112
+ 'd1', 'execute', database, '--remote', '--env=', '--json',
113
+ '--command', "select name, sql from sqlite_master where type = 'table'",
114
+ ]);
115
+ if (listed.code !== 0)
116
+ throw new Error(`could not list D1's tables: ${listed.stderr.trim() || listed.stdout.trim()}`);
117
+ const json = listed.stdout.slice(listed.stdout.indexOf('['));
118
+ const rows = JSON.parse(json)[0]?.results ?? [];
119
+ const tables = backupTables(rows);
120
+ if (!tables.includes(options.requiredTable)) {
121
+ throw new Error(`D1 lists no \`${options.requiredTable}\` table (saw: ${rows.map((r) => r.name).join(', ') || 'nothing'})`);
122
+ }
123
+ let sql = '';
124
+ for (const table of tables) {
125
+ const dump = `${destination}.${table}.sql`;
126
+ const exported = wrangler(['d1', 'export', database, '--remote', '--env=', '--table', table, '--no-schema', '--output', dump, '-y']);
127
+ if (exported.code !== 0) {
128
+ throw new Error(`wrangler d1 export --table ${table} failed: ${exported.stderr.trim() || exported.stdout.trim()}`);
129
+ }
130
+ if (!existsSync(dump))
131
+ throw new Error(`wrangler d1 export --table ${table} reported success and wrote no file at ${dump}`);
132
+ sql += `${readFileSync(dump, 'utf8')}\n`;
133
+ }
134
+ if (!/INSERT INTO/i.test(sql)) {
135
+ throw new Error('the D1 export contains no rows — refusing to snapshot an empty database as if it were the library');
136
+ }
137
+ const db = new Database(destination, { create: true });
138
+ try {
139
+ // `exec`, not `run`: each is many statements and `run` would execute only the first,
140
+ // leaving a file with a schema and no rows that looks perfectly valid.
141
+ db.exec(options.schemaSql);
142
+ // 🔴 A table D1 holds that the schema does not declare is created from D1's OWN DDL, never
143
+ // dropped: measured on music's first post-cutover pull (2026-09-23), `art_lookups` — a
144
+ // cache its art job creates where it first needs it — made the whole pull refuse with
145
+ // `no such table`, because its rows were exported and its table was not in the schema.
146
+ const have = new Set(db.query("SELECT name FROM sqlite_master WHERE type = 'table'").all().map((r) => r.name));
147
+ for (const row of rows) {
148
+ if (!tables.includes(row.name) || have.has(row.name) || !row.sql)
149
+ continue;
150
+ db.exec(row.sql.replace(/^\s*CREATE\s+TABLE\s+(?!IF\s+NOT\s+EXISTS)/i, 'CREATE TABLE IF NOT EXISTS '));
151
+ }
152
+ db.exec(sql);
153
+ if (options.afterLoadSql)
154
+ db.exec(options.afterLoadSql);
155
+ }
156
+ finally {
157
+ db.close();
158
+ }
159
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "cursedbelt-server",
3
- "version": "4.16.0",
3
+ "version": "4.17.1",
4
4
  "license": "ISC",
5
5
  "type": "module",
6
6
  "description": "The app-facing Bun/Hono server tier of the cursedbelt split — storage, sharing, activity, guard, sync. React-free; cursedbelt-core below it.",
@@ -135,3 +135,17 @@ export function backupFor(
135
135
  ): DatabaseBackup {
136
136
  return flavor === 'local' ? createLocalBackup(local()) : createTimeTravelBackup(remote());
137
137
  }
138
+
139
+ // After a Worker cutover the live rows are D1's, and a Mac-hosted backup must pull them down
140
+ // rather than snapshot the frozen file — see `./pullD1.ts`. Here, not in the `./d1` barrel,
141
+ // for the reason this file's header gives: it reaches `bun:sqlite`.
142
+ export {
143
+ type BackupSourceChoice,
144
+ type BackupSourceInput,
145
+ backupTables,
146
+ chooseBackupSource,
147
+ type PullD1Options,
148
+ pullD1ToSqlite,
149
+ servedRuntime,
150
+ type Wrangler,
151
+ } from './pullD1.js';
@@ -0,0 +1,186 @@
1
+ import { Database } from 'bun:sqlite';
2
+ import { afterEach, describe, expect, test } from 'bun:test';
3
+ import { mkdtempSync, rmSync, writeFileSync } from 'node:fs';
4
+ import { tmpdir } from 'node:os';
5
+ import { join } from 'node:path';
6
+ import { backupTables, chooseBackupSource, pullD1ToSqlite, servedRuntime, type Wrangler } from './localBackup.js';
7
+
8
+ const dirs: string[] = [];
9
+ const scratch = (): string => {
10
+ const dir = mkdtempSync(join(tmpdir(), 'd1-pull-'));
11
+ dirs.push(dir);
12
+ return dir;
13
+ };
14
+ afterEach(() => {
15
+ for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true });
16
+ });
17
+
18
+ describe('🔴 chooseBackupSource — the one whose failure looks like a green backup', () => {
19
+ const WORKER = 'worker';
20
+
21
+ test('a ported app whose production hostname is still the Mac reads the MAC file', () => {
22
+ const choice = chooseBackupSource({ fromFileFlag: false, runtimeMarker: WORKER, servedRuntime: 'bun' });
23
+ expect(choice.kind).toBe('file');
24
+ expect(choice.why).toContain("still served by 'bun'");
25
+ });
26
+
27
+ test('once the production hostname answers as the Worker, it pulls D1', () => {
28
+ expect(chooseBackupSource({ fromFileFlag: false, runtimeMarker: WORKER, servedRuntime: WORKER }).kind).toBe('d1');
29
+ });
30
+
31
+ test('🔴 an unreadable probe REFUSES — it never guesses a side', () => {
32
+ const choice = chooseBackupSource({ fromFileFlag: false, runtimeMarker: WORKER, servedRuntime: null });
33
+ expect(choice.kind).toBe('refuse');
34
+ expect(choice.why).toContain('--from-file');
35
+ });
36
+
37
+ test('an app that never declared a worker runtime never pulls', () => {
38
+ for (const marker of [undefined, 'bun', '']) {
39
+ expect(chooseBackupSource({ fromFileFlag: false, runtimeMarker: marker, servedRuntime: null }).kind).toBe('file');
40
+ }
41
+ });
42
+
43
+ test('--from-file wins over everything, including a live Worker', () => {
44
+ const choice = chooseBackupSource({ fromFileFlag: true, runtimeMarker: WORKER, servedRuntime: WORKER });
45
+ expect(choice.kind).toBe('file');
46
+ expect(choice.why).toContain('on purpose');
47
+ });
48
+ });
49
+
50
+ describe('servedRuntime', () => {
51
+ const serve = (body: unknown, status = 200) =>
52
+ Bun.serve({ port: 0, fetch: () => Response.json(body, { status }) });
53
+
54
+ test('reads `runtime` off /healthz', async () => {
55
+ const server = serve({ runtime: 'worker' });
56
+ try {
57
+ expect(await servedRuntime(`http://127.0.0.1:${server.port}`)).toBe('worker');
58
+ } finally {
59
+ server.stop(true);
60
+ }
61
+ });
62
+
63
+ test('🔴 a body with no runtime, or an error, is UNKNOWN — never "not the Worker"', async () => {
64
+ const bare = serve({ ok: true });
65
+ const broken = serve({ runtime: 'worker' }, 503);
66
+ try {
67
+ const noSleep = { attempts: 2, sleep: async () => undefined };
68
+ expect(await servedRuntime(`http://127.0.0.1:${bare.port}`, noSleep)).toBeNull();
69
+ expect(await servedRuntime(`http://127.0.0.1:${broken.port}`, noSleep)).toBeNull();
70
+ } finally {
71
+ bare.stop(true);
72
+ broken.stop(true);
73
+ }
74
+ });
75
+ });
76
+
77
+ describe('pulling D1 down — table by table, because a whole-database export refuses fts5', () => {
78
+ const MASTER = [
79
+ { name: '_cf_KV', sql: 'CREATE TABLE _cf_KV (key TEXT PRIMARY KEY)' },
80
+ { name: 'items', sql: 'CREATE TABLE items ( id TEXT PRIMARY KEY )' },
81
+ { name: 'items_fts', sql: 'CREATE VIRTUAL TABLE items_fts USING fts5(text)' },
82
+ { name: 'items_fts_config', sql: "CREATE TABLE 'items_fts_config'(k PRIMARY KEY, v)" },
83
+ { name: 'items_fts_data', sql: "CREATE TABLE 'items_fts_data'(id INTEGER PRIMARY KEY, block BLOB)" },
84
+ { name: 'items_fts_docsize', sql: "CREATE TABLE 'items_fts_docsize'(id INTEGER PRIMARY KEY, sz BLOB)" },
85
+ { name: 'items_fts_idx', sql: "CREATE TABLE 'items_fts_idx'(segid, term, pgno)" },
86
+ { name: 'items_fts_map', sql: 'CREATE TABLE items_fts_map ( rowid INTEGER PRIMARY KEY AUTOINCREMENT )' },
87
+ { name: 'meta', sql: 'CREATE TABLE meta ( key TEXT PRIMARY KEY )' },
88
+ { name: 'sqlite_sequence', sql: 'CREATE TABLE sqlite_sequence(name,seq)' },
89
+ { name: 'sqlite_stat1', sql: 'CREATE TABLE sqlite_stat1(tbl,idx,stat)' },
90
+ ];
91
+ const SCHEMA = [
92
+ 'CREATE TABLE items (id TEXT PRIMARY KEY, name TEXT NOT NULL DEFAULT "");',
93
+ 'CREATE TABLE items_fts_map (rowid INTEGER PRIMARY KEY AUTOINCREMENT, item_id TEXT, text TEXT);',
94
+ 'CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT);',
95
+ 'CREATE VIRTUAL TABLE items_fts USING fts5(text);',
96
+ ].join('\n');
97
+ const REBUILD = "INSERT INTO items_fts(items_fts) VALUES('delete-all'); INSERT INTO items_fts(rowid, text) SELECT rowid, text FROM items_fts_map;";
98
+
99
+ test('🔴 the virtual table and its shadows stay out, the MAP and the sequence come in', () => {
100
+ expect(backupTables(MASTER)).toEqual(['items', 'items_fts_map', 'meta', 'sqlite_sequence']);
101
+ });
102
+
103
+ const fake = (rowsFor: (table: string) => string | null): { run: Wrangler; exported: string[]; seen: string[][] } => {
104
+ const exported: string[] = [];
105
+ const seen: string[][] = [];
106
+ const run: Wrangler = (args) => {
107
+ seen.push(args);
108
+ if (args[1] === 'execute') return { code: 0, stdout: `noise\n${JSON.stringify([{ results: MASTER }])}`, stderr: '' };
109
+ const table = args[args.indexOf('--table') + 1] as string;
110
+ const out = args[args.indexOf('--output') + 1] as string;
111
+ const rows = rowsFor(table);
112
+ if (rows === null) return { code: 1, stdout: '', stderr: `✘ export of ${table} failed` };
113
+ exported.push(table);
114
+ writeFileSync(out, rows);
115
+ return { code: 0, stdout: '', stderr: '' };
116
+ };
117
+ return { run, exported, seen };
118
+ };
119
+ const pull = (dir: string, run: Wrangler) =>
120
+ pullD1ToSqlite({
121
+ destination: join(dir, 'pulled.sqlite'),
122
+ database: 'collections',
123
+ schemaSql: SCHEMA,
124
+ requiredTable: 'items',
125
+ afterLoadSql: REBUILD,
126
+ wrangler: run,
127
+ });
128
+
129
+ test('a pull is a real, searchable sqlite file built from the schema and the rows, of the named database', async () => {
130
+ const dir = scratch();
131
+ const { run, exported, seen } = fake((table) =>
132
+ table === 'items'
133
+ ? "INSERT INTO items (id, name) VALUES ('itm_1', 'Travels 5388.jpg');"
134
+ : table === 'items_fts_map'
135
+ ? "INSERT INTO items_fts_map (rowid, item_id, text) VALUES (1, 'itm_1', 'travels 5388');"
136
+ : '',
137
+ );
138
+ await pull(dir, run);
139
+ expect(exported).toEqual(['items', 'items_fts_map', 'meta', 'sqlite_sequence']);
140
+ expect(seen.every((args) => args[2] === 'collections')).toBe(true);
141
+ const db = new Database(join(dir, 'pulled.sqlite'), { readonly: true });
142
+ const hits = db.query("select count(*) as n from items_fts where items_fts match '5388'").get() as { n: number };
143
+ db.close();
144
+ expect(hits.n).toBe(1);
145
+ });
146
+
147
+ test('🔴 a table D1 holds that the schema does not declare is created from D1\'s own DDL', async () => {
148
+ const dir = scratch();
149
+ const master = [...MASTER, { name: 'art_lookups', sql: 'CREATE TABLE art_lookups (key TEXT PRIMARY KEY, art_key TEXT)' }];
150
+ const run: Wrangler = (args) => {
151
+ if (args[1] === 'execute') return { code: 0, stdout: JSON.stringify([{ results: master }]), stderr: '' };
152
+ const table = args[args.indexOf('--table') + 1] as string;
153
+ const out = args[args.indexOf('--output') + 1] as string;
154
+ writeFileSync(
155
+ out,
156
+ table === 'items' ? "INSERT INTO items (id) VALUES ('i1');" : table === 'art_lookups' ? "INSERT INTO art_lookups VALUES ('k', 'art/1');" : '',
157
+ );
158
+ return { code: 0, stdout: '', stderr: '' };
159
+ };
160
+ await pull(dir, run);
161
+ const db = new Database(join(dir, 'pulled.sqlite'), { readonly: true });
162
+ const row = db.query('select art_key from art_lookups').get() as { art_key: string } | null;
163
+ db.close();
164
+ expect(row?.art_key).toBe('art/1');
165
+ });
166
+
167
+ test('🔴 a database without the required table is refused as the wrong one', async () => {
168
+ const dir = scratch();
169
+ const { run } = fake(() => '');
170
+ await expect(
171
+ pullD1ToSqlite({ destination: join(dir, 'x.sqlite'), database: 'x', schemaSql: SCHEMA, requiredTable: 'tracks', wrangler: run }),
172
+ ).rejects.toThrow(/lists no `tracks` table/);
173
+ });
174
+
175
+ test('🔴 one table failing to export THROWS — never a snapshot missing that table', async () => {
176
+ const dir = scratch();
177
+ const { run } = fake((table) => (table === 'meta' ? null : "INSERT INTO items (id) VALUES ('x');"));
178
+ await expect(pull(dir, run)).rejects.toThrow(/--table meta failed/);
179
+ });
180
+
181
+ test('🔴 an export with no rows at all THROWS — never an empty library dated today', async () => {
182
+ const dir = scratch();
183
+ const { run } = fake(() => '');
184
+ await expect(pull(dir, run)).rejects.toThrow(/no rows/);
185
+ });
186
+ });
@@ -0,0 +1,195 @@
1
+ /**
2
+ * After a Worker cutover, the Mac's nightly backup must snapshot the LIVE rows — D1 — and not
3
+ * the Mac's frozen file. The three pieces of that, which `apps/collections` wrote first
4
+ * (2026-09-19/22) and `apps/music` needed next, so they live here once:
5
+ *
6
+ * · {@link chooseBackupSource} — which side holds the live rows, asked of the production
7
+ * hostname's `/healthz` rather than of a config flag, and REFUSING when it cannot tell;
8
+ * · {@link servedRuntime} — that probe;
9
+ * · {@link pullD1ToSqlite} — the live database pulled down into a real SQLite file, table by
10
+ * table ({@link backupTables}), so the app's existing whole-file snapshot path is unchanged.
11
+ *
12
+ * ## 🔴 Why the production hostname and not `cursed.app.runtime`
13
+ *
14
+ * An app declares `runtime: "worker"` when it is PORTED, which can be months before a single
15
+ * row moves. Keying on the marker points the nightly backup at an empty D1 the day the port
16
+ * lands. `/healthz` publishes `runtime` for exactly this question — ask the hostname who
17
+ * answered. And an unreadable answer REFUSES: guessing the Mac snapshots a frozen file after
18
+ * the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
19
+ * D1 snapshots a stale copy while the Mac is live.
20
+ *
21
+ * ## 🔴 Why table by table, and through `wrangler`
22
+ *
23
+ * `wrangler d1 export` REFUSES a whole database holding a virtual table — *"cannot export
24
+ * databases with Virtual Tables (fts5)"*, measured on collections' first post-cutover night —
25
+ * and a per-table export is allowed. `wrangler` rather than the REST API because the export
26
+ * is typed SQL: the REST API's JSON cannot tell the REAL `5.0` from the INTEGER `5`
27
+ * (`./http.ts` has the measurement), and a restore must put back what was there.
28
+ */
29
+
30
+ import { Database } from 'bun:sqlite';
31
+ import { existsSync, readFileSync } from 'node:fs';
32
+
33
+ export type BackupSourceChoice = { kind: 'file' | 'd1' | 'refuse'; why: string };
34
+
35
+ export interface BackupSourceInput {
36
+ /** `--from-file`: the operator overriding all of this on purpose. */
37
+ fromFileFlag: boolean;
38
+ /** `package.json`'s `cursed.app.runtime`. NOT sufficient on its own — see the header. */
39
+ runtimeMarker: string | undefined;
40
+ /** What the PRODUCTION hostname's `/healthz` said its `runtime` was; `null` = no usable answer. */
41
+ servedRuntime: string | null;
42
+ }
43
+
44
+ /** Which side a backup must read. Pure, so every branch is testable without a network. */
45
+ export function chooseBackupSource(input: BackupSourceInput): BackupSourceChoice {
46
+ if (input.fromFileFlag) {
47
+ return { kind: 'file', why: "--from-file was given: snapshotting this Mac's sqlite on purpose" };
48
+ }
49
+ if (input.runtimeMarker !== 'worker') {
50
+ return { kind: 'file', why: 'this app does not declare a worker runtime, so the Mac holds the live rows' };
51
+ }
52
+ if (input.servedRuntime === null) {
53
+ return {
54
+ kind: 'refuse',
55
+ why:
56
+ "could not read `runtime` from the production hostname's /healthz, so which side holds the live " +
57
+ 'rows is unknown. Guessing the Mac would snapshot a database that may have been frozen at the ' +
58
+ 'cutover; guessing D1 would snapshot a stale copy while the Mac is still serving. Neither is a ' +
59
+ 'backup. Fix the probe, or run `bun run backup -- --from-file` if you know the Mac is live.',
60
+ };
61
+ }
62
+ if (input.servedRuntime === 'worker') {
63
+ return { kind: 'd1', why: 'the production hostname is served by the Worker, so the live rows are in D1' };
64
+ }
65
+ return {
66
+ kind: 'file',
67
+ why: `the production hostname is still served by '${input.servedRuntime}', so the Mac holds the live rows`,
68
+ };
69
+ }
70
+
71
+ /**
72
+ * What `/healthz` says `runtime` is at `url`, or `null` if no usable answer came back — which
73
+ * means UNKNOWN and is never conflated with "not the Worker". Three attempts, because one blip
74
+ * must not cost a night's backup; a short timeout, because a hung fetch leaves launchd with a
75
+ * job that never exits.
76
+ */
77
+ export async function servedRuntime(
78
+ url: string,
79
+ { attempts = 3, sleep = (ms: number) => new Promise<void>((r) => setTimeout(r, ms)) } = {},
80
+ ): Promise<string | null> {
81
+ for (let attempt = 1; attempt <= attempts; attempt++) {
82
+ try {
83
+ const response = await fetch(`${url.replace(/\/+$/, '')}/healthz`, {
84
+ signal: AbortSignal.timeout(10_000),
85
+ headers: { 'cache-control': 'no-cache' },
86
+ });
87
+ if (!response.ok) throw new Error(`HTTP ${response.status}`);
88
+ const body = (await response.json()) as { runtime?: unknown };
89
+ // A body with no `runtime` is not an answer.
90
+ if (typeof body.runtime === 'string' && body.runtime) return body.runtime;
91
+ return null;
92
+ } catch {
93
+ if (attempt === attempts) return null;
94
+ await sleep(2000 * attempt);
95
+ }
96
+ }
97
+ return null;
98
+ }
99
+
100
+ /** One `wrangler` invocation, as {@link pullD1ToSqlite} needs it. Injected so a spec runs none. */
101
+ export type Wrangler = (args: string[]) => { code: number; stdout: string; stderr: string };
102
+
103
+ /**
104
+ * The tables a snapshot carries, from D1's own `sqlite_master` rows — every real table, and
105
+ * none of what cannot or must not be re-inserted: a virtual table and its shadows (`_config`,
106
+ * `_data`, `_docsize`, `_idx`, `_content` — an inverted index, not rows), `_cf_*`
107
+ * (Cloudflare's bookkeeping), and every `sqlite_*` table except `sqlite_sequence`, whose
108
+ * AUTOINCREMENT high-water marks belong in a restore.
109
+ */
110
+ export function backupTables(rows: ReadonlyArray<{ name: string; sql: string | null }>): string[] {
111
+ const virtual = rows.filter((r) => /^\s*CREATE\s+VIRTUAL\s+TABLE/i.test(r.sql ?? '')).map((r) => r.name);
112
+ const shadow = new Set(
113
+ virtual.flatMap((vt) => ['config', 'data', 'docsize', 'idx', 'content'].map((suffix) => `${vt}_${suffix}`)),
114
+ );
115
+ return rows
116
+ .map((r) => r.name)
117
+ .filter((name) => !virtual.includes(name) && !shadow.has(name))
118
+ .filter((name) => !name.startsWith('_cf_'))
119
+ .filter((name) => name === 'sqlite_sequence' || !name.startsWith('sqlite_'))
120
+ .sort();
121
+ }
122
+
123
+ export interface PullD1Options {
124
+ /** The SQLite file to create. */
125
+ destination: string;
126
+ /** The D1 database's NAME as `wrangler.jsonc` declares it at the top level. */
127
+ database: string;
128
+ /** The whole schema — normally the app's `db/schema.sql`. */
129
+ schemaSql: string;
130
+ /** A table that must be listed, or the pull is refused as the wrong database. */
131
+ requiredTable: string;
132
+ /** Run after the rows are loaded — where an app REBUILDS its search index (never copied). */
133
+ afterLoadSql?: string;
134
+ wrangler: Wrangler;
135
+ }
136
+
137
+ /**
138
+ * Pull the live D1 database into a real SQLite file.
139
+ *
140
+ * 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
141
+ * SQLite file as success, so an export that produced nothing would sail through the
142
+ * magic-byte check and `integrity_check` and land on the shelf with today's date on it.
143
+ */
144
+ export async function pullD1ToSqlite(options: PullD1Options): Promise<void> {
145
+ const { destination, database, wrangler } = options;
146
+ const listed = wrangler([
147
+ 'd1', 'execute', database, '--remote', '--env=', '--json',
148
+ '--command', "select name, sql from sqlite_master where type = 'table'",
149
+ ]);
150
+ if (listed.code !== 0) throw new Error(`could not list D1's tables: ${listed.stderr.trim() || listed.stdout.trim()}`);
151
+ const json = listed.stdout.slice(listed.stdout.indexOf('['));
152
+ const rows = (JSON.parse(json) as Array<{ results: Array<{ name: string; sql: string | null }> }>)[0]?.results ?? [];
153
+ const tables = backupTables(rows);
154
+ if (!tables.includes(options.requiredTable)) {
155
+ throw new Error(
156
+ `D1 lists no \`${options.requiredTable}\` table (saw: ${rows.map((r) => r.name).join(', ') || 'nothing'})`,
157
+ );
158
+ }
159
+
160
+ let sql = '';
161
+ for (const table of tables) {
162
+ const dump = `${destination}.${table}.sql`;
163
+ const exported = wrangler(['d1', 'export', database, '--remote', '--env=', '--table', table, '--no-schema', '--output', dump, '-y']);
164
+ if (exported.code !== 0) {
165
+ throw new Error(`wrangler d1 export --table ${table} failed: ${exported.stderr.trim() || exported.stdout.trim()}`);
166
+ }
167
+ if (!existsSync(dump)) throw new Error(`wrangler d1 export --table ${table} reported success and wrote no file at ${dump}`);
168
+ sql += `${readFileSync(dump, 'utf8')}\n`;
169
+ }
170
+ if (!/INSERT INTO/i.test(sql)) {
171
+ throw new Error('the D1 export contains no rows — refusing to snapshot an empty database as if it were the library');
172
+ }
173
+
174
+ const db = new Database(destination, { create: true });
175
+ try {
176
+ // `exec`, not `run`: each is many statements and `run` would execute only the first,
177
+ // leaving a file with a schema and no rows that looks perfectly valid.
178
+ db.exec(options.schemaSql);
179
+ // 🔴 A table D1 holds that the schema does not declare is created from D1's OWN DDL, never
180
+ // dropped: measured on music's first post-cutover pull (2026-09-23), `art_lookups` — a
181
+ // cache its art job creates where it first needs it — made the whole pull refuse with
182
+ // `no such table`, because its rows were exported and its table was not in the schema.
183
+ const have = new Set(
184
+ (db.query("SELECT name FROM sqlite_master WHERE type = 'table'").all() as { name: string }[]).map((r) => r.name),
185
+ );
186
+ for (const row of rows) {
187
+ if (!tables.includes(row.name) || have.has(row.name) || !row.sql) continue;
188
+ db.exec(row.sql.replace(/^\s*CREATE\s+TABLE\s+(?!IF\s+NOT\s+EXISTS)/i, 'CREATE TABLE IF NOT EXISTS '));
189
+ }
190
+ db.exec(sql);
191
+ if (options.afterLoadSql) db.exec(options.afterLoadSql);
192
+ } finally {
193
+ db.close();
194
+ }
195
+ }