cursedbelt-server 4.31.0 → 4.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/server/d1/pullD1.d.ts +57 -8
- package/dist/server/d1/pullD1.js +157 -89
- package/package.json +1 -1
- package/src/server/d1/pullD1.spec.ts +111 -90
- package/src/server/d1/pullD1.ts +162 -84
|
@@ -18,13 +18,30 @@
|
|
|
18
18
|
* the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
|
|
19
19
|
* D1 snapshots a stale copy while the Mac is live.
|
|
20
20
|
*
|
|
21
|
-
* ## 🔴 Why
|
|
21
|
+
* ## 🔴 Why paged `SELECT`s, and never `wrangler d1 export`
|
|
22
22
|
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
23
|
+
* **A D1 export blocks the database from serving anything else for as long as it runs**
|
|
24
|
+
* (Cloudflare's own docs). Measured on family 2026-09-25 00:32–00:33 UTC: a manual `bun run
|
|
25
|
+
* backup` — then a per-table `wrangler d1 export` — held production for ~28 s at a time, and
|
|
26
|
+
* 14 requests to the family Worker sat waiting and threw Error 1101 (wall p50 28 s, CPU p50
|
|
27
|
+
* 1.6 ms). Two earlier throws at 13:56/13:58 had the same shape. So every nightly backup of
|
|
28
|
+
* every app on this seam was a scheduled outage.
|
|
29
|
+
*
|
|
30
|
+
* Since 4.32.0 the rows are read with ordinary `wrangler d1 execute --remote` SELECTs, a page
|
|
31
|
+
* of {@link PAGE_ROWS} rows at a time on a rowid keyset, which D1 serves between other
|
|
32
|
+
* requests like any read. Each value comes back as SQLite's own `quote()` of it — an SQL
|
|
33
|
+
* literal, as TEXT — so the REAL `5.0` stays `5.0` and the INTEGER `5` stays `5` (the reason
|
|
34
|
+
* this used `export` and not the REST API's JSON: `./http.ts` has the measurement), a BLOB is
|
|
35
|
+
* `X'…'`, and `quote()` round-trips a double exactly. The dump is then the same
|
|
36
|
+
* `INSERT INTO …` text an export wrote, so everything after it is unchanged.
|
|
37
|
+
*
|
|
38
|
+
* Still table by table: a whole-database export refused fts5 (*"cannot export databases with
|
|
39
|
+
* Virtual Tables"*), and a virtual table's shadow tables are an index, rebuilt, never copied.
|
|
40
|
+
*
|
|
41
|
+
* 🔴 The trade: an export was one point in time; pages are not. A row written to one table
|
|
42
|
+
* while another is being paged may be in the snapshot without its partner. For a nightly
|
|
43
|
+
* backup that is the right trade against taking production down, and the row-count check
|
|
44
|
+
* below still refuses a load short of what was read.
|
|
28
45
|
*/
|
|
29
46
|
export type BackupSourceChoice = {
|
|
30
47
|
kind: 'file' | 'd1' | 'refuse';
|
|
@@ -50,6 +67,13 @@ export declare function servedRuntime(url: string, { attempts, sleep }?: {
|
|
|
50
67
|
attempts?: number | undefined;
|
|
51
68
|
sleep?: ((ms: number) => Promise<void>) | undefined;
|
|
52
69
|
}): Promise<string | null>;
|
|
70
|
+
/**
|
|
71
|
+
* Rows per paged `SELECT`. The fleet's largest table held 290 rows when this was set (collections'
|
|
72
|
+
* `items`, 2026-09-25) and one `wrangler` call costs ~1 s, so a page this size is one call per
|
|
73
|
+
* table for every app today. A page D1 refuses — too large a response — is halved and asked
|
|
74
|
+
* again, down to one row, before it is a refusal.
|
|
75
|
+
*/
|
|
76
|
+
export declare const PAGE_ROWS = 500;
|
|
53
77
|
/** One `wrangler` invocation, as {@link pullD1ToSqlite} needs it. Injected so a spec runs none. */
|
|
54
78
|
export type Wrangler = (args: string[]) => {
|
|
55
79
|
code: number;
|
|
@@ -69,6 +93,8 @@ export declare function spawnWrangler(options: {
|
|
|
69
93
|
}): Wrangler;
|
|
70
94
|
/**
|
|
71
95
|
* 🔴 Trailing NUL bytes off a `wrangler d1 export` file, in place. Returns how many were removed.
|
|
96
|
+
* {@link pullD1ToSqlite} no longer exports (see the header), so nothing here calls this since
|
|
97
|
+
* 4.32.0; it stays exported for anyone still reading an export file by hand.
|
|
72
98
|
*
|
|
73
99
|
* Measured 2026-09-23 on vault's D1 (wrangler 4.136.3): `wrangler d1 export --table … --output`
|
|
74
100
|
* INTERMITTENTLY ends the file in hundreds of NUL bytes (`entries` +988, `oplog` +1,369, `sync_ops`
|
|
@@ -104,11 +130,34 @@ export interface PullD1Options {
|
|
|
104
130
|
afterLoadSql?: string;
|
|
105
131
|
wrangler: Wrangler;
|
|
106
132
|
}
|
|
133
|
+
/** How a table is walked in pages: by a rowid keyset, or — a `WITHOUT ROWID` table — by offset. */
|
|
134
|
+
export interface TablePlan {
|
|
135
|
+
table: string;
|
|
136
|
+
columns: string[];
|
|
137
|
+
/** The rowid alias to page on, or `null` for a table with none (then `order` + OFFSET). */
|
|
138
|
+
key: string | null;
|
|
139
|
+
order: string[];
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* The paging plan for one table. The key is the first of `rowid`, `_rowid_`, `oid` that no
|
|
143
|
+
* declared column shadows — `items_fts_map` declares a column NAMED `rowid`, so it pages on
|
|
144
|
+
* `_rowid_`, which is the same value.
|
|
145
|
+
*/
|
|
146
|
+
export declare function tablePlan(table: string, ddl: string | null, cols: ReadonlyArray<{
|
|
147
|
+
name: string;
|
|
148
|
+
pk: number;
|
|
149
|
+
}>): TablePlan;
|
|
150
|
+
/** The SELECT for one page: the key (if any) and every column as its `quote()`d SQL literal. */
|
|
151
|
+
export declare function pageSql(plan: TablePlan, limit: number, after: {
|
|
152
|
+
key?: number;
|
|
153
|
+
offset?: number;
|
|
154
|
+
}): string;
|
|
107
155
|
/**
|
|
108
|
-
* Pull the live D1 database into a real SQLite file
|
|
156
|
+
* Pull the live D1 database into a real SQLite file, reading it with paged SELECTs that never
|
|
157
|
+
* block production (see the header — this used to be `wrangler d1 export`, which did).
|
|
109
158
|
*
|
|
110
159
|
* 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
|
|
111
|
-
* SQLite file as success, so
|
|
160
|
+
* SQLite file as success, so a pull that produced nothing would sail through the
|
|
112
161
|
* magic-byte check and `integrity_check` and land on the shelf with today's date on it.
|
|
113
162
|
*/
|
|
114
163
|
export declare function pullD1ToSqlite(options: PullD1Options): Promise<void>;
|
package/dist/server/d1/pullD1.js
CHANGED
|
@@ -18,13 +18,30 @@
|
|
|
18
18
|
* the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
|
|
19
19
|
* D1 snapshots a stale copy while the Mac is live.
|
|
20
20
|
*
|
|
21
|
-
* ## 🔴 Why
|
|
21
|
+
* ## 🔴 Why paged `SELECT`s, and never `wrangler d1 export`
|
|
22
22
|
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
23
|
+
* **A D1 export blocks the database from serving anything else for as long as it runs**
|
|
24
|
+
* (Cloudflare's own docs). Measured on family 2026-09-25 00:32–00:33 UTC: a manual `bun run
|
|
25
|
+
* backup` — then a per-table `wrangler d1 export` — held production for ~28 s at a time, and
|
|
26
|
+
* 14 requests to the family Worker sat waiting and threw Error 1101 (wall p50 28 s, CPU p50
|
|
27
|
+
* 1.6 ms). Two earlier throws at 13:56/13:58 had the same shape. So every nightly backup of
|
|
28
|
+
* every app on this seam was a scheduled outage.
|
|
29
|
+
*
|
|
30
|
+
* Since 4.32.0 the rows are read with ordinary `wrangler d1 execute --remote` SELECTs, a page
|
|
31
|
+
* of {@link PAGE_ROWS} rows at a time on a rowid keyset, which D1 serves between other
|
|
32
|
+
* requests like any read. Each value comes back as SQLite's own `quote()` of it — an SQL
|
|
33
|
+
* literal, as TEXT — so the REAL `5.0` stays `5.0` and the INTEGER `5` stays `5` (the reason
|
|
34
|
+
* this used `export` and not the REST API's JSON: `./http.ts` has the measurement), a BLOB is
|
|
35
|
+
* `X'…'`, and `quote()` round-trips a double exactly. The dump is then the same
|
|
36
|
+
* `INSERT INTO …` text an export wrote, so everything after it is unchanged.
|
|
37
|
+
*
|
|
38
|
+
* Still table by table: a whole-database export refused fts5 (*"cannot export databases with
|
|
39
|
+
* Virtual Tables"*), and a virtual table's shadow tables are an index, rebuilt, never copied.
|
|
40
|
+
*
|
|
41
|
+
* 🔴 The trade: an export was one point in time; pages are not. A row written to one table
|
|
42
|
+
* while another is being paged may be in the snapshot without its partner. For a nightly
|
|
43
|
+
* backup that is the right trade against taking production down, and the row-count check
|
|
44
|
+
* below still refuses a load short of what was read.
|
|
28
45
|
*/
|
|
29
46
|
import { Database } from 'bun:sqlite';
|
|
30
47
|
import { existsSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
|
@@ -82,8 +99,13 @@ export async function servedRuntime(url, { attempts = 3, sleep = (ms) => new Pro
|
|
|
82
99
|
}
|
|
83
100
|
return null;
|
|
84
101
|
}
|
|
85
|
-
/**
|
|
86
|
-
|
|
102
|
+
/**
|
|
103
|
+
* Rows per paged `SELECT`. The fleet's largest table held 290 rows when this was set (collections'
|
|
104
|
+
* `items`, 2026-09-25) and one `wrangler` call costs ~1 s, so a page this size is one call per
|
|
105
|
+
* table for every app today. A page D1 refuses — too large a response — is halved and asked
|
|
106
|
+
* again, down to one row, before it is a refusal.
|
|
107
|
+
*/
|
|
108
|
+
export const PAGE_ROWS = 500;
|
|
87
109
|
/**
|
|
88
110
|
* The argv that runs `wrangler`: THIS bun binary's `x`, never a bare `bunx` — a launchd job's
|
|
89
111
|
* PATH has no `bunx` (task 2112), and vault's and patterns' nightly backups each carried this
|
|
@@ -104,6 +126,8 @@ export function spawnWrangler(options) {
|
|
|
104
126
|
}
|
|
105
127
|
/**
|
|
106
128
|
* 🔴 Trailing NUL bytes off a `wrangler d1 export` file, in place. Returns how many were removed.
|
|
129
|
+
* {@link pullD1ToSqlite} no longer exports (see the header), so nothing here calls this since
|
|
130
|
+
* 4.32.0; it stays exported for anyone still reading an export file by hand.
|
|
107
131
|
*
|
|
108
132
|
* Measured 2026-09-23 on vault's D1 (wrangler 4.136.3): `wrangler d1 export --table … --output`
|
|
109
133
|
* INTERMITTENTLY ends the file in hundreds of NUL bytes (`entries` +988, `oplog` +1,369, `sync_ops`
|
|
@@ -143,104 +167,148 @@ export function backupTables(rows) {
|
|
|
143
167
|
.filter((name) => name === 'sqlite_sequence' || !name.startsWith('sqlite_'))
|
|
144
168
|
.sort();
|
|
145
169
|
}
|
|
170
|
+
const sqlIdent = (name) => `"${name.replaceAll('"', '""')}"`;
|
|
171
|
+
const sqlText = (text) => `'${text.replaceAll("'", "''")}'`;
|
|
172
|
+
/** One `wrangler d1 execute --remote --json` — every statement's result rows, in order. */
|
|
173
|
+
function execute(wrangler, database, sql) {
|
|
174
|
+
const r = wrangler(['d1', 'execute', database, '--remote', '--env=', '--json', '--command', sql]);
|
|
175
|
+
if (r.code !== 0)
|
|
176
|
+
throw new Error(r.stderr.trim() || r.stdout.trim() || `wrangler exited ${r.code}`);
|
|
177
|
+
const parsed = JSON.parse(r.stdout.slice(r.stdout.indexOf('[')));
|
|
178
|
+
return parsed.map((statement) => statement.results ?? []);
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* The paging plan for one table. The key is the first of `rowid`, `_rowid_`, `oid` that no
|
|
182
|
+
* declared column shadows — `items_fts_map` declares a column NAMED `rowid`, so it pages on
|
|
183
|
+
* `_rowid_`, which is the same value.
|
|
184
|
+
*/
|
|
185
|
+
export function tablePlan(table, ddl, cols) {
|
|
186
|
+
const columns = cols.map((c) => c.name);
|
|
187
|
+
if (columns.length === 0)
|
|
188
|
+
throw new Error(`D1 reports no columns for \`${table}\``);
|
|
189
|
+
const lower = new Set(columns.map((c) => c.toLowerCase()));
|
|
190
|
+
const withoutRowid = /\bWITHOUT\s+ROWID\b/i.test(ddl ?? '');
|
|
191
|
+
const key = withoutRowid ? null : (['rowid', '_rowid_', 'oid'].find((alias) => !lower.has(alias)) ?? null);
|
|
192
|
+
const pk = cols.filter((c) => c.pk > 0).sort((a, b) => a.pk - b.pk).map((c) => c.name);
|
|
193
|
+
return { table, columns, key, order: pk.length > 0 ? pk : columns };
|
|
194
|
+
}
|
|
195
|
+
/** The SELECT for one page: the key (if any) and every column as its `quote()`d SQL literal. */
|
|
196
|
+
export function pageSql(plan, limit, after) {
|
|
197
|
+
const values = plan.columns.map((c, i) => `quote(${sqlIdent(c)}) AS c${i}`).join(', ');
|
|
198
|
+
if (plan.key) {
|
|
199
|
+
const where = after.key === undefined ? '' : ` WHERE ${plan.key} > ${after.key}`;
|
|
200
|
+
return `SELECT ${plan.key} AS k, ${values} FROM ${sqlIdent(plan.table)}${where} ORDER BY ${plan.key} LIMIT ${limit}`;
|
|
201
|
+
}
|
|
202
|
+
const order = plan.order.map(sqlIdent).join(', ');
|
|
203
|
+
return `SELECT ${values} FROM ${sqlIdent(plan.table)} ORDER BY ${order} LIMIT ${limit} OFFSET ${after.offset ?? 0}`;
|
|
204
|
+
}
|
|
205
|
+
/** Every row of one table as `INSERT INTO …;` lines, read in pages that never block D1. */
|
|
206
|
+
function dumpTable(wrangler, database, plan) {
|
|
207
|
+
const into = `INSERT INTO ${sqlIdent(plan.table)} (${plan.columns.map(sqlIdent).join(', ')}) VALUES (`;
|
|
208
|
+
const lines = [];
|
|
209
|
+
let limit = PAGE_ROWS;
|
|
210
|
+
let after = {};
|
|
211
|
+
for (;;) {
|
|
212
|
+
let page;
|
|
213
|
+
try {
|
|
214
|
+
page = execute(wrangler, database, pageSql(plan, limit, after))[0] ?? [];
|
|
215
|
+
}
|
|
216
|
+
catch (error) {
|
|
217
|
+
if (limit > 1) {
|
|
218
|
+
limit = Math.max(1, Math.floor(limit / 2));
|
|
219
|
+
continue;
|
|
220
|
+
}
|
|
221
|
+
throw new Error(`could not read \`${plan.table}\` from D1: ${error.message}`);
|
|
222
|
+
}
|
|
223
|
+
for (const row of page) {
|
|
224
|
+
const literals = plan.columns.map((_, i) => row[`c${i}`]);
|
|
225
|
+
if (literals.some((v) => typeof v !== 'string')) {
|
|
226
|
+
throw new Error(`D1 returned a non-literal value paging \`${plan.table}\` — refusing a snapshot that may not restore`);
|
|
227
|
+
}
|
|
228
|
+
lines.push(`${into}${literals.join(', ')});`);
|
|
229
|
+
}
|
|
230
|
+
if (page.length < limit)
|
|
231
|
+
break;
|
|
232
|
+
after = plan.key ? { key: Number(page[page.length - 1]?.k) } : { offset: (after.offset ?? 0) + page.length };
|
|
233
|
+
}
|
|
234
|
+
return { text: lines.join('\n'), rows: lines.length };
|
|
235
|
+
}
|
|
146
236
|
/**
|
|
147
|
-
* Pull the live D1 database into a real SQLite file
|
|
237
|
+
* Pull the live D1 database into a real SQLite file, reading it with paged SELECTs that never
|
|
238
|
+
* block production (see the header — this used to be `wrangler d1 export`, which did).
|
|
148
239
|
*
|
|
149
240
|
* 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
|
|
150
|
-
* SQLite file as success, so
|
|
241
|
+
* SQLite file as success, so a pull that produced nothing would sail through the
|
|
151
242
|
* magic-byte check and `integrity_check` and land on the shelf with today's date on it.
|
|
152
243
|
*/
|
|
153
244
|
export async function pullD1ToSqlite(options) {
|
|
154
245
|
const { destination, database, wrangler } = options;
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
throw new Error(`could not list D1's tables: ${
|
|
161
|
-
|
|
162
|
-
const rows = JSON.parse(json)[0]?.results ?? [];
|
|
246
|
+
let rows;
|
|
247
|
+
try {
|
|
248
|
+
rows = (execute(wrangler, database, "select name, sql from sqlite_master where type = 'table'")[0] ?? []);
|
|
249
|
+
}
|
|
250
|
+
catch (error) {
|
|
251
|
+
throw new Error(`could not list D1's tables: ${error.message}`);
|
|
252
|
+
}
|
|
163
253
|
const tables = backupTables(rows);
|
|
164
254
|
if (!tables.includes(options.requiredTable)) {
|
|
165
255
|
throw new Error(`D1 lists no \`${options.requiredTable}\` table (saw: ${rows.map((r) => r.name).join(', ') || 'nothing'})`);
|
|
166
256
|
}
|
|
167
|
-
|
|
168
|
-
|
|
257
|
+
// Every table's columns in ONE call: one statement each, answered in order.
|
|
258
|
+
let columnSets;
|
|
169
259
|
try {
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
260
|
+
columnSets = execute(wrangler, database, tables.map((t) => `SELECT name, pk FROM pragma_table_info(${sqlText(t)});`).join(' '));
|
|
261
|
+
}
|
|
262
|
+
catch (error) {
|
|
263
|
+
throw new Error(`could not read D1's columns: ${error.message}`);
|
|
264
|
+
}
|
|
265
|
+
const dumps = tables.map((table, i) => {
|
|
266
|
+
const cols = (columnSets[i] ?? []);
|
|
267
|
+
const plan = tablePlan(table, rows.find((r) => r.name === table)?.sql ?? null, cols);
|
|
268
|
+
return { table, ...dumpTable(wrangler, database, plan) };
|
|
269
|
+
});
|
|
270
|
+
if (dumps.every((d) => d.rows === 0)) {
|
|
271
|
+
throw new Error('the D1 pull contains no rows — refusing to snapshot an empty database as if it were the library');
|
|
272
|
+
}
|
|
273
|
+
const db = new Database(destination, { create: true });
|
|
274
|
+
try {
|
|
275
|
+
// `exec`, not `run`: each is many statements and `run` would execute only the first,
|
|
276
|
+
// leaving a file with a schema and no rows that looks perfectly valid.
|
|
277
|
+
db.exec(options.schemaSql);
|
|
278
|
+
// 🔴 A table D1 holds that the schema does not declare is created from D1's OWN DDL, never
|
|
279
|
+
// dropped: measured on music's first post-cutover pull (2026-09-23), `art_lookups` — a
|
|
280
|
+
// cache its art job creates where it first needs it — made the whole pull refuse with
|
|
281
|
+
// `no such table`, because its rows were exported and its table was not in the schema.
|
|
282
|
+
const have = new Set(db.query("SELECT name FROM sqlite_master WHERE type = 'table'").all().map((r) => r.name));
|
|
283
|
+
for (const row of rows) {
|
|
284
|
+
if (!tables.includes(row.name) || have.has(row.name) || !row.sql)
|
|
285
|
+
continue;
|
|
286
|
+
db.exec(row.sql.replace(/^\s*CREATE\s+TABLE\s+(?!IF\s+NOT\s+EXISTS)/i, 'CREATE TABLE IF NOT EXISTS '));
|
|
196
287
|
}
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
for (const { text } of dumps)
|
|
214
|
-
if (text.trim())
|
|
215
|
-
db.exec(text);
|
|
216
|
-
if (options.afterLoadSql)
|
|
217
|
-
db.exec(options.afterLoadSql);
|
|
218
|
-
// 🔴 Every table holds as many rows as its dump has INSERTs, or this is not a backup: the
|
|
219
|
-
// check that makes a silently truncated load — the NUL padding, or whatever does it next —
|
|
220
|
-
// a REFUSAL. A row count only; nothing here reads what the rows hold.
|
|
221
|
-
const short = [];
|
|
222
|
-
for (const { table, text } of dumps) {
|
|
223
|
-
if (table === 'sqlite_sequence')
|
|
224
|
-
continue;
|
|
225
|
-
const expected = insertLines(text);
|
|
226
|
-
const loaded = db.query(`SELECT COUNT(*) AS n FROM "${table.replaceAll('"', '""')}"`).get().n;
|
|
227
|
-
if (loaded < expected)
|
|
228
|
-
short.push(`${table}: ${loaded} of ${expected}`);
|
|
229
|
-
}
|
|
230
|
-
if (short.length > 0) {
|
|
231
|
-
db.close();
|
|
232
|
-
rmSync(destination, { force: true });
|
|
233
|
-
throw new Error(`the D1 pull did not load every exported row — ${short.join('; ')}. Refusing the snapshot.`);
|
|
234
|
-
}
|
|
288
|
+
// One table at a time, so a fault is located rather than smeared over the whole load.
|
|
289
|
+
for (const { text } of dumps)
|
|
290
|
+
if (text)
|
|
291
|
+
db.exec(text);
|
|
292
|
+
if (options.afterLoadSql)
|
|
293
|
+
db.exec(options.afterLoadSql);
|
|
294
|
+
// 🔴 Every table holds as many rows as were read from D1, or this is not a backup: the check
|
|
295
|
+
// that makes a silently truncated load (task 2128's NUL padding, or whatever does it next)
|
|
296
|
+
// a REFUSAL. A row count only; nothing here reads what the rows hold.
|
|
297
|
+
const short = [];
|
|
298
|
+
for (const { table, rows: expected } of dumps) {
|
|
299
|
+
if (table === 'sqlite_sequence')
|
|
300
|
+
continue;
|
|
301
|
+
const loaded = db.query(`SELECT COUNT(*) AS n FROM ${sqlIdent(table)}`).get().n;
|
|
302
|
+
if (loaded < expected)
|
|
303
|
+
short.push(`${table}: ${loaded} of ${expected}`);
|
|
235
304
|
}
|
|
236
|
-
|
|
305
|
+
if (short.length > 0) {
|
|
237
306
|
db.close();
|
|
307
|
+
rmSync(destination, { force: true });
|
|
308
|
+
throw new Error(`the D1 pull did not load every row it read — ${short.join('; ')}. Refusing the snapshot.`);
|
|
238
309
|
}
|
|
239
310
|
}
|
|
240
311
|
finally {
|
|
241
|
-
|
|
242
|
-
// would ever delete — eighteen were found beside vault's snapshot (task 2128).
|
|
243
|
-
for (const table of tables)
|
|
244
|
-
rmSync(`${destination}.${table}.sql`, { force: true });
|
|
312
|
+
db.close();
|
|
245
313
|
}
|
|
246
314
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cursedbelt-server",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.32.0",
|
|
4
4
|
"license": "ISC",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"description": "The app-facing Bun/Hono server tier of the cursedbelt split — storage, sharing, activity, guard, sync. React-free; cursedbelt-core below it.",
|
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
type Wrangler,
|
|
13
13
|
wranglerArgv,
|
|
14
14
|
} from './localBackup.js';
|
|
15
|
+
import { PAGE_ROWS, pageSql, tablePlan } from './pullD1.js';
|
|
15
16
|
|
|
16
17
|
const dirs: string[] = [];
|
|
17
18
|
const scratch = (): string => {
|
|
@@ -82,7 +83,7 @@ describe('servedRuntime', () => {
|
|
|
82
83
|
});
|
|
83
84
|
});
|
|
84
85
|
|
|
85
|
-
describe('pulling D1 down — table by table,
|
|
86
|
+
describe('pulling D1 down — paged SELECTs, table by table, never an export', () => {
|
|
86
87
|
const MASTER = [
|
|
87
88
|
{ name: '_cf_KV', sql: 'CREATE TABLE _cf_KV (key TEXT PRIMARY KEY)' },
|
|
88
89
|
{ name: 'items', sql: 'CREATE TABLE items ( id TEXT PRIMARY KEY )' },
|
|
@@ -108,21 +109,32 @@ describe('pulling D1 down — table by table, because a whole-database export re
|
|
|
108
109
|
expect(backupTables(MASTER)).toEqual(['items', 'items_fts_map', 'meta', 'sqlite_sequence']);
|
|
109
110
|
});
|
|
110
111
|
|
|
111
|
-
|
|
112
|
-
|
|
112
|
+
/**
|
|
113
|
+
* A fake `wrangler` backed by a REAL sqlite: `d1 execute` runs the SQL against `source` and
|
|
114
|
+
* answers the way wrangler's `--json` does (noise, then one result set per statement), and
|
|
115
|
+
* `d1 export` — which must never be called any more — is recorded so a test can say so.
|
|
116
|
+
*/
|
|
117
|
+
const d1 = (source: Database, opts: { failWhen?: (sql: string) => boolean; master?: typeof MASTER } = {}) => {
|
|
113
118
|
const seen: string[][] = [];
|
|
114
119
|
const run: Wrangler = (args) => {
|
|
115
120
|
seen.push(args);
|
|
116
|
-
if (args[1]
|
|
117
|
-
const
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
121
|
+
if (args[1] !== 'execute') return { code: 0, stdout: 'Downloaded successfully', stderr: '' };
|
|
122
|
+
const sql = args[args.indexOf('--command') + 1] as string;
|
|
123
|
+
if (opts.failWhen?.(sql)) return { code: 1, stdout: '', stderr: '✘ D1 said no' };
|
|
124
|
+
if (sql.startsWith('select name, sql from sqlite_master')) {
|
|
125
|
+
return { code: 0, stdout: `noise\n${JSON.stringify([{ results: opts.master ?? MASTER }])}`, stderr: '' };
|
|
126
|
+
}
|
|
127
|
+
const statements = sql.split(/;\s*/).filter(Boolean);
|
|
128
|
+
const results = statements.map((statement) => ({ results: source.query(statement).all() }));
|
|
129
|
+
return { code: 0, stdout: `noise\n${JSON.stringify(results)}`, stderr: '' };
|
|
124
130
|
};
|
|
125
|
-
return { run,
|
|
131
|
+
return { run, seen, exports: () => seen.filter((a) => a[1] === 'export').length };
|
|
132
|
+
};
|
|
133
|
+
const live = (setup = ''): Database => {
|
|
134
|
+
const db = new Database(':memory:');
|
|
135
|
+
db.exec(SCHEMA);
|
|
136
|
+
if (setup) db.exec(setup);
|
|
137
|
+
return db;
|
|
126
138
|
};
|
|
127
139
|
const pull = (dir: string, run: Wrangler) =>
|
|
128
140
|
pullD1ToSqlite({
|
|
@@ -133,77 +145,99 @@ describe('pulling D1 down — table by table, because a whole-database export re
|
|
|
133
145
|
afterLoadSql: REBUILD,
|
|
134
146
|
wrangler: run,
|
|
135
147
|
});
|
|
148
|
+
const opened = (dir: string) => new Database(join(dir, 'pulled.sqlite'), { readonly: true });
|
|
136
149
|
|
|
137
150
|
test('a pull is a real, searchable sqlite file built from the schema and the rows, of the named database', async () => {
|
|
138
151
|
const dir = scratch();
|
|
139
|
-
const { run,
|
|
140
|
-
|
|
141
|
-
? "INSERT INTO items (id, name) VALUES ('itm_1', 'Travels 5388.jpg');"
|
|
142
|
-
: table === 'items_fts_map'
|
|
143
|
-
? "INSERT INTO items_fts_map (rowid, item_id, text) VALUES (1, 'itm_1', 'travels 5388');"
|
|
144
|
-
: '',
|
|
152
|
+
const { run, seen } = d1(
|
|
153
|
+
live("INSERT INTO items VALUES ('itm_1', 'Travels 5388.jpg'); INSERT INTO items_fts_map (item_id, text) VALUES ('itm_1', 'travels 5388');"),
|
|
145
154
|
);
|
|
146
155
|
await pull(dir, run);
|
|
147
|
-
expect(exported).toEqual(['items', 'items_fts_map', 'meta', 'sqlite_sequence']);
|
|
148
156
|
expect(seen.every((args) => args[2] === 'collections')).toBe(true);
|
|
149
|
-
const db =
|
|
157
|
+
const db = opened(dir);
|
|
150
158
|
const hits = db.query("select count(*) as n from items_fts where items_fts match '5388'").get() as { n: number };
|
|
159
|
+
expect((db.query('select seq from sqlite_sequence').get() as { seq: number }).seq).toBe(1);
|
|
151
160
|
db.close();
|
|
152
161
|
expect(hits.n).toBe(1);
|
|
153
162
|
});
|
|
154
163
|
|
|
155
|
-
test('🔴
|
|
164
|
+
test('🔴 it NEVER runs `wrangler d1 export` — an export blocks production D1 (family, 2026-09-25: Error 1101)', async () => {
|
|
156
165
|
const dir = scratch();
|
|
157
|
-
const
|
|
158
|
-
const run: Wrangler = (args) => {
|
|
159
|
-
if (args[1] === 'execute') return { code: 0, stdout: JSON.stringify([{ results: master }]), stderr: '' };
|
|
160
|
-
const table = args[args.indexOf('--table') + 1] as string;
|
|
161
|
-
const out = args[args.indexOf('--output') + 1] as string;
|
|
162
|
-
writeFileSync(
|
|
163
|
-
out,
|
|
164
|
-
table === 'items' ? "INSERT INTO items (id) VALUES ('i1');" : table === 'art_lookups' ? "INSERT INTO art_lookups VALUES ('k', 'art/1');" : '',
|
|
165
|
-
);
|
|
166
|
-
return { code: 0, stdout: '', stderr: '' };
|
|
167
|
-
};
|
|
166
|
+
const { run, seen, exports } = d1(live("INSERT INTO items VALUES ('i1', 'a');"));
|
|
168
167
|
await pull(dir, run);
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
168
|
+
expect(exports()).toBe(0);
|
|
169
|
+
expect(seen.every((args) => args[1] === 'execute' && args.includes('--remote'))).toBe(true);
|
|
170
|
+
// Every statement it sends is a read.
|
|
171
|
+
const sent = seen.map((args) => args[args.indexOf('--command') + 1] as string);
|
|
172
|
+
expect(sent.every((sql) => /^\s*select/i.test(sql))).toBe(true);
|
|
173
173
|
});
|
|
174
174
|
|
|
175
|
-
test('🔴
|
|
175
|
+
test('🔴 types survive exactly — REAL 5.0 stays REAL, INTEGER 5 stays INTEGER, a BLOB stays its bytes, quotes and NULL too', async () => {
|
|
176
176
|
const dir = scratch();
|
|
177
|
-
const
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
177
|
+
const source = live(
|
|
178
|
+
"CREATE TABLE typed (a, b, c, d, e); INSERT INTO typed VALUES (5.0, 5, x'00ff10', 'it''s\nlines', NULL); INSERT INTO items VALUES ('i1', 'x');",
|
|
179
|
+
);
|
|
180
|
+
const master = [...MASTER, { name: 'typed', sql: 'CREATE TABLE typed (a, b, c, d, e)' }];
|
|
181
|
+
await pull(dir, d1(source, { master }).run);
|
|
182
|
+
const db = opened(dir);
|
|
183
|
+
const row = db.query('select typeof(a) ta, typeof(b) tb, hex(c) hc, d, typeof(e) te, a + 0.1 s from typed').get();
|
|
184
|
+
db.close();
|
|
185
|
+
expect(row).toEqual({ ta: 'real', tb: 'integer', hc: '00FF10', d: "it's\nlines", te: 'null', s: 5.1 });
|
|
181
186
|
});
|
|
182
187
|
|
|
183
|
-
test('
|
|
188
|
+
test('a table larger than one page is read in full on the rowid keyset — and a table with a column NAMED rowid pages on _rowid_', async () => {
|
|
184
189
|
const dir = scratch();
|
|
185
|
-
|
|
186
|
-
const
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
// `meta` "succeeds" once without writing — measured on collections, 2026-09-23.
|
|
191
|
-
if (table === 'meta' && flaky-- > 0) return { code: 0, stdout: 'Downloaded successfully', stderr: '' };
|
|
192
|
-
writeFileSync(out, table === 'items' ? "INSERT INTO items (id) VALUES ('i1');" : '');
|
|
193
|
-
return { code: 0, stdout: '', stderr: '' };
|
|
194
|
-
};
|
|
190
|
+
const n = PAGE_ROWS * 2 + 7;
|
|
191
|
+
const source = live(
|
|
192
|
+
`WITH RECURSIVE s(i) AS (SELECT 1 UNION ALL SELECT i + 1 FROM s WHERE i < ${n}) INSERT INTO items_fts_map (item_id, text) SELECT 'i' || i, 't' FROM s; INSERT INTO items VALUES ('i1', 'x');`,
|
|
193
|
+
);
|
|
194
|
+
const { run, seen } = d1(source);
|
|
195
195
|
await pull(dir, run);
|
|
196
|
-
const
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
196
|
+
const db = opened(dir);
|
|
197
|
+
expect((db.query('select count(*) n from items_fts_map').get() as { n: number }).n).toBe(n);
|
|
198
|
+
db.close();
|
|
199
|
+
const mapPages = seen.map((a) => a[a.indexOf('--command') + 1] as string).filter((sql) => sql.includes('FROM "items_fts_map"'));
|
|
200
|
+
expect(mapPages).toHaveLength(3);
|
|
201
|
+
expect(mapPages[1]).toContain('WHERE _rowid_ >');
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
test('a WITHOUT ROWID table pages by its primary key and OFFSET', () => {
|
|
205
|
+
const plan = tablePlan('kv', 'CREATE TABLE kv (k TEXT PRIMARY KEY, v) WITHOUT ROWID', [
|
|
206
|
+
{ name: 'k', pk: 1 },
|
|
207
|
+
{ name: 'v', pk: 0 },
|
|
208
|
+
]);
|
|
209
|
+
expect(plan.key).toBeNull();
|
|
210
|
+
expect(pageSql(plan, 10, { offset: 20 })).toBe('SELECT quote("k") AS c0, quote("v") AS c1 FROM "kv" ORDER BY "k" LIMIT 10 OFFSET 20');
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
test('a page D1 refuses is halved and asked again; a table that cannot be read at one row THROWS', async () => {
|
|
214
|
+
const dir = scratch();
|
|
215
|
+
const source = live("INSERT INTO items VALUES ('i1', 'a'); INSERT INTO meta VALUES ('k', 'v');");
|
|
216
|
+
await pull(dir, d1(source, { failWhen: (sql) => sql.includes('FROM "meta"') && sql.includes(`LIMIT ${PAGE_ROWS}`) }).run);
|
|
217
|
+
const db = opened(dir);
|
|
218
|
+
expect((db.query('select value from meta').get() as { value: string }).value).toBe('v');
|
|
219
|
+
db.close();
|
|
220
|
+
await expect(pull(scratch(), d1(source, { failWhen: (sql) => sql.includes('FROM "meta"') }).run)).rejects.toThrow(
|
|
221
|
+
/could not read `meta` from D1/,
|
|
222
|
+
);
|
|
201
223
|
});
|
|
202
224
|
|
|
203
|
-
test('🔴
|
|
225
|
+
test('🔴 a table D1 holds that the schema does not declare is created from D1\'s own DDL', async () => {
|
|
226
|
+
const dir = scratch();
|
|
227
|
+
const ddl = 'CREATE TABLE art_lookups (key TEXT PRIMARY KEY, art_key TEXT)';
|
|
228
|
+
const source = live(`${ddl}; INSERT INTO art_lookups VALUES ('k', 'art/1'); INSERT INTO items VALUES ('i1', 'x');`);
|
|
229
|
+
await pull(dir, d1(source, { master: [...MASTER, { name: 'art_lookups', sql: ddl }] }).run);
|
|
230
|
+
const db = opened(dir);
|
|
231
|
+
const row = db.query('select art_key from art_lookups').get() as { art_key: string } | null;
|
|
232
|
+
db.close();
|
|
233
|
+
expect(row?.art_key).toBe('art/1');
|
|
234
|
+
});
|
|
235
|
+
|
|
236
|
+
test('🔴 a database without the required table is refused as the wrong one', async () => {
|
|
204
237
|
const dir = scratch();
|
|
205
|
-
|
|
206
|
-
|
|
238
|
+
await expect(
|
|
239
|
+
pullD1ToSqlite({ destination: join(dir, 'x.sqlite'), database: 'x', schemaSql: SCHEMA, requiredTable: 'tracks', wrangler: d1(live()).run }),
|
|
240
|
+
).rejects.toThrow(/lists no `tracks` table/);
|
|
207
241
|
});
|
|
208
242
|
|
|
209
243
|
test("🔴 bun:sqlite's exec stops at a NUL with no error — the defect, pinned (task 2128)", () => {
|
|
@@ -214,38 +248,27 @@ describe('pulling D1 down — table by table, because a whole-database export re
|
|
|
214
248
|
db.close();
|
|
215
249
|
});
|
|
216
250
|
|
|
217
|
-
test('🔴 a
|
|
218
|
-
const dir = scratch();
|
|
219
|
-
const { run } = fake((table) =>
|
|
220
|
-
table === 'items'
|
|
221
|
-
? `INSERT INTO items (id, name) VALUES ('itm_1', 'a');\n${'\0'.repeat(988)}`
|
|
222
|
-
: table === 'meta'
|
|
223
|
-
? "INSERT INTO meta (key, value) VALUES ('k', 'v');\nINSERT INTO meta (key, value) VALUES ('k2', 'v2');"
|
|
224
|
-
: '',
|
|
225
|
-
);
|
|
226
|
-
await pull(dir, run);
|
|
227
|
-
const db = new Database(join(dir, 'pulled.sqlite'), { readonly: true });
|
|
228
|
-
expect((db.query('SELECT COUNT(*) AS n FROM meta').get() as { n: number }).n).toBe(2);
|
|
229
|
-
db.close();
|
|
230
|
-
});
|
|
231
|
-
|
|
232
|
-
test('🔴 a load SHORT of its dump — whatever truncated it — is refused, and leaves no snapshot', async () => {
|
|
251
|
+
test('🔴 a load SHORT of what was read — whatever truncated it — is refused, and leaves no snapshot', async () => {
|
|
233
252
|
const dir = scratch();
|
|
234
|
-
//
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
253
|
+
// Whatever truncates a load — task 2128's NUL padding, or the next thing — the table ends
|
|
254
|
+
// up holding fewer rows than were read. A row deleted after the load is exactly that shape.
|
|
255
|
+
const source = live("INSERT INTO items VALUES ('a', 'x'); INSERT INTO items VALUES ('b', 'y');");
|
|
256
|
+
const truncating = pullD1ToSqlite({
|
|
257
|
+
destination: join(dir, 'pulled.sqlite'),
|
|
258
|
+
database: 'collections',
|
|
259
|
+
schemaSql: SCHEMA,
|
|
260
|
+
requiredTable: 'items',
|
|
261
|
+
afterLoadSql: "DELETE FROM items WHERE id = 'b'",
|
|
262
|
+
wrangler: d1(source).run,
|
|
263
|
+
});
|
|
264
|
+
await expect(truncating).rejects.toThrow(/items: 1 of 2/);
|
|
239
265
|
expect(existsSync(join(dir, 'pulled.sqlite'))).toBe(false);
|
|
240
266
|
});
|
|
241
267
|
|
|
242
|
-
test('
|
|
268
|
+
test('no dump file is left beside the snapshot', async () => {
|
|
243
269
|
const dir = scratch();
|
|
244
|
-
|
|
245
|
-
await pull(dir, run);
|
|
270
|
+
await pull(dir, d1(live("INSERT INTO items VALUES ('a', 'x');")).run);
|
|
246
271
|
expect(readdirSync(dir).filter((f) => f.endsWith('.sql'))).toEqual([]);
|
|
247
|
-
const bad = fake((table) => (table === 'meta' ? null : "INSERT INTO items (id) VALUES ('x');"));
|
|
248
|
-
await expect(pull(scratch(), bad.run)).rejects.toThrow();
|
|
249
272
|
});
|
|
250
273
|
|
|
251
274
|
test('stripTrailingNuls strips only TRAILING NULs; wranglerArgv is this bun, never a bare bunx', () => {
|
|
@@ -258,9 +281,7 @@ describe('pulling D1 down — table by table, because a whole-database export re
|
|
|
258
281
|
expect(wranglerArgv('/x/bun')).toEqual(['/x/bun', 'x', 'wrangler']);
|
|
259
282
|
});
|
|
260
283
|
|
|
261
|
-
test('🔴
|
|
262
|
-
|
|
263
|
-
const { run } = fake(() => '');
|
|
264
|
-
await expect(pull(dir, run)).rejects.toThrow(/no rows/);
|
|
284
|
+
test('🔴 a pull with no rows at all THROWS — never an empty library dated today', async () => {
|
|
285
|
+
await expect(pull(scratch(), d1(live()).run)).rejects.toThrow(/no rows/);
|
|
265
286
|
});
|
|
266
287
|
});
|
package/src/server/d1/pullD1.ts
CHANGED
|
@@ -18,13 +18,30 @@
|
|
|
18
18
|
* the cutover (a GREEN backup of nothing current — the worst failure a backup has), guessing
|
|
19
19
|
* D1 snapshots a stale copy while the Mac is live.
|
|
20
20
|
*
|
|
21
|
-
* ## 🔴 Why
|
|
21
|
+
* ## 🔴 Why paged `SELECT`s, and never `wrangler d1 export`
|
|
22
22
|
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
23
|
+
* **A D1 export blocks the database from serving anything else for as long as it runs**
|
|
24
|
+
* (Cloudflare's own docs). Measured on family 2026-09-25 00:32–00:33 UTC: a manual `bun run
|
|
25
|
+
* backup` — then a per-table `wrangler d1 export` — held production for ~28 s at a time, and
|
|
26
|
+
* 14 requests to the family Worker sat waiting and threw Error 1101 (wall p50 28 s, CPU p50
|
|
27
|
+
* 1.6 ms). Two earlier throws at 13:56/13:58 had the same shape. So every nightly backup of
|
|
28
|
+
* every app on this seam was a scheduled outage.
|
|
29
|
+
*
|
|
30
|
+
* Since 4.32.0 the rows are read with ordinary `wrangler d1 execute --remote` SELECTs, a page
|
|
31
|
+
* of {@link PAGE_ROWS} rows at a time on a rowid keyset, which D1 serves between other
|
|
32
|
+
* requests like any read. Each value comes back as SQLite's own `quote()` of it — an SQL
|
|
33
|
+
* literal, as TEXT — so the REAL `5.0` stays `5.0` and the INTEGER `5` stays `5` (the reason
|
|
34
|
+
* this used `export` and not the REST API's JSON: `./http.ts` has the measurement), a BLOB is
|
|
35
|
+
* `X'…'`, and `quote()` round-trips a double exactly. The dump is then the same
|
|
36
|
+
* `INSERT INTO …` text an export wrote, so everything after it is unchanged.
|
|
37
|
+
*
|
|
38
|
+
* Still table by table: a whole-database export refused fts5 (*"cannot export databases with
|
|
39
|
+
* Virtual Tables"*), and a virtual table's shadow tables are an index, rebuilt, never copied.
|
|
40
|
+
*
|
|
41
|
+
* 🔴 The trade: an export was one point in time; pages are not. A row written to one table
|
|
42
|
+
* while another is being paged may be in the snapshot without its partner. For a nightly
|
|
43
|
+
* backup that is the right trade against taking production down, and the row-count check
|
|
44
|
+
* below still refuses a load short of what was read.
|
|
28
45
|
*/
|
|
29
46
|
|
|
30
47
|
import { Database } from 'bun:sqlite';
|
|
@@ -97,8 +114,13 @@ export async function servedRuntime(
|
|
|
97
114
|
return null;
|
|
98
115
|
}
|
|
99
116
|
|
|
100
|
-
/**
|
|
101
|
-
|
|
117
|
+
/**
|
|
118
|
+
* Rows per paged `SELECT`. The fleet's largest table held 290 rows when this was set (collections'
|
|
119
|
+
* `items`, 2026-09-25) and one `wrangler` call costs ~1 s, so a page this size is one call per
|
|
120
|
+
* table for every app today. A page D1 refuses — too large a response — is halved and asked
|
|
121
|
+
* again, down to one row, before it is a refusal.
|
|
122
|
+
*/
|
|
123
|
+
export const PAGE_ROWS = 500;
|
|
102
124
|
|
|
103
125
|
/** One `wrangler` invocation, as {@link pullD1ToSqlite} needs it. Injected so a spec runs none. */
|
|
104
126
|
export type Wrangler = (args: string[]) => { code: number; stdout: string; stderr: string };
|
|
@@ -125,6 +147,8 @@ export function spawnWrangler(options: { cwd: string; env?: Record<string, strin
|
|
|
125
147
|
|
|
126
148
|
/**
|
|
127
149
|
* 🔴 Trailing NUL bytes off a `wrangler d1 export` file, in place. Returns how many were removed.
|
|
150
|
+
* {@link pullD1ToSqlite} no longer exports (see the header), so nothing here calls this since
|
|
151
|
+
* 4.32.0; it stays exported for anyone still reading an export file by hand.
|
|
128
152
|
*
|
|
129
153
|
* Measured 2026-09-23 on vault's D1 (wrangler 4.136.3): `wrangler d1 export --table … --output`
|
|
130
154
|
* INTERMITTENTLY ends the file in hundreds of NUL bytes (`entries` +988, `oplog` +1,369, `sync_ops`
|
|
@@ -180,22 +204,98 @@ export interface PullD1Options {
|
|
|
180
204
|
wrangler: Wrangler;
|
|
181
205
|
}
|
|
182
206
|
|
|
207
|
+
const sqlIdent = (name: string): string => `"${name.replaceAll('"', '""')}"`;
|
|
208
|
+
const sqlText = (text: string): string => `'${text.replaceAll("'", "''")}'`;
|
|
209
|
+
|
|
210
|
+
/** One `wrangler d1 execute --remote --json` — every statement's result rows, in order. */
|
|
211
|
+
function execute(wrangler: Wrangler, database: string, sql: string): Array<Array<Record<string, unknown>>> {
|
|
212
|
+
const r = wrangler(['d1', 'execute', database, '--remote', '--env=', '--json', '--command', sql]);
|
|
213
|
+
if (r.code !== 0) throw new Error(r.stderr.trim() || r.stdout.trim() || `wrangler exited ${r.code}`);
|
|
214
|
+
const parsed = JSON.parse(r.stdout.slice(r.stdout.indexOf('['))) as Array<{ results?: Array<Record<string, unknown>> }>;
|
|
215
|
+
return parsed.map((statement) => statement.results ?? []);
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** How a table is walked in pages: by a rowid keyset, or — a `WITHOUT ROWID` table — by offset. */
|
|
219
|
+
export interface TablePlan {
|
|
220
|
+
table: string;
|
|
221
|
+
columns: string[];
|
|
222
|
+
/** The rowid alias to page on, or `null` for a table with none (then `order` + OFFSET). */
|
|
223
|
+
key: string | null;
|
|
224
|
+
order: string[];
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* The paging plan for one table. The key is the first of `rowid`, `_rowid_`, `oid` that no
|
|
229
|
+
* declared column shadows — `items_fts_map` declares a column NAMED `rowid`, so it pages on
|
|
230
|
+
* `_rowid_`, which is the same value.
|
|
231
|
+
*/
|
|
232
|
+
export function tablePlan(table: string, ddl: string | null, cols: ReadonlyArray<{ name: string; pk: number }>): TablePlan {
|
|
233
|
+
const columns = cols.map((c) => c.name);
|
|
234
|
+
if (columns.length === 0) throw new Error(`D1 reports no columns for \`${table}\``);
|
|
235
|
+
const lower = new Set(columns.map((c) => c.toLowerCase()));
|
|
236
|
+
const withoutRowid = /\bWITHOUT\s+ROWID\b/i.test(ddl ?? '');
|
|
237
|
+
const key = withoutRowid ? null : (['rowid', '_rowid_', 'oid'].find((alias) => !lower.has(alias)) ?? null);
|
|
238
|
+
const pk = cols.filter((c) => c.pk > 0).sort((a, b) => a.pk - b.pk).map((c) => c.name);
|
|
239
|
+
return { table, columns, key, order: pk.length > 0 ? pk : columns };
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** The SELECT for one page: the key (if any) and every column as its `quote()`d SQL literal. */
|
|
243
|
+
export function pageSql(plan: TablePlan, limit: number, after: { key?: number; offset?: number }): string {
|
|
244
|
+
const values = plan.columns.map((c, i) => `quote(${sqlIdent(c)}) AS c${i}`).join(', ');
|
|
245
|
+
if (plan.key) {
|
|
246
|
+
const where = after.key === undefined ? '' : ` WHERE ${plan.key} > ${after.key}`;
|
|
247
|
+
return `SELECT ${plan.key} AS k, ${values} FROM ${sqlIdent(plan.table)}${where} ORDER BY ${plan.key} LIMIT ${limit}`;
|
|
248
|
+
}
|
|
249
|
+
const order = plan.order.map(sqlIdent).join(', ');
|
|
250
|
+
return `SELECT ${values} FROM ${sqlIdent(plan.table)} ORDER BY ${order} LIMIT ${limit} OFFSET ${after.offset ?? 0}`;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/** Every row of one table as `INSERT INTO …;` lines, read in pages that never block D1. */
|
|
254
|
+
function dumpTable(wrangler: Wrangler, database: string, plan: TablePlan): { text: string; rows: number } {
|
|
255
|
+
const into = `INSERT INTO ${sqlIdent(plan.table)} (${plan.columns.map(sqlIdent).join(', ')}) VALUES (`;
|
|
256
|
+
const lines: string[] = [];
|
|
257
|
+
let limit = PAGE_ROWS;
|
|
258
|
+
let after: { key?: number; offset?: number } = {};
|
|
259
|
+
for (;;) {
|
|
260
|
+
let page: Array<Record<string, unknown>>;
|
|
261
|
+
try {
|
|
262
|
+
page = execute(wrangler, database, pageSql(plan, limit, after))[0] ?? [];
|
|
263
|
+
} catch (error) {
|
|
264
|
+
if (limit > 1) {
|
|
265
|
+
limit = Math.max(1, Math.floor(limit / 2));
|
|
266
|
+
continue;
|
|
267
|
+
}
|
|
268
|
+
throw new Error(`could not read \`${plan.table}\` from D1: ${(error as Error).message}`);
|
|
269
|
+
}
|
|
270
|
+
for (const row of page) {
|
|
271
|
+
const literals = plan.columns.map((_, i) => row[`c${i}`]);
|
|
272
|
+
if (literals.some((v) => typeof v !== 'string')) {
|
|
273
|
+
throw new Error(`D1 returned a non-literal value paging \`${plan.table}\` — refusing a snapshot that may not restore`);
|
|
274
|
+
}
|
|
275
|
+
lines.push(`${into}${literals.join(', ')});`);
|
|
276
|
+
}
|
|
277
|
+
if (page.length < limit) break;
|
|
278
|
+
after = plan.key ? { key: Number(page[page.length - 1]?.k) } : { offset: (after.offset ?? 0) + page.length };
|
|
279
|
+
}
|
|
280
|
+
return { text: lines.join('\n'), rows: lines.length };
|
|
281
|
+
}
|
|
282
|
+
|
|
183
283
|
/**
|
|
184
|
-
* Pull the live D1 database into a real SQLite file
|
|
284
|
+
* Pull the live D1 database into a real SQLite file, reading it with paged SELECTs that never
|
|
285
|
+
* block production (see the header — this used to be `wrangler d1 export`, which did).
|
|
185
286
|
*
|
|
186
287
|
* 🔴 It THROWS rather than returning an empty database. Every layer beneath treats a readable
|
|
187
|
-
* SQLite file as success, so
|
|
288
|
+
* SQLite file as success, so a pull that produced nothing would sail through the
|
|
188
289
|
* magic-byte check and `integrity_check` and land on the shelf with today's date on it.
|
|
189
290
|
*/
|
|
190
291
|
export async function pullD1ToSqlite(options: PullD1Options): Promise<void> {
|
|
191
292
|
const { destination, database, wrangler } = options;
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
const rows = (JSON.parse(json) as Array<{ results: Array<{ name: string; sql: string | null }> }>)[0]?.results ?? [];
|
|
293
|
+
let rows: Array<{ name: string; sql: string | null }>;
|
|
294
|
+
try {
|
|
295
|
+
rows = (execute(wrangler, database, "select name, sql from sqlite_master where type = 'table'")[0] ?? []) as typeof rows;
|
|
296
|
+
} catch (error) {
|
|
297
|
+
throw new Error(`could not list D1's tables: ${(error as Error).message}`);
|
|
298
|
+
}
|
|
199
299
|
const tables = backupTables(rows);
|
|
200
300
|
if (!tables.includes(options.requiredTable)) {
|
|
201
301
|
throw new Error(
|
|
@@ -203,78 +303,56 @@ export async function pullD1ToSqlite(options: PullD1Options): Promise<void> {
|
|
|
203
303
|
);
|
|
204
304
|
}
|
|
205
305
|
|
|
206
|
-
|
|
207
|
-
|
|
306
|
+
// Every table's columns in ONE call: one statement each, answered in order.
|
|
307
|
+
let columnSets: Array<Array<Record<string, unknown>>>;
|
|
208
308
|
try {
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
309
|
+
columnSets = execute(wrangler, database, tables.map((t) => `SELECT name, pk FROM pragma_table_info(${sqlText(t)});`).join(' '));
|
|
310
|
+
} catch (error) {
|
|
311
|
+
throw new Error(`could not read D1's columns: ${(error as Error).message}`);
|
|
312
|
+
}
|
|
313
|
+
const dumps = tables.map((table, i) => {
|
|
314
|
+
const cols = (columnSets[i] ?? []) as Array<{ name: string; pk: number }>;
|
|
315
|
+
const plan = tablePlan(table, rows.find((r) => r.name === table)?.sql ?? null, cols);
|
|
316
|
+
return { table, ...dumpTable(wrangler, database, plan) };
|
|
317
|
+
});
|
|
318
|
+
if (dumps.every((d) => d.rows === 0)) {
|
|
319
|
+
throw new Error('the D1 pull contains no rows — refusing to snapshot an empty database as if it were the library');
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
const db = new Database(destination, { create: true });
|
|
323
|
+
try {
|
|
324
|
+
// `exec`, not `run`: each is many statements and `run` would execute only the first,
|
|
325
|
+
// leaving a file with a schema and no rows that looks perfectly valid.
|
|
326
|
+
db.exec(options.schemaSql);
|
|
327
|
+
// 🔴 A table D1 holds that the schema does not declare is created from D1's OWN DDL, never
|
|
328
|
+
// dropped: measured on music's first post-cutover pull (2026-09-23), `art_lookups` — a
|
|
329
|
+
// cache its art job creates where it first needs it — made the whole pull refuse with
|
|
330
|
+
// `no such table`, because its rows were exported and its table was not in the schema.
|
|
331
|
+
const have = new Set(
|
|
332
|
+
(db.query("SELECT name FROM sqlite_master WHERE type = 'table'").all() as { name: string }[]).map((r) => r.name),
|
|
333
|
+
);
|
|
334
|
+
for (const row of rows) {
|
|
335
|
+
if (!tables.includes(row.name) || have.has(row.name) || !row.sql) continue;
|
|
336
|
+
db.exec(row.sql.replace(/^\s*CREATE\s+TABLE\s+(?!IF\s+NOT\s+EXISTS)/i, 'CREATE TABLE IF NOT EXISTS '));
|
|
233
337
|
}
|
|
234
|
-
|
|
235
|
-
|
|
338
|
+
// One table at a time, so a fault is located rather than smeared over the whole load.
|
|
339
|
+
for (const { text } of dumps) if (text) db.exec(text);
|
|
340
|
+
if (options.afterLoadSql) db.exec(options.afterLoadSql);
|
|
341
|
+
// 🔴 Every table holds as many rows as were read from D1, or this is not a backup: the check
|
|
342
|
+
// that makes a silently truncated load (task 2128's NUL padding, or whatever does it next)
|
|
343
|
+
// a REFUSAL. A row count only; nothing here reads what the rows hold.
|
|
344
|
+
const short: string[] = [];
|
|
345
|
+
for (const { table, rows: expected } of dumps) {
|
|
346
|
+
if (table === 'sqlite_sequence') continue;
|
|
347
|
+
const loaded = (db.query(`SELECT COUNT(*) AS n FROM ${sqlIdent(table)}`).get() as { n: number }).n;
|
|
348
|
+
if (loaded < expected) short.push(`${table}: ${loaded} of ${expected}`);
|
|
236
349
|
}
|
|
237
|
-
|
|
238
|
-
const db = new Database(destination, { create: true });
|
|
239
|
-
try {
|
|
240
|
-
// `exec`, not `run`: each is many statements and `run` would execute only the first,
|
|
241
|
-
// leaving a file with a schema and no rows that looks perfectly valid.
|
|
242
|
-
db.exec(options.schemaSql);
|
|
243
|
-
// 🔴 A table D1 holds that the schema does not declare is created from D1's OWN DDL, never
|
|
244
|
-
// dropped: measured on music's first post-cutover pull (2026-09-23), `art_lookups` — a
|
|
245
|
-
// cache its art job creates where it first needs it — made the whole pull refuse with
|
|
246
|
-
// `no such table`, because its rows were exported and its table was not in the schema.
|
|
247
|
-
const have = new Set(
|
|
248
|
-
(db.query("SELECT name FROM sqlite_master WHERE type = 'table'").all() as { name: string }[]).map((r) => r.name),
|
|
249
|
-
);
|
|
250
|
-
for (const row of rows) {
|
|
251
|
-
if (!tables.includes(row.name) || have.has(row.name) || !row.sql) continue;
|
|
252
|
-
db.exec(row.sql.replace(/^\s*CREATE\s+TABLE\s+(?!IF\s+NOT\s+EXISTS)/i, 'CREATE TABLE IF NOT EXISTS '));
|
|
253
|
-
}
|
|
254
|
-
// One table at a time, so a fault is located rather than smeared over the whole load.
|
|
255
|
-
for (const { text } of dumps) if (text.trim()) db.exec(text);
|
|
256
|
-
if (options.afterLoadSql) db.exec(options.afterLoadSql);
|
|
257
|
-
// 🔴 Every table holds as many rows as its dump has INSERTs, or this is not a backup: the
|
|
258
|
-
// check that makes a silently truncated load — the NUL padding, or whatever does it next —
|
|
259
|
-
// a REFUSAL. A row count only; nothing here reads what the rows hold.
|
|
260
|
-
const short: string[] = [];
|
|
261
|
-
for (const { table, text } of dumps) {
|
|
262
|
-
if (table === 'sqlite_sequence') continue;
|
|
263
|
-
const expected = insertLines(text);
|
|
264
|
-
const loaded = (db.query(`SELECT COUNT(*) AS n FROM "${table.replaceAll('"', '""')}"`).get() as { n: number }).n;
|
|
265
|
-
if (loaded < expected) short.push(`${table}: ${loaded} of ${expected}`);
|
|
266
|
-
}
|
|
267
|
-
if (short.length > 0) {
|
|
268
|
-
db.close();
|
|
269
|
-
rmSync(destination, { force: true });
|
|
270
|
-
throw new Error(`the D1 pull did not load every exported row — ${short.join('; ')}. Refusing the snapshot.`);
|
|
271
|
-
}
|
|
272
|
-
} finally {
|
|
350
|
+
if (short.length > 0) {
|
|
273
351
|
db.close();
|
|
352
|
+
rmSync(destination, { force: true });
|
|
353
|
+
throw new Error(`the D1 pull did not load every row it read — ${short.join('; ')}. Refusing the snapshot.`);
|
|
274
354
|
}
|
|
275
355
|
} finally {
|
|
276
|
-
|
|
277
|
-
// would ever delete — eighteen were found beside vault's snapshot (task 2128).
|
|
278
|
-
for (const table of tables) rmSync(`${destination}.${table}.sql`, { force: true });
|
|
356
|
+
db.close();
|
|
279
357
|
}
|
|
280
358
|
}
|