querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,393 @@
1
+ /**
2
+ * The engine entry point, and the whole of the engine's public surface.
3
+ *
4
+ * pure · synchronous · no DOM · no React · fully serializable output
5
+ *
6
+ * Kept worker-ready on purpose (plan.md §5) but deliberately run on the main
7
+ * thread in v0: step-back needs the full event history regardless, so once the
8
+ * UI holds the whole array a worker boundary buys nothing.
9
+ */
10
+ import { emitAnalyzeParseEvents, emitDeleteParseEvents, emitInsertParseEvents, emitParseEvents, emitUpdateParseEvents, parseStatement, } from "./parser/index.js";
11
+ import { analyzeTable, statsToRows } from "./stats.js";
12
+ import { emitDeletePlanEvents, emitPlanEvents, emitUpdatePlanEvents, findIndexNestedLoopJoins, findSortDistinct, isReorderableChain, joinsIn, needsSortSpill, reorderJoinsIfCheaper, } from "./planner/index.js";
13
+ import { buildHeap, createBufferPool } from "./storage/index.js";
14
+ import { buildIndexes, describeUniqueKey, findIndexSpec, indexSpecsOf, pageCountFor } from "./index/index.js";
15
+ import { walkExpr } from "./parser/index.js";
16
+ import { execute } from "./exec/index.js";
17
+ import { executeDelete } from "./exec/delete.js";
18
+ import { executeInsert } from "./exec/insert.js";
19
+ import { executeUpdate } from "./exec/update.js";
20
+ import { UniqueViolation } from "./exec/unique.js";
21
+ import { sortTempPageCount } from "./exec/sort.js";
22
+ import { createLockManager } from "./locks/index.js";
23
+ import { explainResult } from "./explain.js";
24
+ import { resolveWhereSubqueries } from "./subquery.js";
25
+ import { createTracer } from "./trace.js";
26
+ import { DEFAULT_ENGINE_OPTIONS } from "./types.js";
27
+ /** The error a rejected `INSERT` or `UPDATE` reports: which key is taken, by which row, and that nothing changed. */
28
+ function uniqueError(violation, heap, span) {
29
+ const holder = heap.pages.find((p) => p.pageId === violation.holder.pageId)?.rows[violation.holder.slot];
30
+ return {
31
+ message: `Duplicate key — the UNIQUE index on \`${violation.index}\` already holds ${describeUniqueKey(violation.columns, violation.values)}.`,
32
+ hint: `A UNIQUE index allows one row per key (NULLs excepted: NULL is not equal to NULL).${holder ? ` The row that has it: ${JSON.stringify(holder)}.` : ''} Use a different value, or change or delete that row first. Nothing was changed.`,
33
+ from: span.from,
34
+ to: span.to,
35
+ };
36
+ }
37
+ /**
38
+ * `ast`, with its `JOIN`s reordered to the cheapest sequence `reorderJoinsIfCheaper` finds, alongside the order it
39
+ * was actually *typed* in (`writtenOrder`) — present only when reordering genuinely changed something, which is
40
+ * also the signal `emitPlanEvents` uses to narrate it. Otherwise `ast` comes back completely unchanged. Tolerant
41
+ * of a table that turns out not to exist: resolving `db.tables` here is a *look*, not the real validation (that
42
+ * happens right after, against whichever `ast` this returns), so a missing name just means "nothing to reorder,"
43
+ * never a different error than the query would have gotten anyway.
44
+ */
45
+ function applyCostOrder(ast, db, rowsPerPage) {
46
+ if (!isReorderableChain(ast))
47
+ return { ast };
48
+ const names = [ast.from.name, ...(ast.joins ?? []).map((j) => j.table.name)];
49
+ const tables = {};
50
+ for (const name of names) {
51
+ const t = db.tables[name];
52
+ if (!t)
53
+ return { ast };
54
+ tables[name] = t;
55
+ }
56
+ const reordered = reorderJoinsIfCheaper(ast, tables, rowsPerPage);
57
+ return reordered ? { ast: { ...ast, from: reordered.from, joins: reordered.joins }, writtenOrder: names } : { ast };
58
+ }
59
+ export function runQuery(sql, db, options = DEFAULT_ENGINE_OPTIONS, hooks = {}) {
60
+ const { emit, drain } = createTracer();
61
+ const parsed = parseStatement(sql);
62
+ if (!parsed.ok) {
63
+ return { events: drain(), rows: [], error: parsed.error };
64
+ }
65
+ const parsedAst = parsed.ast;
66
+ // EXPLAIN [ANALYZE] SELECT (plan.md §25.4 B6): the inner query, run and then reported as a plan.
67
+ if (parsedAst.kind === 'explain') {
68
+ return explainResult(sql, parsedAst, (inner) => runQuery(inner, db, options, hooks));
69
+ }
70
+ // A chain of 2+ plain inner JOINs may cost less in a different order (plan.md §25.4 C5 slice 3) — resolved here,
71
+ // before any table is looked up for real, so every later step (table/column validation, the heap, the plan
72
+ // itself) already sees whichever order will actually run. Left exactly as written when it is not eligible (a
73
+ // LEFT JOIN anywhere, fewer than two JOINs), already optimal, or names a table that turns out not to exist —
74
+ // the ordinary "no table called X" error just below is exactly as useful either way. `writtenJoinOrder` is only
75
+ // ever set alongside a real change, and is threaded through to `emitPlanEvents` so the trace can say so.
76
+ const { ast, writtenOrder: writtenJoinOrder } = parsedAst.kind === 'select'
77
+ ? applyCostOrder(parsedAst, db, options.rowsPerPage)
78
+ : { ast: parsedAst, writtenOrder: undefined };
79
+ const tableRef = ast.kind === 'insert' ? ast.into : ast.kind === 'update' || ast.kind === 'analyze' ? ast.table : ast.from;
80
+ const table = db.tables[tableRef.name];
81
+ if (!table) {
82
+ const known = Object.keys(db.tables);
83
+ return {
84
+ events: drain(),
85
+ rows: [],
86
+ error: {
87
+ message: `There is no table called \`${tableRef.name}\`.`,
88
+ hint: known.length === 1 ? `The only table is \`${known[0]}\`.` : `Tables: ${known.join(', ')}.`,
89
+ from: tableRef.span.from,
90
+ to: tableRef.span.to,
91
+ },
92
+ };
93
+ }
94
+ // ANALYZE (plan.md §25.4 C1): a real scan, nothing to plan or optimize, and no join/column checks below apply.
95
+ if (ast.kind === 'analyze') {
96
+ emitAnalyzeParseEvents(emit, ast);
97
+ const stats = analyzeTable(table);
98
+ const rows = statsToRows(stats);
99
+ emit(`ANALYZE scanned all ${String(table.rows.length)} row${table.rows.length === 1 ? '' : 's'} of \`${table.name}\` and built statistics for ${String(table.columns.length)} column${table.columns.length === 1 ? '' : 's'}. The next query against \`${table.name}\` reads these instead of scanning the live rows to estimate a plan.`, { stage: 'result', rows });
100
+ return { events: drain(), rows, analyze: { table: table.name, stats } };
101
+ }
102
+ // Every JOIN's own table (plan.md §25.4 C3 slice b: 0 or more, left-deep) — resolved here, alongside the
103
+ // FROM/INTO table above, so all of them go through the same "no table called X" shape.
104
+ const joins = ast.kind === 'select' ? (ast.joins ?? []) : [];
105
+ const joinTables = {};
106
+ for (const j of joins) {
107
+ const t = db.tables[j.table.name];
108
+ if (!t) {
109
+ const known = Object.keys(db.tables);
110
+ return {
111
+ events: drain(),
112
+ rows: [],
113
+ error: {
114
+ message: `There is no table called \`${j.table.name}\`.`,
115
+ hint: known.length === 1 ? `The only table is \`${known[0]}\`.` : `Tables: ${known.join(', ')}.`,
116
+ from: j.table.span.from,
117
+ to: j.table.span.to,
118
+ },
119
+ };
120
+ }
121
+ joinTables[j.table.name] = t;
122
+ }
123
+ const unknownColumn = referencedColumns(ast).find((c) => {
124
+ const named = c.table ? db.tables[c.table] : table;
125
+ return !named || !named.columns.includes(c.name);
126
+ });
127
+ if (unknownColumn) {
128
+ const named = (unknownColumn.table ? db.tables[unknownColumn.table] : table);
129
+ return {
130
+ events: drain(),
131
+ rows: [],
132
+ error: {
133
+ message: `\`${named.name}\` has no column called \`${unknownColumn.name}\`.`,
134
+ hint: `Columns: ${named.columns.join(', ')}.`,
135
+ from: unknownColumn.span.from,
136
+ to: unknownColumn.span.to,
137
+ },
138
+ };
139
+ }
140
+ if (ast.kind === 'insert' && !ast.columns && ast.values.items.length !== table.columns.length) {
141
+ return {
142
+ events: drain(),
143
+ rows: [],
144
+ error: {
145
+ message: `\`${table.name}\` has ${table.columns.length} column${table.columns.length === 1 ? '' : 's'}, but this INSERT gives ${ast.values.items.length} value${ast.values.items.length === 1 ? '' : 's'}.`,
146
+ hint: `Columns: ${table.columns.join(', ')}. Name the columns you're setting if you don't want to supply all of them, e.g. INSERT INTO ${table.name} (${table.columns[0]}) VALUES (...).`,
147
+ from: ast.values.span.from,
148
+ to: ast.values.span.to,
149
+ },
150
+ };
151
+ }
152
+ const heap = buildHeap(table, options.rowsPerPage);
153
+ const trees = buildIndexes(heap, indexSpecsOf(table), undefined, undefined, options.indexBuild, table.clusteredKey);
154
+ const locks = createLockManager(emit);
155
+ if (ast.kind === 'insert') {
156
+ emitInsertParseEvents(emit, ast);
157
+ const row = buildInsertRow(ast, table);
158
+ const lastPage = heap.pages[heap.pages.length - 1];
159
+ const insertNeedsNewPage = lastPage.rows.length >= options.rowsPerPage;
160
+ const pool = createBufferPool(pageCountFor(heap, trees) + (insertNeedsNewPage ? 1 : 0), options, emit);
161
+ let inserted;
162
+ try {
163
+ inserted = executeInsert(row, { emit, pool, locks, heap, trees, table, options }).row;
164
+ }
165
+ catch (e) {
166
+ if (e instanceof UniqueViolation) {
167
+ const error = uniqueError(e, heap, ast.values.span);
168
+ emit(`INSERT is rejected — ${error.message} Nothing was changed.`, { stage: 'result', rows: [] });
169
+ return { events: drain(), rows: [], error };
170
+ }
171
+ throw e;
172
+ }
173
+ hooks.afterWrite?.(trees);
174
+ const rows = [...table.rows, inserted];
175
+ emit(`1 row inserted, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.writes} write${pool.stats.writes === 1 ? '' : 's'}. ${table.name} now has ${rows.length} row${rows.length === 1 ? '' : 's'}.`, { stage: 'result', rows: [inserted] });
176
+ return {
177
+ events: drain(),
178
+ rows: [inserted],
179
+ write: { kind: 'insert', table: table.name, insertedCount: 1, rows },
180
+ };
181
+ }
182
+ if (ast.kind === 'delete') {
183
+ emitDeleteParseEvents(emit, ast);
184
+ // Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
185
+ // the parse-stage narration above already showed the query exactly as written.
186
+ const resolved = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
187
+ if (!resolved.ok)
188
+ return { events: drain(), rows: [], error: resolved.error };
189
+ const deleteAst = { ...ast, ...(resolved.where ? { where: resolved.where } : {}) };
190
+ const plan = emitDeletePlanEvents(emit, deleteAst, table, options.rowsPerPage);
191
+ const pool = createBufferPool(pageCountFor(heap, trees), options, emit);
192
+ const { deleted } = executeDelete(plan, { emit, pool, locks, heap, trees, table, options });
193
+ hooks.afterWrite?.(trees);
194
+ const removed = new Set(deleted);
195
+ const rows = table.rows.filter((r) => !removed.has(r));
196
+ emit(`${deleted.length} row${deleted.length === 1 ? '' : 's'} deleted, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.evictions} eviction${pool.stats.evictions === 1 ? '' : 's'}. ${table.name} now has ${rows.length} row${rows.length === 1 ? '' : 's'}.`, { stage: 'result', rows: deleted });
197
+ return {
198
+ events: drain(),
199
+ rows: deleted,
200
+ write: { kind: 'delete', table: table.name, deletedCount: deleted.length, rows },
201
+ };
202
+ }
203
+ if (ast.kind === 'update') {
204
+ emitUpdateParseEvents(emit, ast);
205
+ // Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
206
+ // the parse-stage narration above already showed the query exactly as written.
207
+ const resolved = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
208
+ if (!resolved.ok)
209
+ return { events: drain(), rows: [], error: resolved.error };
210
+ const updateAst = { ...ast, ...(resolved.where ? { where: resolved.where } : {}) };
211
+ const plan = emitUpdatePlanEvents(emit, updateAst, table, options.rowsPerPage);
212
+ const pool = createBufferPool(pageCountFor(heap, trees), options, emit);
213
+ let changed;
214
+ try {
215
+ changed = executeUpdate(plan, ast.set.assignments, { emit, pool, locks, heap, trees, table, options });
216
+ }
217
+ catch (e) {
218
+ if (e instanceof UniqueViolation) {
219
+ const error = uniqueError(e, heap, ast.set.span);
220
+ emit(`UPDATE is rejected — ${error.message} Nothing was changed.`, { stage: 'result', rows: [] });
221
+ return { events: drain(), rows: [], error };
222
+ }
223
+ throw e;
224
+ }
225
+ const { before, after } = changed;
226
+ hooks.afterWrite?.(trees);
227
+ const replacements = new Map(before.map((row, i) => [row, after[i]]));
228
+ const rows = table.rows.map((r) => replacements.get(r) ?? r);
229
+ emit(`${after.length} row${after.length === 1 ? '' : 's'} updated, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.evictions} eviction${pool.stats.evictions === 1 ? '' : 's'}.`, { stage: 'result', rows: after });
230
+ return {
231
+ events: drain(),
232
+ rows: after,
233
+ write: { kind: 'update', table: table.name, updatedCount: after.length, rows },
234
+ };
235
+ }
236
+ // Always the query exactly as typed — `ast` may already be the cost-reordered version by this point (plan.md
237
+ // §25.4 C5 slice 3), but the parse stage narrates source text, never a planning decision; `parsedAst` never was.
238
+ emitParseEvents(emit, parsedAst.kind === 'select' ? parsedAst : ast);
239
+ // Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
240
+ // the parse-stage narration above already showed the query exactly as written.
241
+ const resolvedWhere = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
242
+ if (!resolvedWhere.ok)
243
+ return { events: drain(), rows: [], error: resolvedWhere.error };
244
+ const selectAst = { ...ast, ...(resolvedWhere.where ? { where: resolvedWhere.where } : {}) };
245
+ const plan = emitPlanEvents(emit, selectAst, table, options.rowsPerPage, options.aggregateStrategy ?? 'hash', joinTables, options.joinStrategy, writtenJoinOrder);
246
+ // Every joined table gets its own heap, its pages numbered right after whatever came before it — so all of them
247
+ // compete for the same buffer frames, the same way a Sort's spill pages do (plan.md §25.4 C3 slice b generalizes
248
+ // this from one join partner to a whole left-deep chain). Nested-loop, hash and sort-merge never consult any
249
+ // side's own index, so there is no B+Tree to build for a join partner unless that particular step runs
250
+ // index-nested-loop (plan.md §25.4 B4) — each one that does gets its one needed index built next, past every
251
+ // heap and every earlier probe index, so nothing collides.
252
+ const joinHeaps = {};
253
+ let pageBase = pageCountFor(heap, trees);
254
+ for (const j of joins) {
255
+ const jHeap = buildHeap(joinTables[j.table.name], options.rowsPerPage, pageBase);
256
+ joinHeaps[j.table.name] = jHeap;
257
+ pageBase += jHeap.pages.length;
258
+ }
259
+ const joinTrees = {};
260
+ for (const ij of findIndexNestedLoopJoins(plan)) {
261
+ if (ij.right.op !== 'IndexProbe')
262
+ continue; // unreachable: an index-nested-loop Join's own inner side is always a probe
263
+ const jTable = joinTables[ij.rightTable];
264
+ const probeIndex = findIndexSpec(jTable, ij.right.column);
265
+ if (!probeIndex)
266
+ continue; // unreachable: effectiveJoinAlgorithm already confirmed this index exists
267
+ const jTrees = buildIndexes(joinHeaps[ij.rightTable], [probeIndex], undefined, pageBase, options.indexBuild, jTable.clusteredKey);
268
+ joinTrees[ij.rightTable] = jTrees;
269
+ pageBase = pageCountFor(joinHeaps[ij.rightTable], jTrees);
270
+ }
271
+ const heapsEnd = pageBase;
272
+ // Every `sort-merge` Join in the chain needs its own two-block reservation — one per side, since each sorts
273
+ // independently — placed in chain order (the innermost join's scratch first, mirroring the order joins actually
274
+ // execute in) right after every table's heap and any probe index. A single join's worst-case left input is just
275
+ // the FROM table itself (`table.rows.length`, unchanged from before chains existed); a later step's worst case is
276
+ // the product of every earlier table's row count, since no join can ever produce more rows than that.
277
+ const sortMergeBases = new Map();
278
+ let worstCaseLeftRows = table.rows.length;
279
+ for (const j of joinsIn(plan)) {
280
+ const jTable = joinTables[j.rightTable];
281
+ if (j.algorithm === 'sort-merge') {
282
+ const leftSize = sortTempPageCount(worstCaseLeftRows, options.rowsPerPage, options.bufferFrames);
283
+ const leftBase = pageBase;
284
+ pageBase += leftSize;
285
+ const rightSize = sortTempPageCount(jTable.rows.length, options.rowsPerPage, options.bufferFrames);
286
+ const rightBase = pageBase;
287
+ pageBase += rightSize;
288
+ sortMergeBases.set(j, { leftBase, rightBase });
289
+ }
290
+ worstCaseLeftRows *= jTable.rows.length;
291
+ }
292
+ // A Sort's (or a SortAggregate's own internal sort's) spill needs its own page ids, reserved up front — sized
293
+ // against the whole table (the worst case; a WHERE clause can only shrink what actually reaches it) so the
294
+ // reservation is never too small to discover only at runtime. This and the sort-merge reservation above never
295
+ // both apply: ORDER BY/GROUP BY still can't combine with JOIN at all.
296
+ const sortTemp = needsSortSpill(plan)
297
+ ? sortTempPageCount(table.rows.length, options.rowsPerPage, options.bufferFrames)
298
+ : pageBase - heapsEnd;
299
+ // A sort-based DISTINCT (plan.md §25.4 B2) sorts what the projection produced, which after a JOIN can be as many
300
+ // rows as every joined table's product (capped — these are teaching-sized tables) — and it gets its own block of
301
+ // scratch pages *after* whichever other sort the plan has, so no two sorts share a page id.
302
+ const distinctRows = joins.length > 0
303
+ ? Math.min(joins.reduce((product, j) => product * joinTables[j.table.name].rows.length, table.rows.length), 20_000)
304
+ : table.rows.length;
305
+ const distinctSpill = findSortDistinct(plan)
306
+ ? sortTempPageCount(distinctRows, options.rowsPerPage, options.bufferFrames)
307
+ : 0;
308
+ const distinctBase = heapsEnd + sortTemp;
309
+ const pool = createBufferPool(distinctBase + distinctSpill, options, emit);
310
+ const rows = execute(plan, {
311
+ emit,
312
+ pool,
313
+ locks,
314
+ heap,
315
+ trees,
316
+ table,
317
+ options,
318
+ distinctBase,
319
+ ...(joins.length > 0
320
+ ? {
321
+ joinPartners: Object.fromEntries(joins.map((j) => [
322
+ j.table.name,
323
+ { table: joinTables[j.table.name], heap: joinHeaps[j.table.name], trees: joinTrees[j.table.name] ?? {} },
324
+ ])),
325
+ }
326
+ : {}),
327
+ ...(sortMergeBases.size > 0 ? { sortMergeBases } : {}),
328
+ });
329
+ return { events: drain(), rows };
330
+ }
331
+ function referencedColumns(ast) {
332
+ const refs = [];
333
+ const push = (ref) => refs.push({ name: ref.name, ...(ref.table === undefined ? {} : { table: ref.table }), span: ref.span });
334
+ if (ast.kind === 'select' && ast.select.kind === 'columns') {
335
+ for (const item of ast.select.items) {
336
+ if (item.kind === 'column')
337
+ push(item);
338
+ else if (item.kind === 'expression')
339
+ walkExpr(item.expr, (e) => { if (e.kind === 'column')
340
+ push(e); });
341
+ else if (item.arg.kind === 'column')
342
+ push(item.arg);
343
+ }
344
+ }
345
+ if (ast.kind !== 'insert' && ast.kind !== 'analyze' && ast.where) {
346
+ walkExpr(ast.where, (e) => {
347
+ if (e.kind === 'column')
348
+ push(e);
349
+ });
350
+ }
351
+ if (ast.kind === 'select') {
352
+ for (const join of ast.joins ?? []) {
353
+ if (join.on.left.kind === 'column')
354
+ push(join.on.left);
355
+ if (join.on.right.kind === 'column')
356
+ push(join.on.right);
357
+ }
358
+ }
359
+ if (ast.kind === 'select' && ast.groupBy)
360
+ refs.push(...ast.groupBy.columns);
361
+ if (ast.kind === 'select' && ast.having) {
362
+ // A HAVING predicate names grouped columns and aggregate arguments — each must exist.
363
+ walkExpr(ast.having, (e) => {
364
+ if (e.kind === 'column')
365
+ push(e);
366
+ else if (e.kind === 'aggregate' && e.arg.kind === 'column')
367
+ push(e.arg);
368
+ });
369
+ }
370
+ if (ast.kind === 'select' && ast.orderBy)
371
+ refs.push({ name: ast.orderBy.column, span: ast.orderBy.span });
372
+ if (ast.kind === 'insert' && ast.columns)
373
+ refs.push(...ast.columns.names);
374
+ if (ast.kind === 'update') {
375
+ refs.push(...ast.set.assignments.map((a) => ({ name: a.column, span: a.columnSpan })));
376
+ // A SET value can read columns (`SET n = n + 1`), and each must exist.
377
+ for (const a of ast.set.assignments)
378
+ walkExpr(a.value, (e) => { if (e.kind === 'column')
379
+ push(e); });
380
+ }
381
+ return refs;
382
+ }
383
+ /** Builds the new row: every table column defaults to `null`, then the given values overwrite it. */
384
+ function buildInsertRow(ast, table) {
385
+ const row = {};
386
+ for (const column of table.columns)
387
+ row[column] = null;
388
+ const names = ast.columns ? ast.columns.names.map((c) => c.name) : table.columns;
389
+ names.forEach((name, i) => {
390
+ row[name] = ast.values.items[i].value;
391
+ });
392
+ return row;
393
+ }
@@ -0,0 +1,165 @@
1
+ /**
2
+ * 48 rows at 4 rows per page is 12 pages against 8 buffer frames, so a full
3
+ * sequential scan evicts. That contrast is the point: plan.md §9 wants the
4
+ * scan's eviction storm shown next to an index lookup that touches almost
5
+ * nothing. A table small enough to never evict would make the buffer pool —
6
+ * the product's best differentiator — look inert.
7
+ */
8
+ const NAMES = [
9
+ 'ada', 'alan', 'barbara', 'brendan', 'carmack', 'carol', 'dennis', 'dijkstra',
10
+ 'donald', 'edsger', 'engelbart', 'fowler', 'frances', 'grace', 'guido', 'hamilton',
11
+ 'hedy', 'hopper', 'iverson', 'james', 'jean', 'john', 'katherine', 'ken',
12
+ 'knuth', 'lamport', 'larry', 'leslie', 'linus', 'lynn', 'margaret', 'mary',
13
+ 'matz', 'niklaus', 'radia', 'richard', 'ritchie', 'rob', 'robert', 'shafi',
14
+ 'stephen', 'thompson', 'tim', 'tony', 'torvalds', 'turing', 'vint', 'yukihiro',
15
+ ];
16
+ function seedRows() {
17
+ return NAMES.map((name, i) => ({
18
+ id: i + 1,
19
+ name,
20
+ email: `${name}@example.com`,
21
+ age: 20 + ((i * 7) % 45),
22
+ }));
23
+ }
24
+ /**
25
+ * The starting dataset. The user can edit these rows and re-run (plan.md §4).
26
+ */
27
+ export function createSeedDatabase() {
28
+ return {
29
+ tables: {
30
+ users: {
31
+ name: 'users',
32
+ columns: ['id', 'name', 'email', 'age'],
33
+ rows: seedRows(),
34
+ indexedColumns: ['id'],
35
+ },
36
+ },
37
+ };
38
+ }
39
+ const AUTHOR_NAMES = ['ada', 'alan', 'grace', 'donald', 'barbara'];
40
+ // `authorId` is a real foreign key shape: two authors with several posts,
41
+ // one with exactly one, and one with none at all — the ordinary distribution
42
+ // a real join produces, not every row matching neatly. Small on purpose: v1
43
+ // JOIN is nested-loop only, so every row on one side rescans the *other*
44
+ // side whole — small tables keep that honest even before an index-aware join
45
+ // strategy exists to speed it up.
46
+ const POST_AUTHORS = [1, 1, 1, 2, 2, 3, 3, 3, 3, 4];
47
+ /**
48
+ * A small second table, joinable against a small first one (plan.md §22.2) —
49
+ * `authors`/`posts` rather than reusing the 48-row `users`, so a full,
50
+ * unfiltered join stays small enough to read as a trace. Not part of
51
+ * `datasets.ts`'s preset library: those are each one table by design (§22.7);
52
+ * this pair exists to give `/tools/planner` something to JOIN against.
53
+ */
54
+ export function createJoinDemoDatabase() {
55
+ return {
56
+ tables: {
57
+ authors: {
58
+ name: 'authors',
59
+ columns: ['id', 'name'],
60
+ rows: AUTHOR_NAMES.map((name, i) => ({ id: i + 1, name })),
61
+ indexedColumns: ['id'],
62
+ },
63
+ posts: {
64
+ name: 'posts',
65
+ columns: ['id', 'authorId', 'title'],
66
+ rows: POST_AUTHORS.map((authorId, i) => ({
67
+ id: i + 1,
68
+ authorId,
69
+ title: `${AUTHOR_NAMES[authorId - 1]}'s post ${String(i + 1)}`,
70
+ })),
71
+ // Three indexes on purpose. `id` and `title` give `/tools/planner`'s index-selection rule a second
72
+ // candidate to weigh (plan.md §22.2); `authorId` is the foreign key, and it *repeats* — 4 distinct
73
+ // values among 10 rows — so `WHERE authorId = 1` walks an index that holds three entries for one value
74
+ // and returns all three rows (plan.md §25.4 B3). Until then a B+Tree here was unique-only, and
75
+ // indexing this column would have silently returned one row instead of three.
76
+ indexedColumns: ['id', 'title', 'authorId'],
77
+ },
78
+ },
79
+ };
80
+ }
81
+ /**
82
+ * Four tables, shaped as a tree rather than a chain — `departments` is the hub, connecting to both `employees`
83
+ * and `projects`; `timesheets` hangs off `employees` alone (plan.md §25.4 C4, `/tools/join-order` + the
84
+ * Playground's own join-order note). Sized so join *order* has a real, measured answer rather than an obvious
85
+ * one: `timesheets` is the one genuinely large table (5,000 rows against the other three's tens or hundreds), so
86
+ * whichever join brings it in pays a real cost — but *when* that happens (how large the accumulated join is by
87
+ * the time it does) is what actually separates a good order from a bad one, not simply "big table first" or
88
+ * "big table last." Tuned empirically (the project's own "measure, don't guess" convention) rather than derived:
89
+ * `FROM departments JOIN projects ON departments.id = projects.deptId JOIN employees ON departments.id =
90
+ * employees.deptId JOIN timesheets ON employees.id = timesheets.employeeId` — a natural-looking "hub, then each
91
+ * branch" order — costs 5× the DP-optimal `departments → employees → timesheets → projects`, because visiting
92
+ * `projects` before drilling into `employees` → `timesheets` lets the *other* branch's join happen against an
93
+ * already-larger intermediate. Every foreign key is uniform (rows divide evenly across parents), so the only
94
+ * lever is order, not skew.
95
+ */
96
+ export function createJoinOrderDemoDatabase() {
97
+ const DEPARTMENT_COUNT = 5;
98
+ const EMPLOYEES_PER_DEPARTMENT = 40;
99
+ const PROJECTS_PER_DEPARTMENT = 6;
100
+ const TIMESHEETS_PER_EMPLOYEE = 25;
101
+ const employeeCount = DEPARTMENT_COUNT * EMPLOYEES_PER_DEPARTMENT;
102
+ return {
103
+ tables: {
104
+ departments: {
105
+ name: 'departments',
106
+ columns: ['id', 'name'],
107
+ rows: Array.from({ length: DEPARTMENT_COUNT }, (_, i) => ({ id: i + 1, name: `dept-${String(i + 1)}` })),
108
+ indexedColumns: ['id'],
109
+ },
110
+ employees: {
111
+ name: 'employees',
112
+ columns: ['id', 'deptId', 'name'],
113
+ rows: Array.from({ length: employeeCount }, (_, i) => ({
114
+ id: i + 1,
115
+ deptId: Math.floor(i / EMPLOYEES_PER_DEPARTMENT) + 1,
116
+ name: `employee-${String(i + 1)}`,
117
+ })),
118
+ indexedColumns: ['id', 'deptId'],
119
+ },
120
+ projects: {
121
+ name: 'projects',
122
+ columns: ['id', 'deptId', 'name'],
123
+ rows: Array.from({ length: DEPARTMENT_COUNT * PROJECTS_PER_DEPARTMENT }, (_, i) => ({
124
+ id: i + 1,
125
+ deptId: Math.floor(i / PROJECTS_PER_DEPARTMENT) + 1,
126
+ name: `project-${String(i + 1)}`,
127
+ })),
128
+ indexedColumns: ['id', 'deptId'],
129
+ },
130
+ timesheets: {
131
+ name: 'timesheets',
132
+ columns: ['id', 'employeeId', 'hours'],
133
+ rows: Array.from({ length: employeeCount * TIMESHEETS_PER_EMPLOYEE }, (_, i) => ({
134
+ id: i + 1,
135
+ employeeId: Math.floor(i / TIMESHEETS_PER_EMPLOYEE) + 1,
136
+ hours: 1 + (i % 8),
137
+ })),
138
+ indexedColumns: ['id', 'employeeId'],
139
+ },
140
+ },
141
+ };
142
+ }
143
+ /**
144
+ * `createJoinDemoDatabase`'s pair plus a larger third table, for `/compare`'s JOIN mode (plan.md §25.4 B4). Ten
145
+ * posts on three pages cannot show what an index join is *for* — a probe of `posts` costs more than one scan
146
+ * of it. `comments` is the inner table that can: 120 rows on 30 pages, twelve per post and scattered across them
147
+ * (every tenth row is one post's), indexed on `postId` (a foreign key, so it repeats). One outer row probing that
148
+ * index reads a few tree pages and the pages holding its own twelve comments — well under the thirty a scan reads —
149
+ * while ten outer rows probing it cost more than one scan of the lot, which is the other half of the lesson.
150
+ */
151
+ export function createJoinCompareDatabase() {
152
+ const base = createJoinDemoDatabase();
153
+ return {
154
+ tables: {
155
+ ...base.tables,
156
+ comments: {
157
+ name: 'comments',
158
+ columns: ['id', 'postId', 'body'],
159
+ // `i * 7 mod 10` visits every post id in a scrambled order, so one post's comments sit on many pages.
160
+ rows: Array.from({ length: 120 }, (_, i) => ({ id: i + 1, postId: ((i * 7) % 10) + 1, body: `comment ${String(i + 1)}` })),
161
+ indexedColumns: ['postId'],
162
+ },
163
+ },
164
+ };
165
+ }