querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,393 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The engine entry point, and the whole of the engine's public surface.
|
|
3
|
+
*
|
|
4
|
+
* pure · synchronous · no DOM · no React · fully serializable output
|
|
5
|
+
*
|
|
6
|
+
* Kept worker-ready on purpose (plan.md §5) but deliberately run on the main
|
|
7
|
+
* thread in v0: step-back needs the full event history regardless, so once the
|
|
8
|
+
* UI holds the whole array a worker boundary buys nothing.
|
|
9
|
+
*/
|
|
10
|
+
import { emitAnalyzeParseEvents, emitDeleteParseEvents, emitInsertParseEvents, emitParseEvents, emitUpdateParseEvents, parseStatement, } from "./parser/index.js";
|
|
11
|
+
import { analyzeTable, statsToRows } from "./stats.js";
|
|
12
|
+
import { emitDeletePlanEvents, emitPlanEvents, emitUpdatePlanEvents, findIndexNestedLoopJoins, findSortDistinct, isReorderableChain, joinsIn, needsSortSpill, reorderJoinsIfCheaper, } from "./planner/index.js";
|
|
13
|
+
import { buildHeap, createBufferPool } from "./storage/index.js";
|
|
14
|
+
import { buildIndexes, describeUniqueKey, findIndexSpec, indexSpecsOf, pageCountFor } from "./index/index.js";
|
|
15
|
+
import { walkExpr } from "./parser/index.js";
|
|
16
|
+
import { execute } from "./exec/index.js";
|
|
17
|
+
import { executeDelete } from "./exec/delete.js";
|
|
18
|
+
import { executeInsert } from "./exec/insert.js";
|
|
19
|
+
import { executeUpdate } from "./exec/update.js";
|
|
20
|
+
import { UniqueViolation } from "./exec/unique.js";
|
|
21
|
+
import { sortTempPageCount } from "./exec/sort.js";
|
|
22
|
+
import { createLockManager } from "./locks/index.js";
|
|
23
|
+
import { explainResult } from "./explain.js";
|
|
24
|
+
import { resolveWhereSubqueries } from "./subquery.js";
|
|
25
|
+
import { createTracer } from "./trace.js";
|
|
26
|
+
import { DEFAULT_ENGINE_OPTIONS } from "./types.js";
|
|
27
|
+
/** The error a rejected `INSERT` or `UPDATE` reports: which key is taken, by which row, and that nothing changed. */
|
|
28
|
+
function uniqueError(violation, heap, span) {
|
|
29
|
+
const holder = heap.pages.find((p) => p.pageId === violation.holder.pageId)?.rows[violation.holder.slot];
|
|
30
|
+
return {
|
|
31
|
+
message: `Duplicate key — the UNIQUE index on \`${violation.index}\` already holds ${describeUniqueKey(violation.columns, violation.values)}.`,
|
|
32
|
+
hint: `A UNIQUE index allows one row per key (NULLs excepted: NULL is not equal to NULL).${holder ? ` The row that has it: ${JSON.stringify(holder)}.` : ''} Use a different value, or change or delete that row first. Nothing was changed.`,
|
|
33
|
+
from: span.from,
|
|
34
|
+
to: span.to,
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* `ast`, with its `JOIN`s reordered to the cheapest sequence `reorderJoinsIfCheaper` finds, alongside the order it
|
|
39
|
+
* was actually *typed* in (`writtenOrder`) — present only when reordering genuinely changed something, which is
|
|
40
|
+
* also the signal `emitPlanEvents` uses to narrate it. Otherwise `ast` comes back completely unchanged. Tolerant
|
|
41
|
+
* of a table that turns out not to exist: resolving `db.tables` here is a *look*, not the real validation (that
|
|
42
|
+
* happens right after, against whichever `ast` this returns), so a missing name just means "nothing to reorder,"
|
|
43
|
+
* never a different error than the query would have gotten anyway.
|
|
44
|
+
*/
|
|
45
|
+
function applyCostOrder(ast, db, rowsPerPage) {
|
|
46
|
+
if (!isReorderableChain(ast))
|
|
47
|
+
return { ast };
|
|
48
|
+
const names = [ast.from.name, ...(ast.joins ?? []).map((j) => j.table.name)];
|
|
49
|
+
const tables = {};
|
|
50
|
+
for (const name of names) {
|
|
51
|
+
const t = db.tables[name];
|
|
52
|
+
if (!t)
|
|
53
|
+
return { ast };
|
|
54
|
+
tables[name] = t;
|
|
55
|
+
}
|
|
56
|
+
const reordered = reorderJoinsIfCheaper(ast, tables, rowsPerPage);
|
|
57
|
+
return reordered ? { ast: { ...ast, from: reordered.from, joins: reordered.joins }, writtenOrder: names } : { ast };
|
|
58
|
+
}
|
|
59
|
+
export function runQuery(sql, db, options = DEFAULT_ENGINE_OPTIONS, hooks = {}) {
|
|
60
|
+
const { emit, drain } = createTracer();
|
|
61
|
+
const parsed = parseStatement(sql);
|
|
62
|
+
if (!parsed.ok) {
|
|
63
|
+
return { events: drain(), rows: [], error: parsed.error };
|
|
64
|
+
}
|
|
65
|
+
const parsedAst = parsed.ast;
|
|
66
|
+
// EXPLAIN [ANALYZE] SELECT (plan.md §25.4 B6): the inner query, run and then reported as a plan.
|
|
67
|
+
if (parsedAst.kind === 'explain') {
|
|
68
|
+
return explainResult(sql, parsedAst, (inner) => runQuery(inner, db, options, hooks));
|
|
69
|
+
}
|
|
70
|
+
// A chain of 2+ plain inner JOINs may cost less in a different order (plan.md §25.4 C5 slice 3) — resolved here,
|
|
71
|
+
// before any table is looked up for real, so every later step (table/column validation, the heap, the plan
|
|
72
|
+
// itself) already sees whichever order will actually run. Left exactly as written when it is not eligible (a
|
|
73
|
+
// LEFT JOIN anywhere, fewer than two JOINs), already optimal, or names a table that turns out not to exist —
|
|
74
|
+
// the ordinary "no table called X" error just below is exactly as useful either way. `writtenJoinOrder` is only
|
|
75
|
+
// ever set alongside a real change, and is threaded through to `emitPlanEvents` so the trace can say so.
|
|
76
|
+
const { ast, writtenOrder: writtenJoinOrder } = parsedAst.kind === 'select'
|
|
77
|
+
? applyCostOrder(parsedAst, db, options.rowsPerPage)
|
|
78
|
+
: { ast: parsedAst, writtenOrder: undefined };
|
|
79
|
+
const tableRef = ast.kind === 'insert' ? ast.into : ast.kind === 'update' || ast.kind === 'analyze' ? ast.table : ast.from;
|
|
80
|
+
const table = db.tables[tableRef.name];
|
|
81
|
+
if (!table) {
|
|
82
|
+
const known = Object.keys(db.tables);
|
|
83
|
+
return {
|
|
84
|
+
events: drain(),
|
|
85
|
+
rows: [],
|
|
86
|
+
error: {
|
|
87
|
+
message: `There is no table called \`${tableRef.name}\`.`,
|
|
88
|
+
hint: known.length === 1 ? `The only table is \`${known[0]}\`.` : `Tables: ${known.join(', ')}.`,
|
|
89
|
+
from: tableRef.span.from,
|
|
90
|
+
to: tableRef.span.to,
|
|
91
|
+
},
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
// ANALYZE (plan.md §25.4 C1): a real scan, nothing to plan or optimize, and no join/column checks below apply.
|
|
95
|
+
if (ast.kind === 'analyze') {
|
|
96
|
+
emitAnalyzeParseEvents(emit, ast);
|
|
97
|
+
const stats = analyzeTable(table);
|
|
98
|
+
const rows = statsToRows(stats);
|
|
99
|
+
emit(`ANALYZE scanned all ${String(table.rows.length)} row${table.rows.length === 1 ? '' : 's'} of \`${table.name}\` and built statistics for ${String(table.columns.length)} column${table.columns.length === 1 ? '' : 's'}. The next query against \`${table.name}\` reads these instead of scanning the live rows to estimate a plan.`, { stage: 'result', rows });
|
|
100
|
+
return { events: drain(), rows, analyze: { table: table.name, stats } };
|
|
101
|
+
}
|
|
102
|
+
// Every JOIN's own table (plan.md §25.4 C3 slice b: 0 or more, left-deep) — resolved here, alongside the
|
|
103
|
+
// FROM/INTO table above, so all of them go through the same "no table called X" shape.
|
|
104
|
+
const joins = ast.kind === 'select' ? (ast.joins ?? []) : [];
|
|
105
|
+
const joinTables = {};
|
|
106
|
+
for (const j of joins) {
|
|
107
|
+
const t = db.tables[j.table.name];
|
|
108
|
+
if (!t) {
|
|
109
|
+
const known = Object.keys(db.tables);
|
|
110
|
+
return {
|
|
111
|
+
events: drain(),
|
|
112
|
+
rows: [],
|
|
113
|
+
error: {
|
|
114
|
+
message: `There is no table called \`${j.table.name}\`.`,
|
|
115
|
+
hint: known.length === 1 ? `The only table is \`${known[0]}\`.` : `Tables: ${known.join(', ')}.`,
|
|
116
|
+
from: j.table.span.from,
|
|
117
|
+
to: j.table.span.to,
|
|
118
|
+
},
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
joinTables[j.table.name] = t;
|
|
122
|
+
}
|
|
123
|
+
const unknownColumn = referencedColumns(ast).find((c) => {
|
|
124
|
+
const named = c.table ? db.tables[c.table] : table;
|
|
125
|
+
return !named || !named.columns.includes(c.name);
|
|
126
|
+
});
|
|
127
|
+
if (unknownColumn) {
|
|
128
|
+
const named = (unknownColumn.table ? db.tables[unknownColumn.table] : table);
|
|
129
|
+
return {
|
|
130
|
+
events: drain(),
|
|
131
|
+
rows: [],
|
|
132
|
+
error: {
|
|
133
|
+
message: `\`${named.name}\` has no column called \`${unknownColumn.name}\`.`,
|
|
134
|
+
hint: `Columns: ${named.columns.join(', ')}.`,
|
|
135
|
+
from: unknownColumn.span.from,
|
|
136
|
+
to: unknownColumn.span.to,
|
|
137
|
+
},
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
if (ast.kind === 'insert' && !ast.columns && ast.values.items.length !== table.columns.length) {
|
|
141
|
+
return {
|
|
142
|
+
events: drain(),
|
|
143
|
+
rows: [],
|
|
144
|
+
error: {
|
|
145
|
+
message: `\`${table.name}\` has ${table.columns.length} column${table.columns.length === 1 ? '' : 's'}, but this INSERT gives ${ast.values.items.length} value${ast.values.items.length === 1 ? '' : 's'}.`,
|
|
146
|
+
hint: `Columns: ${table.columns.join(', ')}. Name the columns you're setting if you don't want to supply all of them, e.g. INSERT INTO ${table.name} (${table.columns[0]}) VALUES (...).`,
|
|
147
|
+
from: ast.values.span.from,
|
|
148
|
+
to: ast.values.span.to,
|
|
149
|
+
},
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
const heap = buildHeap(table, options.rowsPerPage);
|
|
153
|
+
const trees = buildIndexes(heap, indexSpecsOf(table), undefined, undefined, options.indexBuild, table.clusteredKey);
|
|
154
|
+
const locks = createLockManager(emit);
|
|
155
|
+
if (ast.kind === 'insert') {
|
|
156
|
+
emitInsertParseEvents(emit, ast);
|
|
157
|
+
const row = buildInsertRow(ast, table);
|
|
158
|
+
const lastPage = heap.pages[heap.pages.length - 1];
|
|
159
|
+
const insertNeedsNewPage = lastPage.rows.length >= options.rowsPerPage;
|
|
160
|
+
const pool = createBufferPool(pageCountFor(heap, trees) + (insertNeedsNewPage ? 1 : 0), options, emit);
|
|
161
|
+
let inserted;
|
|
162
|
+
try {
|
|
163
|
+
inserted = executeInsert(row, { emit, pool, locks, heap, trees, table, options }).row;
|
|
164
|
+
}
|
|
165
|
+
catch (e) {
|
|
166
|
+
if (e instanceof UniqueViolation) {
|
|
167
|
+
const error = uniqueError(e, heap, ast.values.span);
|
|
168
|
+
emit(`INSERT is rejected — ${error.message} Nothing was changed.`, { stage: 'result', rows: [] });
|
|
169
|
+
return { events: drain(), rows: [], error };
|
|
170
|
+
}
|
|
171
|
+
throw e;
|
|
172
|
+
}
|
|
173
|
+
hooks.afterWrite?.(trees);
|
|
174
|
+
const rows = [...table.rows, inserted];
|
|
175
|
+
emit(`1 row inserted, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.writes} write${pool.stats.writes === 1 ? '' : 's'}. ${table.name} now has ${rows.length} row${rows.length === 1 ? '' : 's'}.`, { stage: 'result', rows: [inserted] });
|
|
176
|
+
return {
|
|
177
|
+
events: drain(),
|
|
178
|
+
rows: [inserted],
|
|
179
|
+
write: { kind: 'insert', table: table.name, insertedCount: 1, rows },
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
if (ast.kind === 'delete') {
|
|
183
|
+
emitDeleteParseEvents(emit, ast);
|
|
184
|
+
// Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
|
|
185
|
+
// the parse-stage narration above already showed the query exactly as written.
|
|
186
|
+
const resolved = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
|
|
187
|
+
if (!resolved.ok)
|
|
188
|
+
return { events: drain(), rows: [], error: resolved.error };
|
|
189
|
+
const deleteAst = { ...ast, ...(resolved.where ? { where: resolved.where } : {}) };
|
|
190
|
+
const plan = emitDeletePlanEvents(emit, deleteAst, table, options.rowsPerPage);
|
|
191
|
+
const pool = createBufferPool(pageCountFor(heap, trees), options, emit);
|
|
192
|
+
const { deleted } = executeDelete(plan, { emit, pool, locks, heap, trees, table, options });
|
|
193
|
+
hooks.afterWrite?.(trees);
|
|
194
|
+
const removed = new Set(deleted);
|
|
195
|
+
const rows = table.rows.filter((r) => !removed.has(r));
|
|
196
|
+
emit(`${deleted.length} row${deleted.length === 1 ? '' : 's'} deleted, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.evictions} eviction${pool.stats.evictions === 1 ? '' : 's'}. ${table.name} now has ${rows.length} row${rows.length === 1 ? '' : 's'}.`, { stage: 'result', rows: deleted });
|
|
197
|
+
return {
|
|
198
|
+
events: drain(),
|
|
199
|
+
rows: deleted,
|
|
200
|
+
write: { kind: 'delete', table: table.name, deletedCount: deleted.length, rows },
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
if (ast.kind === 'update') {
|
|
204
|
+
emitUpdateParseEvents(emit, ast);
|
|
205
|
+
// Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
|
|
206
|
+
// the parse-stage narration above already showed the query exactly as written.
|
|
207
|
+
const resolved = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
|
|
208
|
+
if (!resolved.ok)
|
|
209
|
+
return { events: drain(), rows: [], error: resolved.error };
|
|
210
|
+
const updateAst = { ...ast, ...(resolved.where ? { where: resolved.where } : {}) };
|
|
211
|
+
const plan = emitUpdatePlanEvents(emit, updateAst, table, options.rowsPerPage);
|
|
212
|
+
const pool = createBufferPool(pageCountFor(heap, trees), options, emit);
|
|
213
|
+
let changed;
|
|
214
|
+
try {
|
|
215
|
+
changed = executeUpdate(plan, ast.set.assignments, { emit, pool, locks, heap, trees, table, options });
|
|
216
|
+
}
|
|
217
|
+
catch (e) {
|
|
218
|
+
if (e instanceof UniqueViolation) {
|
|
219
|
+
const error = uniqueError(e, heap, ast.set.span);
|
|
220
|
+
emit(`UPDATE is rejected — ${error.message} Nothing was changed.`, { stage: 'result', rows: [] });
|
|
221
|
+
return { events: drain(), rows: [], error };
|
|
222
|
+
}
|
|
223
|
+
throw e;
|
|
224
|
+
}
|
|
225
|
+
const { before, after } = changed;
|
|
226
|
+
hooks.afterWrite?.(trees);
|
|
227
|
+
const replacements = new Map(before.map((row, i) => [row, after[i]]));
|
|
228
|
+
const rows = table.rows.map((r) => replacements.get(r) ?? r);
|
|
229
|
+
emit(`${after.length} row${after.length === 1 ? '' : 's'} updated, from ${pool.stats.misses} disk read${pool.stats.misses === 1 ? '' : 's'} and ${pool.stats.evictions} eviction${pool.stats.evictions === 1 ? '' : 's'}.`, { stage: 'result', rows: after });
|
|
230
|
+
return {
|
|
231
|
+
events: drain(),
|
|
232
|
+
rows: after,
|
|
233
|
+
write: { kind: 'update', table: table.name, updatedCount: after.length, rows },
|
|
234
|
+
};
|
|
235
|
+
}
|
|
236
|
+
// Always the query exactly as typed — `ast` may already be the cost-reordered version by this point (plan.md
|
|
237
|
+
// §25.4 C5 slice 3), but the parse stage narrates source text, never a planning decision; `parsedAst` never was.
|
|
238
|
+
emitParseEvents(emit, parsedAst.kind === 'select' ? parsedAst : ast);
|
|
239
|
+
// Any `IN (subquery)` in the WHERE clause (plan.md §25.4 C3 slice c) runs exactly once here, before planning —
|
|
240
|
+
// the parse-stage narration above already showed the query exactly as written.
|
|
241
|
+
const resolvedWhere = resolveWhereSubqueries(ast.where, sql, (s) => runQuery(s, db, options));
|
|
242
|
+
if (!resolvedWhere.ok)
|
|
243
|
+
return { events: drain(), rows: [], error: resolvedWhere.error };
|
|
244
|
+
const selectAst = { ...ast, ...(resolvedWhere.where ? { where: resolvedWhere.where } : {}) };
|
|
245
|
+
const plan = emitPlanEvents(emit, selectAst, table, options.rowsPerPage, options.aggregateStrategy ?? 'hash', joinTables, options.joinStrategy, writtenJoinOrder);
|
|
246
|
+
// Every joined table gets its own heap, its pages numbered right after whatever came before it — so all of them
|
|
247
|
+
// compete for the same buffer frames, the same way a Sort's spill pages do (plan.md §25.4 C3 slice b generalizes
|
|
248
|
+
// this from one join partner to a whole left-deep chain). Nested-loop, hash and sort-merge never consult any
|
|
249
|
+
// side's own index, so there is no B+Tree to build for a join partner unless that particular step runs
|
|
250
|
+
// index-nested-loop (plan.md §25.4 B4) — each one that does gets its one needed index built next, past every
|
|
251
|
+
// heap and every earlier probe index, so nothing collides.
|
|
252
|
+
const joinHeaps = {};
|
|
253
|
+
let pageBase = pageCountFor(heap, trees);
|
|
254
|
+
for (const j of joins) {
|
|
255
|
+
const jHeap = buildHeap(joinTables[j.table.name], options.rowsPerPage, pageBase);
|
|
256
|
+
joinHeaps[j.table.name] = jHeap;
|
|
257
|
+
pageBase += jHeap.pages.length;
|
|
258
|
+
}
|
|
259
|
+
const joinTrees = {};
|
|
260
|
+
for (const ij of findIndexNestedLoopJoins(plan)) {
|
|
261
|
+
if (ij.right.op !== 'IndexProbe')
|
|
262
|
+
continue; // unreachable: an index-nested-loop Join's own inner side is always a probe
|
|
263
|
+
const jTable = joinTables[ij.rightTable];
|
|
264
|
+
const probeIndex = findIndexSpec(jTable, ij.right.column);
|
|
265
|
+
if (!probeIndex)
|
|
266
|
+
continue; // unreachable: effectiveJoinAlgorithm already confirmed this index exists
|
|
267
|
+
const jTrees = buildIndexes(joinHeaps[ij.rightTable], [probeIndex], undefined, pageBase, options.indexBuild, jTable.clusteredKey);
|
|
268
|
+
joinTrees[ij.rightTable] = jTrees;
|
|
269
|
+
pageBase = pageCountFor(joinHeaps[ij.rightTable], jTrees);
|
|
270
|
+
}
|
|
271
|
+
const heapsEnd = pageBase;
|
|
272
|
+
// Every `sort-merge` Join in the chain needs its own two-block reservation — one per side, since each sorts
|
|
273
|
+
// independently — placed in chain order (the innermost join's scratch first, mirroring the order joins actually
|
|
274
|
+
// execute in) right after every table's heap and any probe index. A single join's worst-case left input is just
|
|
275
|
+
// the FROM table itself (`table.rows.length`, unchanged from before chains existed); a later step's worst case is
|
|
276
|
+
// the product of every earlier table's row count, since no join can ever produce more rows than that.
|
|
277
|
+
const sortMergeBases = new Map();
|
|
278
|
+
let worstCaseLeftRows = table.rows.length;
|
|
279
|
+
for (const j of joinsIn(plan)) {
|
|
280
|
+
const jTable = joinTables[j.rightTable];
|
|
281
|
+
if (j.algorithm === 'sort-merge') {
|
|
282
|
+
const leftSize = sortTempPageCount(worstCaseLeftRows, options.rowsPerPage, options.bufferFrames);
|
|
283
|
+
const leftBase = pageBase;
|
|
284
|
+
pageBase += leftSize;
|
|
285
|
+
const rightSize = sortTempPageCount(jTable.rows.length, options.rowsPerPage, options.bufferFrames);
|
|
286
|
+
const rightBase = pageBase;
|
|
287
|
+
pageBase += rightSize;
|
|
288
|
+
sortMergeBases.set(j, { leftBase, rightBase });
|
|
289
|
+
}
|
|
290
|
+
worstCaseLeftRows *= jTable.rows.length;
|
|
291
|
+
}
|
|
292
|
+
// A Sort's (or a SortAggregate's own internal sort's) spill needs its own page ids, reserved up front — sized
|
|
293
|
+
// against the whole table (the worst case; a WHERE clause can only shrink what actually reaches it) so the
|
|
294
|
+
// reservation is never too small to discover only at runtime. This and the sort-merge reservation above never
|
|
295
|
+
// both apply: ORDER BY/GROUP BY still can't combine with JOIN at all.
|
|
296
|
+
const sortTemp = needsSortSpill(plan)
|
|
297
|
+
? sortTempPageCount(table.rows.length, options.rowsPerPage, options.bufferFrames)
|
|
298
|
+
: pageBase - heapsEnd;
|
|
299
|
+
// A sort-based DISTINCT (plan.md §25.4 B2) sorts what the projection produced, which after a JOIN can be as many
|
|
300
|
+
// rows as every joined table's product (capped — these are teaching-sized tables) — and it gets its own block of
|
|
301
|
+
// scratch pages *after* whichever other sort the plan has, so no two sorts share a page id.
|
|
302
|
+
const distinctRows = joins.length > 0
|
|
303
|
+
? Math.min(joins.reduce((product, j) => product * joinTables[j.table.name].rows.length, table.rows.length), 20_000)
|
|
304
|
+
: table.rows.length;
|
|
305
|
+
const distinctSpill = findSortDistinct(plan)
|
|
306
|
+
? sortTempPageCount(distinctRows, options.rowsPerPage, options.bufferFrames)
|
|
307
|
+
: 0;
|
|
308
|
+
const distinctBase = heapsEnd + sortTemp;
|
|
309
|
+
const pool = createBufferPool(distinctBase + distinctSpill, options, emit);
|
|
310
|
+
const rows = execute(plan, {
|
|
311
|
+
emit,
|
|
312
|
+
pool,
|
|
313
|
+
locks,
|
|
314
|
+
heap,
|
|
315
|
+
trees,
|
|
316
|
+
table,
|
|
317
|
+
options,
|
|
318
|
+
distinctBase,
|
|
319
|
+
...(joins.length > 0
|
|
320
|
+
? {
|
|
321
|
+
joinPartners: Object.fromEntries(joins.map((j) => [
|
|
322
|
+
j.table.name,
|
|
323
|
+
{ table: joinTables[j.table.name], heap: joinHeaps[j.table.name], trees: joinTrees[j.table.name] ?? {} },
|
|
324
|
+
])),
|
|
325
|
+
}
|
|
326
|
+
: {}),
|
|
327
|
+
...(sortMergeBases.size > 0 ? { sortMergeBases } : {}),
|
|
328
|
+
});
|
|
329
|
+
return { events: drain(), rows };
|
|
330
|
+
}
|
|
331
|
+
function referencedColumns(ast) {
|
|
332
|
+
const refs = [];
|
|
333
|
+
const push = (ref) => refs.push({ name: ref.name, ...(ref.table === undefined ? {} : { table: ref.table }), span: ref.span });
|
|
334
|
+
if (ast.kind === 'select' && ast.select.kind === 'columns') {
|
|
335
|
+
for (const item of ast.select.items) {
|
|
336
|
+
if (item.kind === 'column')
|
|
337
|
+
push(item);
|
|
338
|
+
else if (item.kind === 'expression')
|
|
339
|
+
walkExpr(item.expr, (e) => { if (e.kind === 'column')
|
|
340
|
+
push(e); });
|
|
341
|
+
else if (item.arg.kind === 'column')
|
|
342
|
+
push(item.arg);
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
if (ast.kind !== 'insert' && ast.kind !== 'analyze' && ast.where) {
|
|
346
|
+
walkExpr(ast.where, (e) => {
|
|
347
|
+
if (e.kind === 'column')
|
|
348
|
+
push(e);
|
|
349
|
+
});
|
|
350
|
+
}
|
|
351
|
+
if (ast.kind === 'select') {
|
|
352
|
+
for (const join of ast.joins ?? []) {
|
|
353
|
+
if (join.on.left.kind === 'column')
|
|
354
|
+
push(join.on.left);
|
|
355
|
+
if (join.on.right.kind === 'column')
|
|
356
|
+
push(join.on.right);
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
if (ast.kind === 'select' && ast.groupBy)
|
|
360
|
+
refs.push(...ast.groupBy.columns);
|
|
361
|
+
if (ast.kind === 'select' && ast.having) {
|
|
362
|
+
// A HAVING predicate names grouped columns and aggregate arguments — each must exist.
|
|
363
|
+
walkExpr(ast.having, (e) => {
|
|
364
|
+
if (e.kind === 'column')
|
|
365
|
+
push(e);
|
|
366
|
+
else if (e.kind === 'aggregate' && e.arg.kind === 'column')
|
|
367
|
+
push(e.arg);
|
|
368
|
+
});
|
|
369
|
+
}
|
|
370
|
+
if (ast.kind === 'select' && ast.orderBy)
|
|
371
|
+
refs.push({ name: ast.orderBy.column, span: ast.orderBy.span });
|
|
372
|
+
if (ast.kind === 'insert' && ast.columns)
|
|
373
|
+
refs.push(...ast.columns.names);
|
|
374
|
+
if (ast.kind === 'update') {
|
|
375
|
+
refs.push(...ast.set.assignments.map((a) => ({ name: a.column, span: a.columnSpan })));
|
|
376
|
+
// A SET value can read columns (`SET n = n + 1`), and each must exist.
|
|
377
|
+
for (const a of ast.set.assignments)
|
|
378
|
+
walkExpr(a.value, (e) => { if (e.kind === 'column')
|
|
379
|
+
push(e); });
|
|
380
|
+
}
|
|
381
|
+
return refs;
|
|
382
|
+
}
|
|
383
|
+
/** Builds the new row: every table column defaults to `null`, then the given values overwrite it. */
|
|
384
|
+
function buildInsertRow(ast, table) {
|
|
385
|
+
const row = {};
|
|
386
|
+
for (const column of table.columns)
|
|
387
|
+
row[column] = null;
|
|
388
|
+
const names = ast.columns ? ast.columns.names.map((c) => c.name) : table.columns;
|
|
389
|
+
names.forEach((name, i) => {
|
|
390
|
+
row[name] = ast.values.items[i].value;
|
|
391
|
+
});
|
|
392
|
+
return row;
|
|
393
|
+
}
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 48 rows at 4 rows per page is 12 pages against 8 buffer frames, so a full
|
|
3
|
+
* sequential scan evicts. That contrast is the point: plan.md §9 wants the
|
|
4
|
+
* scan's eviction storm shown next to an index lookup that touches almost
|
|
5
|
+
* nothing. A table small enough to never evict would make the buffer pool —
|
|
6
|
+
* the product's best differentiator — look inert.
|
|
7
|
+
*/
|
|
8
|
+
const NAMES = [
|
|
9
|
+
'ada', 'alan', 'barbara', 'brendan', 'carmack', 'carol', 'dennis', 'dijkstra',
|
|
10
|
+
'donald', 'edsger', 'engelbart', 'fowler', 'frances', 'grace', 'guido', 'hamilton',
|
|
11
|
+
'hedy', 'hopper', 'iverson', 'james', 'jean', 'john', 'katherine', 'ken',
|
|
12
|
+
'knuth', 'lamport', 'larry', 'leslie', 'linus', 'lynn', 'margaret', 'mary',
|
|
13
|
+
'matz', 'niklaus', 'radia', 'richard', 'ritchie', 'rob', 'robert', 'shafi',
|
|
14
|
+
'stephen', 'thompson', 'tim', 'tony', 'torvalds', 'turing', 'vint', 'yukihiro',
|
|
15
|
+
];
|
|
16
|
+
function seedRows() {
|
|
17
|
+
return NAMES.map((name, i) => ({
|
|
18
|
+
id: i + 1,
|
|
19
|
+
name,
|
|
20
|
+
email: `${name}@example.com`,
|
|
21
|
+
age: 20 + ((i * 7) % 45),
|
|
22
|
+
}));
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The starting dataset. The user can edit these rows and re-run (plan.md §4).
|
|
26
|
+
*/
|
|
27
|
+
export function createSeedDatabase() {
|
|
28
|
+
return {
|
|
29
|
+
tables: {
|
|
30
|
+
users: {
|
|
31
|
+
name: 'users',
|
|
32
|
+
columns: ['id', 'name', 'email', 'age'],
|
|
33
|
+
rows: seedRows(),
|
|
34
|
+
indexedColumns: ['id'],
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
};
|
|
38
|
+
}
|
|
39
|
+
const AUTHOR_NAMES = ['ada', 'alan', 'grace', 'donald', 'barbara'];
|
|
40
|
+
// `authorId` is a real foreign key shape: two authors with several posts,
|
|
41
|
+
// one with exactly one, and one with none at all — the ordinary distribution
|
|
42
|
+
// a real join produces, not every row matching neatly. Small on purpose: v1
|
|
43
|
+
// JOIN is nested-loop only, so every row on one side rescans the *other*
|
|
44
|
+
// side whole — small tables keep that honest even before an index-aware join
|
|
45
|
+
// strategy exists to speed it up.
|
|
46
|
+
const POST_AUTHORS = [1, 1, 1, 2, 2, 3, 3, 3, 3, 4];
|
|
47
|
+
/**
|
|
48
|
+
* A small second table, joinable against a small first one (plan.md §22.2) —
|
|
49
|
+
* `authors`/`posts` rather than reusing the 48-row `users`, so a full,
|
|
50
|
+
* unfiltered join stays small enough to read as a trace. Not part of
|
|
51
|
+
* `datasets.ts`'s preset library: those are each one table by design (§22.7);
|
|
52
|
+
* this pair exists to give `/tools/planner` something to JOIN against.
|
|
53
|
+
*/
|
|
54
|
+
export function createJoinDemoDatabase() {
|
|
55
|
+
return {
|
|
56
|
+
tables: {
|
|
57
|
+
authors: {
|
|
58
|
+
name: 'authors',
|
|
59
|
+
columns: ['id', 'name'],
|
|
60
|
+
rows: AUTHOR_NAMES.map((name, i) => ({ id: i + 1, name })),
|
|
61
|
+
indexedColumns: ['id'],
|
|
62
|
+
},
|
|
63
|
+
posts: {
|
|
64
|
+
name: 'posts',
|
|
65
|
+
columns: ['id', 'authorId', 'title'],
|
|
66
|
+
rows: POST_AUTHORS.map((authorId, i) => ({
|
|
67
|
+
id: i + 1,
|
|
68
|
+
authorId,
|
|
69
|
+
title: `${AUTHOR_NAMES[authorId - 1]}'s post ${String(i + 1)}`,
|
|
70
|
+
})),
|
|
71
|
+
// Three indexes on purpose. `id` and `title` give `/tools/planner`'s index-selection rule a second
|
|
72
|
+
// candidate to weigh (plan.md §22.2); `authorId` is the foreign key, and it *repeats* — 4 distinct
|
|
73
|
+
// values among 10 rows — so `WHERE authorId = 1` walks an index that holds three entries for one value
|
|
74
|
+
// and returns all three rows (plan.md §25.4 B3). Until then a B+Tree here was unique-only, and
|
|
75
|
+
// indexing this column would have silently returned one row instead of three.
|
|
76
|
+
indexedColumns: ['id', 'title', 'authorId'],
|
|
77
|
+
},
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Four tables, shaped as a tree rather than a chain — `departments` is the hub, connecting to both `employees`
|
|
83
|
+
* and `projects`; `timesheets` hangs off `employees` alone (plan.md §25.4 C4, `/tools/join-order` + the
|
|
84
|
+
* Playground's own join-order note). Sized so join *order* has a real, measured answer rather than an obvious
|
|
85
|
+
* one: `timesheets` is the one genuinely large table (5,000 rows against the other three's tens or hundreds), so
|
|
86
|
+
* whichever join brings it in pays a real cost — but *when* that happens (how large the accumulated join is by
|
|
87
|
+
* the time it does) is what actually separates a good order from a bad one, not simply "big table first" or
|
|
88
|
+
* "big table last." Tuned empirically (the project's own "measure, don't guess" convention) rather than derived:
|
|
89
|
+
* `FROM departments JOIN projects ON departments.id = projects.deptId JOIN employees ON departments.id =
|
|
90
|
+
* employees.deptId JOIN timesheets ON employees.id = timesheets.employeeId` — a natural-looking "hub, then each
|
|
91
|
+
* branch" order — costs 5× the DP-optimal `departments → employees → timesheets → projects`, because visiting
|
|
92
|
+
* `projects` before drilling into `employees` → `timesheets` lets the *other* branch's join happen against an
|
|
93
|
+
* already-larger intermediate. Every foreign key is uniform (rows divide evenly across parents), so the only
|
|
94
|
+
* lever is order, not skew.
|
|
95
|
+
*/
|
|
96
|
+
export function createJoinOrderDemoDatabase() {
|
|
97
|
+
const DEPARTMENT_COUNT = 5;
|
|
98
|
+
const EMPLOYEES_PER_DEPARTMENT = 40;
|
|
99
|
+
const PROJECTS_PER_DEPARTMENT = 6;
|
|
100
|
+
const TIMESHEETS_PER_EMPLOYEE = 25;
|
|
101
|
+
const employeeCount = DEPARTMENT_COUNT * EMPLOYEES_PER_DEPARTMENT;
|
|
102
|
+
return {
|
|
103
|
+
tables: {
|
|
104
|
+
departments: {
|
|
105
|
+
name: 'departments',
|
|
106
|
+
columns: ['id', 'name'],
|
|
107
|
+
rows: Array.from({ length: DEPARTMENT_COUNT }, (_, i) => ({ id: i + 1, name: `dept-${String(i + 1)}` })),
|
|
108
|
+
indexedColumns: ['id'],
|
|
109
|
+
},
|
|
110
|
+
employees: {
|
|
111
|
+
name: 'employees',
|
|
112
|
+
columns: ['id', 'deptId', 'name'],
|
|
113
|
+
rows: Array.from({ length: employeeCount }, (_, i) => ({
|
|
114
|
+
id: i + 1,
|
|
115
|
+
deptId: Math.floor(i / EMPLOYEES_PER_DEPARTMENT) + 1,
|
|
116
|
+
name: `employee-${String(i + 1)}`,
|
|
117
|
+
})),
|
|
118
|
+
indexedColumns: ['id', 'deptId'],
|
|
119
|
+
},
|
|
120
|
+
projects: {
|
|
121
|
+
name: 'projects',
|
|
122
|
+
columns: ['id', 'deptId', 'name'],
|
|
123
|
+
rows: Array.from({ length: DEPARTMENT_COUNT * PROJECTS_PER_DEPARTMENT }, (_, i) => ({
|
|
124
|
+
id: i + 1,
|
|
125
|
+
deptId: Math.floor(i / PROJECTS_PER_DEPARTMENT) + 1,
|
|
126
|
+
name: `project-${String(i + 1)}`,
|
|
127
|
+
})),
|
|
128
|
+
indexedColumns: ['id', 'deptId'],
|
|
129
|
+
},
|
|
130
|
+
timesheets: {
|
|
131
|
+
name: 'timesheets',
|
|
132
|
+
columns: ['id', 'employeeId', 'hours'],
|
|
133
|
+
rows: Array.from({ length: employeeCount * TIMESHEETS_PER_EMPLOYEE }, (_, i) => ({
|
|
134
|
+
id: i + 1,
|
|
135
|
+
employeeId: Math.floor(i / TIMESHEETS_PER_EMPLOYEE) + 1,
|
|
136
|
+
hours: 1 + (i % 8),
|
|
137
|
+
})),
|
|
138
|
+
indexedColumns: ['id', 'employeeId'],
|
|
139
|
+
},
|
|
140
|
+
},
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* `createJoinDemoDatabase`'s pair plus a larger third table, for `/compare`'s JOIN mode (plan.md §25.4 B4). Ten
|
|
145
|
+
* posts on three pages cannot show what an index join is *for* — a probe of `posts` costs more than one scan
|
|
146
|
+
* of it. `comments` is the inner table that can: 120 rows on 30 pages, twelve per post and scattered across them
|
|
147
|
+
* (every tenth row is one post's), indexed on `postId` (a foreign key, so it repeats). One outer row probing that
|
|
148
|
+
* index reads a few tree pages and the pages holding its own twelve comments — well under the thirty a scan reads —
|
|
149
|
+
* while ten outer rows probing it cost more than one scan of the lot, which is the other half of the lesson.
|
|
150
|
+
*/
|
|
151
|
+
export function createJoinCompareDatabase() {
|
|
152
|
+
const base = createJoinDemoDatabase();
|
|
153
|
+
return {
|
|
154
|
+
tables: {
|
|
155
|
+
...base.tables,
|
|
156
|
+
comments: {
|
|
157
|
+
name: 'comments',
|
|
158
|
+
columns: ['id', 'postId', 'body'],
|
|
159
|
+
// `i * 7 mod 10` visits every post id in a scrambled order, so one post's comments sit on many pages.
|
|
160
|
+
rows: Array.from({ length: 120 }, (_, i) => ({ id: i + 1, postId: ((i * 7) % 10) + 1, body: `comment ${String(i + 1)}` })),
|
|
161
|
+
indexedColumns: ['postId'],
|
|
162
|
+
},
|
|
163
|
+
},
|
|
164
|
+
};
|
|
165
|
+
}
|