querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,267 @@
1
+ import { buildAggregatePlan, buildJoinPlan, buildPlan, distinctNodeFor, isAggregateQuery, projectListFor } from "./buildPlan.js";
2
+ import { estimatePlan, rootEstimate } from "./cost.js";
3
+ import { costOfOrder, joinGraphFromAst, runJoinOrderDP, writtenOrderFromAst } from "./joinOrder.js";
4
+ import { OPTIMIZER_RULES, leftmostPrefixNote, optimize } from "./optimize.js";
5
+ import { exprToSql, joinsIn, subqueryResolutionNotes, toPlanDisplay } from "./plan.js";
6
+ /**
7
+ * Emits the planner and optimizer stages, and returns the final plan for the
8
+ * executor to run. The plan is built bottom-up one operator at a time so the
9
+ * tree can be watched growing, the same way the AST was.
10
+ *
11
+ * Every emitted tree carries the cost model's per-node estimates (plan.md
12
+ * §22.2), which is also why `rowsPerPage` is threaded down: a scan's estimated
13
+ * page count depends on it. `joinTables` is every `JOIN`'d table's own data,
14
+ * keyed by name, needed only so the cost model can estimate that side of each
15
+ * `Join` too — `undefined` for every query without one.
16
+ */
17
+ /** Does any node of `plan` read through an index? */
18
+ function usesIndex(plan) {
19
+ if (plan.op === 'IndexScan' || plan.op === 'IndexRangeScan' || plan.op === 'IndexOnlyScan' || plan.op === 'IndexProbe')
20
+ return true;
21
+ if (plan.op === 'Join')
22
+ return usesIndex(plan.left) || usesIndex(plan.right);
23
+ return 'child' in plan ? usesIndex(plan.child) : false;
24
+ }
25
+ /** A join algorithm as a phrase for the "nothing to rewrite" note. */
26
+ function describeJoin(algorithm, outer, inner) {
27
+ switch (algorithm) {
28
+ case 'hash':
29
+ return 'a hash join over two full scans';
30
+ case 'sort-merge':
31
+ return 'a sort-merge join over two full scans';
32
+ case 'index-nested-loop':
33
+ return `an index nested-loop join — a full scan of ${outer}, then a B+Tree probe of ${inner} for each row`;
34
+ case 'nested-loop':
35
+ return 'a nested-loop join over two full scans';
36
+ }
37
+ }
38
+ export function emitPlanEvents(emit, ast, table, rowsPerPage, aggregateStrategy = 'hash', joinTables,
39
+ /**
40
+ * `undefined` (the shipped default, plan.md §25.4 C5 slice 2) means every `JOIN`'s algorithm is chosen by
41
+ * cost — `join-algorithm-selection` runs in the optimize stage below. An explicit value means the caller is
42
+ * forcing one, `/compare`'s and `/tools/planner`'s whole reason for setting this at all, and that rule is left
43
+ * out of `enabled` entirely so it can never override the forced choice. Either way the plan stage below still
44
+ * builds every step structurally as `nested-loop` first, the same cost-oblivious starting point a bare `SeqScan`
45
+ * already is for `index-selection` — only `structuralJoinStrategy` (the resolved, concrete value) drives it.
46
+ */
47
+ joinStrategy,
48
+ /**
49
+ * The table order this query was actually *typed* in — passed only when real execution (plan.md §25.4 C5
50
+ * slice 3) chose a different one, in which case `ast.from`/`ast.joins` below already reflect the chosen order,
51
+ * not the written one; `runQuery.ts`'s `applyCostOrder` is the only caller that ever sets this.
52
+ */
53
+ writtenJoinOrder) {
54
+ const structuralJoinStrategy = joinStrategy ?? 'nested-loop';
55
+ // Every table this query names, keyed by name — including `table` itself. A single join only ever needed the
56
+ // *other* one (the recursion's own `table` param always being the right one to check first), but a chain
57
+ // (plan.md §25.4 C3 slice b) re-derives that check at every nesting level against whatever table THAT level's
58
+ // own `Join` node concerns — which is `table` (the original FROM table) again once the recursion reaches the
59
+ // bottom of the chain — so `table.name` has to resolve through this same map at every depth, not just the top.
60
+ const otherTables = joinTables && Object.keys(joinTables).length > 0 ? { ...joinTables, [table.name]: table } : undefined;
61
+ const display = (plan, changed) => toPlanDisplay(plan, changed, estimatePlan(plan, table, rowsPerPage, otherTables));
62
+ const leftScan = { op: 'SeqScan', table: ast.from.name };
63
+ // Said up front, before the trace below starts naming a different FROM table than the one typed (plan.md
64
+ // §25.4 C5 slice 3): real execution reordered this chain because it was honestly cheaper to.
65
+ if (writtenJoinOrder && writtenJoinOrder.length >= 2 && joinTables) {
66
+ const allTables = { ...joinTables, [table.name]: table };
67
+ const { nodes, edges } = joinGraphFromAst(ast, allTables, rowsPerPage);
68
+ const chosenOrder = [ast.from.name, ...(ast.joins ?? []).map((j) => j.table.name)];
69
+ const writtenCost = costOfOrder(nodes, edges, writtenJoinOrder);
70
+ const chosenCost = costOfOrder(nodes, edges, chosenOrder);
71
+ const ratio = writtenCost.pages / Math.max(1, chosenCost.pages);
72
+ emit(`This query's JOINs run in a different order than written (plan.md §25.4 C5 slice 3): a cost-based join-order search chose \`${chosenOrder.join(' → ')}\` over the written \`${writtenJoinOrder.join(' → ')}\` — an estimated ${String(chosenCost.pages)} pages against ${String(writtenCost.pages)}, about ${ratio.toFixed(1)}× cheaper. Unlike the algorithm each step runs (plan.md §25.4 C5 slice 2), there is no "declines but stays" version of this: every left-deep order of the same tables returns the same rows, so the cheaper one always wins.`, {
73
+ stage: 'plan',
74
+ planTree: display(leftScan),
75
+ whyNot: {
76
+ chosen: { label: chosenOrder.join(' → '), estPages: chosenCost.pages },
77
+ runnerUp: { label: writtenJoinOrder.join(' → '), estPages: writtenCost.pages },
78
+ },
79
+ });
80
+ }
81
+ emit(`Start at the bottom: every plan begins by reading a table, so the base of the tree is a sequential scan of \`${ast.from.name}\`.`, { stage: 'plan', planTree: display(leftScan) });
82
+ let current = leftScan;
83
+ const allJoins = ast.joins ?? [];
84
+ for (let i = 0; i < allJoins.length; i++) {
85
+ const join = allJoins[i];
86
+ // Rebuilt from scratch each step (cheap — a handful of tables at most) rather than grown incrementally, so this
87
+ // stays the one place that knows how to fold a JOIN chain and never drifts from `buildJoinPlan` itself.
88
+ current = buildJoinPlan(allJoins.slice(0, i + 1), ast.from.name, structuralJoinStrategy, joinTables);
89
+ const j = current;
90
+ const algorithm = j.algorithm;
91
+ // Asked for an index join, but this step's inner table has no index on the join column: say so, and run the plain one.
92
+ const fallback = structuralJoinStrategy === 'index-nested-loop' && algorithm !== 'index-nested-loop'
93
+ ? `You asked for an index nested-loop join, but \`${join.table.name}\` has no index on \`${j.rightColumn}\` for it to probe — so this runs as an ordinary nested-loop. `
94
+ : '';
95
+ const mechanism = algorithm === 'index-nested-loop'
96
+ ? `for every row \`${j.leftTable}\` produces, it descends \`${j.rightTable}\`'s B+Tree on \`${j.rightColumn}\` with that row's \`${j.leftColumn}\` — a few page reads per outer row instead of a rescan of the whole inner table, which is why a small outer over a big indexed inner is where this algorithm wins`
97
+ : algorithm === 'hash'
98
+ ? `it builds an in-memory hash table over all of \`${j.rightTable}\`, keyed on \`${j.rightColumn}\`, once — then probes it with each \`${j.leftTable}\` row's \`${j.leftColumn}\`, reading \`${j.rightTable}\` exactly once regardless of how many rows \`${j.leftTable}\` has`
99
+ : algorithm === 'sort-merge'
100
+ ? `it sorts both \`${j.leftTable}\` and \`${j.rightTable}\` on the join key, then merges the two sorted lists in a single pass — no per-row rescan and no table held in memory, at the cost of sorting both sides first`
101
+ : `for every row \`${j.leftTable}\` produces, it rescans all of \`${j.rightTable}\` looking for a match on \`${j.leftColumn}\` = \`${j.rightColumn}\` — nested-loop, the simplest join algorithm, and the most expensive once neither side has an index to speed the inner scan up`;
102
+ const outer = join.joinType === 'left'
103
+ ? ` This is a LEFT JOIN: every row from \`${j.leftTable}\` survives even when nothing on \`${j.rightTable}\` matches it, with \`${j.rightTable}\`'s columns NULL.`
104
+ : '';
105
+ emit(`${fallback}JOIN adds ${algorithm === 'index-nested-loop' ? 'an index probe on' : 'a second scan of'} \`${j.rightTable}\` and a Join above ${i === 0 ? 'that' : 'the chain so far'}: ${mechanism}.${outer}`, { stage: 'plan', planTree: display(current) });
106
+ }
107
+ // A second opinion on join *order* (plan.md §25.4 C4), for two or more JOINs only — a single one has no other
108
+ // order to consider, and a chain `writtenJoinOrder` already reordered for real has nothing left to say here
109
+ // (the note above already said it). Otherwise informational: this never changes `current` or what actually
110
+ // executes, only what gets said about it — the same "estimate beside the real plan, never a decision" stance
111
+ // C1's stats and C2's q-error already take, and (plan.md §25.4 C5 slice 3) still the only stance a chain with a
112
+ // `LEFT JOIN` in it ever gets, since that is never reordered for real. `/tools/join-order` is the same search,
113
+ // with the whole subset lattice visible.
114
+ if (allJoins.length >= 2 && joinTables && !writtenJoinOrder) {
115
+ const allTables = { ...joinTables, [table.name]: table };
116
+ const { nodes, edges } = joinGraphFromAst(ast, allTables, rowsPerPage);
117
+ const dp = runJoinOrderDP(nodes, edges);
118
+ const written = writtenOrderFromAst(ast);
119
+ const writtenCost = costOfOrder(nodes, edges, written);
120
+ const ratio = writtenCost.pages / dp.winner.pages;
121
+ const hasLeftJoin = allJoins.some((j) => j.joinType === 'left');
122
+ emit(ratio >= 1.05
123
+ ? `A cost-based join-order search (plan.md §25.4 C4) would instead join these ${String(nodes.length)} tables as ${dp.winner.tables.join(' → ')}: an estimated ${String(dp.winner.pages)} pages, against the ${String(writtenCost.pages)} the order actually written above costs — about ${ratio.toFixed(1)}× cheaper.${hasLeftJoin
124
+ ? ' This chain is never reordered for real (plan.md §25.4 C5 slice 3): a LEFT JOIN is not commutative or associative in general, so the order above is what actually runs.'
125
+ : ' This query still runs in the order it was written; nothing here changes that.'}`
126
+ : `A cost-based join-order search (plan.md §25.4 C4) agrees with the order written above — among left-deep orderings of these ${String(nodes.length)} tables, this is already about as cheap as it gets.`, {
127
+ stage: 'plan',
128
+ planTree: display(current),
129
+ ...(ratio >= 1.05
130
+ ? {
131
+ whyNot: {
132
+ chosen: { label: written.join(' → '), estPages: writtenCost.pages },
133
+ runnerUp: { label: dp.winner.tables.join(' → '), estPages: dp.winner.pages },
134
+ },
135
+ }
136
+ : {}),
137
+ });
138
+ }
139
+ if (ast.where) {
140
+ current = { op: 'Filter', predicate: ast.where, child: current };
141
+ const subqueryNotes = subqueryResolutionNotes(ast.where);
142
+ emit(`The WHERE clause becomes a Filter sitting above the scan: \`${exprToSql(ast.where)}\`.${subqueryNotes.length > 0 ? ` ${subqueryNotes.map((n) => `${n[0].toUpperCase()}${n.slice(1)}.`).join(' ')}` : ''}`, { stage: 'plan', planTree: display(current) });
143
+ }
144
+ if (isAggregateQuery(ast)) {
145
+ current = buildAggregatePlan(ast, current, aggregateStrategy);
146
+ const groupBy = ast.groupBy?.columns.map((c) => c.name) ?? [];
147
+ const aggregateLabels = current.op === 'HashAggregate' || current.op === 'SortAggregate'
148
+ ? current.aggregates.map((a) => a.key).join(', ')
149
+ : '';
150
+ const nodeName = aggregateStrategy === 'sort' ? 'SortAggregate' : 'HashAggregate';
151
+ const mechanism = aggregateStrategy === 'sort'
152
+ ? `by sorting on ${groupBy.join(', ') || 'nothing'} first — spilling through the buffer pool exactly as ORDER BY's Sort does — then collapsing runs of equal keys in one pass`
153
+ : 'with a hash table keyed on the group';
154
+ emit(groupBy.length > 0
155
+ ? `GROUP BY becomes a ${nodeName} above that: rows are grouped by ${groupBy.join(', ')}${aggregateLabels ? `, computing ${aggregateLabels} per group` : ''}, ${mechanism}. Like Sort, it cannot produce a row until it has seen every one — the last group is not known to be complete until the input runs out.`
156
+ : `${aggregateLabels} becomes a ${nodeName} above that: with no GROUP BY, the whole table is one group, so it collapses every row into one${aggregateStrategy === 'sort' ? ', sorting first even though there is only ever one group to produce' : ''}. Like Sort, it cannot produce a row until it has seen all of them.`, { stage: 'plan', planTree: display(current) });
157
+ }
158
+ if (ast.having) {
159
+ current = { op: 'Having', predicate: ast.having, child: current };
160
+ emit(`HAVING becomes a Having filter above the aggregate: \`${exprToSql(ast.having)}\` is tested on each finished group, after its aggregates are computed. That is the difference from WHERE, which filters rows *before* they are grouped and so can never mention an aggregate.`, { stage: 'plan', planTree: display(current) });
161
+ }
162
+ if (ast.orderBy) {
163
+ current = {
164
+ op: 'Sort',
165
+ column: ast.orderBy.column,
166
+ direction: ast.orderBy.direction,
167
+ child: current,
168
+ };
169
+ emit(`ORDER BY becomes a Sort above that: every row, ordered by \`${ast.orderBy.column}\` ${ast.orderBy.direction === 'asc' ? 'ascending' : 'descending'}. Unlike everything below it, Sort cannot produce a row until it has seen all of them.`, { stage: 'plan', planTree: display(current) });
170
+ }
171
+ const columns = projectListFor(ast.select);
172
+ current = { op: 'Project', columns, child: current };
173
+ emit(columns.kind === 'star'
174
+ ? 'The SELECT list becomes a Project on top. `*` keeps every column, so it passes rows through unchanged.'
175
+ : columns.kind === 'columns'
176
+ ? `The SELECT list becomes a Project on top, keeping ${columns.names.join(', ')}.`
177
+ : `The SELECT list becomes a Project on top that *computes* each item for every row: ${columns.items
178
+ .map((i) => `\`${i.sql}\``)
179
+ .join(', ')}. Arithmetic on a NULL is NULL, and dividing by zero gives NULL.`, { stage: 'plan', planTree: display(current) });
180
+ if (ast.distinct) {
181
+ current = distinctNodeFor(ast, current, aggregateStrategy);
182
+ emit(aggregateStrategy === 'sort'
183
+ ? `DISTINCT becomes a SortDistinct above the Project: it sorts every projected row${ast.orderBy ? ` (by \`${ast.orderBy.column}\` first, so your ORDER BY survives)` : ''} — spilling through the buffer pool exactly as ORDER BY's Sort does — so equal rows sit next to each other, then keeps one of each run. Like Sort it cannot produce a row until it has seen all of them.`
184
+ : 'DISTINCT becomes a HashDistinct above the Project: it remembers every row it has passed in an in-memory hash set and drops a repeat the moment it sees one. It streams — the first new row comes out immediately, and a LIMIT above it can stop the scan early.', { stage: 'plan', planTree: display(current) });
185
+ }
186
+ if (ast.limit !== undefined) {
187
+ const count = ast.limit.value;
188
+ const offset = ast.limit.offset ?? 0;
189
+ current = { op: 'Limit', count, ...(offset > 0 ? { offset } : {}), child: current };
190
+ emit(offset > 0
191
+ ? `LIMIT ... OFFSET becomes a Limit on top: it discards the first ${plural(offset, 'row')} that come through, then passes ${plural(count, 'row')} and stops. The discarded rows are still produced by everything beneath — an OFFSET saves the client nothing, only the network.`
192
+ : `LIMIT becomes a Limit on top: it stops pulling rows once ${plural(count, 'row')} ${count === 1 ? 'has' : 'have'} come through — whatever is beneath it is never asked for the rest.`, { stage: 'plan', planTree: display(current) });
193
+ }
194
+ // The canonical plan is a direct translation of the query text. Everything
195
+ // below is the optimizer earning its keep.
196
+ const canonical = buildPlan(ast, aggregateStrategy, structuralJoinStrategy, joinTables);
197
+ let optimised = canonical;
198
+ let before = canonical;
199
+ // `join-algorithm-selection` is the one rule left out when the caller forced a specific `joinStrategy` — see
200
+ // this function's own parameter doc above.
201
+ const enabledRules = joinStrategy === undefined ? new Set(OPTIMIZER_RULES) : new Set(OPTIMIZER_RULES.filter((r) => r !== 'join-algorithm-selection'));
202
+ for (const step of optimize(canonical, table, rowsPerPage, otherTables ?? {}, enabledRules)) {
203
+ emit((step.rule === 'index-selection' ||
204
+ step.rule === 'range-index-selection' ||
205
+ step.rule === 'index-only-scan') &&
206
+ step.changed.size > 0
207
+ ? withCostJustification(step.label, step.plan)
208
+ : step.label, {
209
+ stage: 'optimize',
210
+ rule: step.rule,
211
+ before: display(before),
212
+ after: display(step.plan, step.changed),
213
+ ...(step.whyNot ? { whyNot: step.whyNot } : {}),
214
+ }, step.prompt ? { predictable: step.prompt } : undefined);
215
+ before = step.plan;
216
+ optimised = step.plan;
217
+ }
218
+ if (optimised === canonical) {
219
+ const rows = rootEstimate(canonical, table, rowsPerPage, otherTables).estRows;
220
+ const chainJoins = joinsIn(canonical);
221
+ const note = chainJoins.length === 1
222
+ ? `No rewrite rule applies to this JOIN — ${describeJoin(chainJoins[0].algorithm, chainJoins[0].leftTable, chainJoins[0].rightTable)} is what runs.`
223
+ : chainJoins.length > 1
224
+ ? `No rewrite rule applies to this chain of ${String(chainJoins.length)} JOINs — ${chainJoins
225
+ .map((j) => `\`${j.leftTable}\` × \`${j.rightTable}\` runs as ${describeJoin(j.algorithm, j.leftTable, j.rightTable)}`)
226
+ .join('; ')}.`
227
+ : ast.where
228
+ ? `The cost model estimates about ${plural(rows, 'row')} out of ${String(table.rows.length)}, and no access path beats reading the table.${leftmostPrefixNote(table, ast.where) ? ` ${leftmostPrefixNote(table, ast.where)}` : ''}`
229
+ : 'With no WHERE there is nothing to narrow the scan — every row is returned.';
230
+ emit(`No rewrite rule fires on this plan — the optimizer leaves it exactly as written. ${note}`, {
231
+ stage: 'optimize',
232
+ rule: 'none',
233
+ before: display(canonical),
234
+ after: display(canonical),
235
+ });
236
+ }
237
+ // Rules fired, but none of them could use an index the table has: if that is because a composite index cannot be
238
+ // entered, say so here — the "nothing fired" note above never runs once, say, predicate pushdown has.
239
+ if (optimised !== canonical && allJoins.length === 0 && !usesIndex(optimised)) {
240
+ const unusable = leftmostPrefixNote(table, ast.where);
241
+ if (unusable) {
242
+ emit(`No index is used. ${unusable}`, {
243
+ stage: 'optimize',
244
+ rule: 'none',
245
+ before: display(optimised),
246
+ after: display(optimised),
247
+ });
248
+ }
249
+ }
250
+ return optimised;
251
+ /**
252
+ * Appends the scan-vs-index page estimate that makes an accepted rewrite's cost visible, not just asserted. Only
253
+ * ever called for a step that actually changed the plan: `index-selection` and `range-index-selection` already
254
+ * decline anything that would not be cheaper (plan.md §25.4 C5 — see their own doc comments in `optimize.ts`), and
255
+ * `index-only-scan` only ever narrows an `IndexScan` the gate already accepted, never chosen on its own, so it can
256
+ * only be cheaper still. The comparison below is therefore always a confirmation, never a surprise — a second,
257
+ * independent read of the same numbers `optimize.ts`'s own gate already compared, not a competing judgment.
258
+ */
259
+ function withCostJustification(label, chosen) {
260
+ const scanPages = Math.max(1, Math.ceil(table.rows.length / rowsPerPage));
261
+ const rows = rootEstimate(canonical, table, rowsPerPage, otherTables).estRows;
262
+ const indexPages = rootEstimate(chosen, table, rowsPerPage, otherTables).estPages;
263
+ const facts = `the sequential scan reads all ${String(scanPages)} pages to return about ${plural(rows, 'row')}; the index lookup reads about ${String(indexPages)}`;
264
+ return `${label} Estimated cost agrees: ${facts}.`;
265
+ }
266
+ }
267
+ const plural = (n, noun) => `${String(n)} ${noun}${n === 1 ? '' : 's'}`;
@@ -0,0 +1,57 @@
1
+ import { buildDeletePlan } from "./buildPlan.js";
2
+ import { estimatePlan } from "./cost.js";
3
+ import { optimize, WRITE_PATH_RULES } from "./optimize.js";
4
+ import { exprToSql, subqueryResolutionNotes, toPlanDisplay } from "./plan.js";
5
+ /** Strips the throwaway `Project *` wrapper `emitDeletePlanEvents` builds around a delete's scan. */
6
+ function unwrap(plan) {
7
+ return plan.op === 'Project' ? plan.child : plan;
8
+ }
9
+ /**
10
+ * Emits the plan/optimize stages for a `DELETE`, and returns the final access
11
+ * plan (`SeqScan` or `IndexScan`, whichever the optimizer chose) the executor
12
+ * uses to find the rows to remove.
13
+ *
14
+ * Reuses the same rules a `SELECT` uses — they are written against
15
+ * `Project → …`, so the scan is wrapped in a throwaway `Project *` for the
16
+ * optimizer's sake, and every tree shown to the reader has that wrapper
17
+ * stripped again: nothing about a `DELETE` cares what columns a `Project`
18
+ * would keep. `WRITE_PATH_RULES` excludes `range-index-selection` specifically
19
+ * — see its own doc comment in `optimize.ts` — so the plan this returns is
20
+ * always a `SeqScan` or a single-row `IndexScan`, the two shapes
21
+ * `executeDelete` (`exec/delete.ts`) knows how to run.
22
+ */
23
+ export function emitDeletePlanEvents(emit, ast, table, rowsPerPage) {
24
+ const display = (plan, changed) => toPlanDisplay(unwrap(plan), changed, estimatePlan(unwrap(plan), table, rowsPerPage));
25
+ const scan = { op: 'SeqScan', table: ast.from.name };
26
+ emit(`Start at the bottom: every plan begins by reading a table, so the base of the tree is a sequential scan of \`${ast.from.name}\`.`, { stage: 'plan', planTree: display(scan) });
27
+ let current = scan;
28
+ if (ast.where) {
29
+ current = { op: 'Filter', predicate: ast.where, child: current };
30
+ const subqueryNotes = subqueryResolutionNotes(ast.where);
31
+ emit(`The WHERE clause becomes a Filter sitting above the scan: \`${exprToSql(ast.where)}\` — the rows it passes are the rows DELETE removes.${subqueryNotes.length > 0 ? ` ${subqueryNotes.map((n) => `${n[0].toUpperCase()}${n.slice(1)}.`).join(' ')}` : ''}`, { stage: 'plan', planTree: display(current) });
32
+ }
33
+ const canonical = buildDeletePlan(ast);
34
+ const wrapped = { op: 'Project', columns: { kind: 'star' }, child: canonical };
35
+ let optimisedWrapped = wrapped;
36
+ let before = wrapped;
37
+ for (const step of optimize(wrapped, table, rowsPerPage, {}, WRITE_PATH_RULES)) {
38
+ emit(step.label, {
39
+ stage: 'optimize',
40
+ rule: step.rule,
41
+ before: display(before),
42
+ after: display(step.plan, step.changed),
43
+ ...(step.whyNot ? { whyNot: step.whyNot } : {}),
44
+ }, step.prompt ? { predictable: step.prompt } : undefined);
45
+ before = step.plan;
46
+ optimisedWrapped = step.plan;
47
+ }
48
+ if (optimisedWrapped === wrapped) {
49
+ emit('No rewrite rule fires on this plan — the optimizer leaves it exactly as written.', {
50
+ stage: 'optimize',
51
+ rule: 'none',
52
+ before: display(wrapped),
53
+ after: display(wrapped),
54
+ });
55
+ }
56
+ return unwrap(optimisedWrapped);
57
+ }
@@ -0,0 +1,51 @@
1
+ import { buildUpdatePlan } from "./buildPlan.js";
2
+ import { estimatePlan } from "./cost.js";
3
+ import { optimize, WRITE_PATH_RULES } from "./optimize.js";
4
+ import { exprToSql, subqueryResolutionNotes, toPlanDisplay } from "./plan.js";
5
+ /** Strips the throwaway `Project *` wrapper `emitUpdatePlanEvents` builds around an update's scan. */
6
+ function unwrap(plan) {
7
+ return plan.op === 'Project' ? plan.child : plan;
8
+ }
9
+ /**
10
+ * Emits the plan/optimize stages for an `UPDATE`, and returns the final
11
+ * access plan (`SeqScan` or `IndexScan`) the executor uses to find the rows
12
+ * to modify — the mirror of `emitDeletePlanEvents`, since only the WHERE
13
+ * clause is planned; the SET clause has nothing for a plan or an optimizer
14
+ * rule to act on. Uses `WRITE_PATH_RULES` for the same reason
15
+ * `emitDeletePlanEvents` does — see its own doc comment.
16
+ */
17
+ export function emitUpdatePlanEvents(emit, ast, table, rowsPerPage) {
18
+ const display = (plan, changed) => toPlanDisplay(unwrap(plan), changed, estimatePlan(unwrap(plan), table, rowsPerPage));
19
+ const scan = { op: 'SeqScan', table: ast.table.name };
20
+ emit(`Start at the bottom: every plan begins by reading a table, so the base of the tree is a sequential scan of \`${ast.table.name}\`.`, { stage: 'plan', planTree: display(scan) });
21
+ let current = scan;
22
+ if (ast.where) {
23
+ current = { op: 'Filter', predicate: ast.where, child: current };
24
+ const subqueryNotes = subqueryResolutionNotes(ast.where);
25
+ emit(`The WHERE clause becomes a Filter sitting above the scan: \`${exprToSql(ast.where)}\` — the rows it passes are the rows UPDATE modifies.${subqueryNotes.length > 0 ? ` ${subqueryNotes.map((n) => `${n[0].toUpperCase()}${n.slice(1)}.`).join(' ')}` : ''}`, { stage: 'plan', planTree: display(current) });
26
+ }
27
+ const canonical = buildUpdatePlan(ast);
28
+ const wrapped = { op: 'Project', columns: { kind: 'star' }, child: canonical };
29
+ let optimisedWrapped = wrapped;
30
+ let before = wrapped;
31
+ for (const step of optimize(wrapped, table, rowsPerPage, {}, WRITE_PATH_RULES)) {
32
+ emit(step.label, {
33
+ stage: 'optimize',
34
+ rule: step.rule,
35
+ before: display(before),
36
+ after: display(step.plan, step.changed),
37
+ ...(step.whyNot ? { whyNot: step.whyNot } : {}),
38
+ }, step.prompt ? { predictable: step.prompt } : undefined);
39
+ before = step.plan;
40
+ optimisedWrapped = step.plan;
41
+ }
42
+ if (optimisedWrapped === wrapped) {
43
+ emit('No rewrite rule fires on this plan — the optimizer leaves it exactly as written.', {
44
+ stage: 'optimize',
45
+ rule: 'none',
46
+ before: display(wrapped),
47
+ after: display(wrapped),
48
+ });
49
+ }
50
+ return unwrap(optimisedWrapped);
51
+ }
@@ -0,0 +1,8 @@
1
+ export { buildAggregatePlan, buildDeletePlan, buildJoinPlan, distinctNodeFor, effectiveJoinAlgorithm, buildPlan, buildUpdatePlan, isAggregateQuery, } from "./buildPlan.js";
2
+ export { boundedRangeSelectivity, estimateIndexLevels, estimatePlan, ndistinct, NO_STATS_EQ_SELECTIVITY, OPEN_RANGE_SELECTIVITY, rootEstimate, selectivityOf, } from "./cost.js";
3
+ export { emitPlanEvents } from "./emit.js";
4
+ export { emitDeletePlanEvents } from "./emitDelete.js";
5
+ export { emitUpdatePlanEvents } from "./emitUpdate.js";
6
+ export { costOfOrder, isReorderableChain, joinGraphFromAst, rebuildJoinsForOrder, reorderJoinsIfCheaper, runJoinOrderDP, writtenOrderFromAst, } from "./joinOrder.js";
7
+ export { optimize, OPTIMIZER_RULES, scanPredicate, WRITE_PATH_RULES } from "./optimize.js";
8
+ export { childOf, conjoin, conjuncts, equalityOn, exprToSql, findIndexNestedLoopJoins, findSort, findSortDistinct, findSortMergeJoins, formatRangeBounds, indexLabel, joinsIn, lookupKeysOf, lookupText, needsSortSpill, rangeOn, toPlanDisplay, } from "./plan.js";
@@ -0,0 +1,252 @@
1
+ /**
2
+ * Selinger-style join ordering by dynamic programming (plan.md §25.4 C4), over the **subset lattice**: for `n`
3
+ * tables there are `2^n - 1` non-empty subsets, and the cheapest way to join every table in a subset is built up
4
+ * from the cheapest way to join every smaller subset one table down — bottom-up, smallest subsets first, exactly
5
+ * the textbook System R algorithm (Selinger et al., 1979).
6
+ *
7
+ * Restricted to **left-deep** orderings on purpose: the engine's own `Join` operator (plan.md §25.4 C3 slice b)
8
+ * only ever executes a left-deep chain, so a bushy plan this module might otherwise consider is one the engine
9
+ * could never actually run. That restriction is also what keeps this simple — a subset's own best plan is just
10
+ * "which table was added last," not "which two smaller subsets were merged," so there is no need to search which
11
+ * way to *split* a subset, only which single table to peel off it.
12
+ *
13
+ * A **cross join is never forbidden**, only costed honestly: when two tables share no join predicate, joining
14
+ * them anyway (`rows(a) × rows(b)`, no selectivity) is a real, always-available option — it just loses the cost
15
+ * comparison so badly that the DP never picks it unless the graph genuinely offers nothing better. Nothing here
16
+ * needs to reason about the join graph's connectivity to avoid a cross join; the real cost already does that.
17
+ *
18
+ * The cost formula mirrors `cost.ts`'s own `Join` case exactly (nested-loop, the engine's default and the one
19
+ * algorithm every join order is meaningful for): `pages = left.pages + left.rows × right.pages`, and
20
+ * `rows = left.rows × right.rows / ndistinct(right's join column)` — see `cost.ts:358-426` for the source this
21
+ * is kept consistent with. Ordering is deliberately scored on **page cost**, not row count: row count alone
22
+ * cannot tell two orders with the same output size but very different rescan cost apart.
23
+ *
24
+ * Visualization-only when it was built (plan.md §25.4 C4's own scoping decision) — `/tools/join-order` and the
25
+ * Playground's informational note only ever *showed* what this function would have chosen. **Plan.md §25.4 C5
26
+ * slice 3 changes that for one well-scoped case**: `runQuery.ts` now actually reorders a chain of 2+ `JOIN`s
27
+ * before running it, *when every one is a plain inner join* — `reorderJoinsIfCheaper` below, and
28
+ * `rebuildJoinsForOrder`, which it's built on. A chain with any `LEFT JOIN` is never reordered: `LEFT JOIN` is
29
+ * not commutative or associative in general (`A LEFT JOIN B` keeps every `A` row regardless of `B`; flipping it,
30
+ * or moving another join between them, can change which rows get NULL-padded and when), and generalizing this
31
+ * DP to respect that would mean tracking a partial order among the steps, not just their cost — a materially
32
+ * different, harder problem this module does not attempt. For an eligible chain, this is the real, final order,
33
+ * not a second opinion — `/tools/join-order` and the note both still exist, but only ever have something to
34
+ * *show* (rather than something that already ran) when a `LEFT JOIN` keeps a chain out of scope.
35
+ */
36
+ import { ndistinct } from "./cost.js";
37
+ const clampRows = (value) => Math.max(0, Math.round(value));
38
+ const popcount = (mask) => {
39
+ let count = 0;
40
+ for (let m = mask; m !== 0; m >>>= 1)
41
+ count += m & 1;
42
+ return count;
43
+ };
44
+ /**
45
+ * Runs the DP. `nodes` must be non-empty; a single table has nothing to order and returns immediately with one
46
+ * singleton subset. Capped implicitly by `2^nodes.length` states — teaching-sized graphs only (plan.md's own
47
+ * datasets never approach the twenty-ish tables where this would start to matter).
48
+ */
49
+ export function runJoinOrderDP(nodes, edges) {
50
+ const n = nodes.length;
51
+ if (n === 0)
52
+ throw new Error('unreachable: runJoinOrderDP needs at least one table');
53
+ const nodeNames = nodes.map((t) => t.name);
54
+ const indexOf = new Map(nodeNames.map((name, i) => [name, i]));
55
+ // An edge is undirected for lookup purposes — keyed by the pair of bit indices, lowest first.
56
+ const edgeFor = new Map();
57
+ for (const e of edges) {
58
+ const ia = indexOf.get(e.a);
59
+ const ib = indexOf.get(e.b);
60
+ if (ia === undefined || ib === undefined)
61
+ continue; // an edge naming a table outside `nodes` is simply unusable
62
+ edgeFor.set(`${String(Math.min(ia, ib))}-${String(Math.max(ia, ib))}`, e);
63
+ }
64
+ const subsets = new Map();
65
+ for (let i = 0; i < n; i++) {
66
+ const t = nodes[i];
67
+ const mask = 1 << i;
68
+ subsets.set(mask, { mask, tables: [t.name], rows: t.rows, pages: t.pages, candidates: [] });
69
+ }
70
+ const masksBySize = Array.from({ length: n + 1 }, () => []);
71
+ for (let mask = 1; mask < 1 << n; mask++)
72
+ masksBySize[popcount(mask)].push(mask);
73
+ for (let size = 2; size <= n; size++) {
74
+ for (const mask of masksBySize[size]) {
75
+ const candidates = [];
76
+ for (let i = 0; i < n; i++) {
77
+ const bit = 1 << i;
78
+ if (!(mask & bit))
79
+ continue;
80
+ const prevMask = mask & ~bit;
81
+ const prev = subsets.get(prevMask);
82
+ if (!prev)
83
+ continue; // unreachable: every smaller mask is filled before this size runs
84
+ const t = nodes[i];
85
+ // The cheapest real join predicate connecting `t` to anything already in `prevMask` — there may be
86
+ // several (a star-shaped graph can have `t` adjacent to more than one table already joined in).
87
+ let viaEdge = null;
88
+ for (let j = 0; j < n; j++) {
89
+ if (!(prevMask & (1 << j)))
90
+ continue;
91
+ const edge = edgeFor.get(`${String(Math.min(i, j))}-${String(Math.max(i, j))}`);
92
+ if (!edge)
93
+ continue;
94
+ const tDistinct = edge.a === t.name ? edge.aNdistinct : edge.bNdistinct;
95
+ const rows = clampRows((prev.rows * t.rows) / Math.max(1, tDistinct));
96
+ if (!viaEdge || rows < viaEdge.rows)
97
+ viaEdge = { rows };
98
+ }
99
+ const rows = viaEdge ? viaEdge.rows : clampRows(prev.rows * t.rows);
100
+ const pages = prev.pages + prev.rows * t.pages;
101
+ candidates.push({ addedTable: t.name, viaEdge: viaEdge !== null, rows, pages });
102
+ }
103
+ candidates.sort((a, b) => a.pages - b.pages);
104
+ const winner = candidates[0];
105
+ const winnerBit = 1 << indexOf.get(winner.addedTable);
106
+ const prevPlan = subsets.get(mask & ~winnerBit);
107
+ subsets.set(mask, {
108
+ mask,
109
+ tables: [...prevPlan.tables, winner.addedTable],
110
+ rows: winner.rows,
111
+ pages: winner.pages,
112
+ candidates,
113
+ });
114
+ }
115
+ }
116
+ const fullMask = (1 << n) - 1;
117
+ return { nodeNames, subsets, winner: subsets.get(fullMask) };
118
+ }
119
+ /**
120
+ * The same cost formula `runJoinOrderDP` searches with, applied to one *specific* order instead of searching for
121
+ * the cheapest one — how much a query's own, as-written join order actually costs, for comparison against the
122
+ * DP's own suggestion. `order` must name every one of `nodes` exactly once; the first is the base scan.
123
+ */
124
+ export function costOfOrder(nodes, edges, order) {
125
+ const byName = new Map(nodes.map((t) => [t.name, t]));
126
+ const edgeFor = new Map();
127
+ for (const e of edges) {
128
+ edgeFor.set(`${e.a}-${e.b}`, e);
129
+ edgeFor.set(`${e.b}-${e.a}`, e);
130
+ }
131
+ const first = byName.get(order[0]);
132
+ let rows = first.rows;
133
+ let pages = first.pages;
134
+ const soFar = [order[0]];
135
+ for (let k = 1; k < order.length; k++) {
136
+ const t = byName.get(order[k]);
137
+ let bestRows = null;
138
+ for (const prevName of soFar) {
139
+ const edge = edgeFor.get(`${prevName}-${t.name}`);
140
+ if (!edge)
141
+ continue;
142
+ const tDistinct = edge.a === t.name ? edge.aNdistinct : edge.bNdistinct;
143
+ const candidateRows = clampRows((rows * t.rows) / Math.max(1, tDistinct));
144
+ if (bestRows === null || candidateRows < bestRows)
145
+ bestRows = candidateRows;
146
+ }
147
+ pages = pages + rows * t.pages;
148
+ rows = bestRows ?? clampRows(rows * t.rows);
149
+ soFar.push(t.name);
150
+ }
151
+ return { rows, pages };
152
+ }
153
+ /**
154
+ * Builds the join graph `runJoinOrderDP` needs straight from a parsed, already-resolved multi-join query — the
155
+ * same table/column resolution `buildJoinPlan` (`buildPlan.ts`) already does per step, just collecting it into a
156
+ * graph instead of folding it into a left-deep `Plan`. `tables` must have an entry for `ast.from.name` and every
157
+ * `ast.joins` table name (`runQuery.ts` already resolves exactly this map before planning).
158
+ */
159
+ export function joinGraphFromAst(ast, tables, rowsPerPage) {
160
+ const heapPages = (t) => Math.max(1, Math.ceil(t.rows.length / rowsPerPage));
161
+ const nodeFor = (name) => {
162
+ const t = tables[name];
163
+ return { name, rows: t.rows.length, pages: heapPages(t) };
164
+ };
165
+ const nodes = [nodeFor(ast.from.name)];
166
+ const edges = [];
167
+ for (const join of ast.joins ?? []) {
168
+ const { left, right } = join.on;
169
+ if (left.kind !== 'column' || right.kind !== 'column')
170
+ continue; // unreachable: checkJoinOn guarantees this
171
+ const newTable = join.table.name;
172
+ // checkJoinChain already guarantees exactly one side names newTable and the other a table already in scope.
173
+ const newTableIsLeftOperand = left.table === newTable;
174
+ const bColumn = newTableIsLeftOperand ? left.name : right.name;
175
+ const aColumn = newTableIsLeftOperand ? right.name : left.name;
176
+ const aTable = (newTableIsLeftOperand ? right.table : left.table);
177
+ nodes.push(nodeFor(newTable));
178
+ edges.push({
179
+ a: aTable,
180
+ aColumn,
181
+ aNdistinct: ndistinct(tables[aTable], aColumn),
182
+ b: newTable,
183
+ bColumn,
184
+ bNdistinct: ndistinct(tables[newTable], bColumn),
185
+ });
186
+ }
187
+ return { nodes, edges };
188
+ }
189
+ /** The table order a multi-join query was actually *written* in — `FROM`'s table, then each `JOIN` in source order. */
190
+ export function writtenOrderFromAst(ast) {
191
+ return [ast.from.name, ...(ast.joins ?? []).map((j) => j.table.name)];
192
+ }
193
+ /** A chain real execution may reorder (plan.md §25.4 C5 slice 3): 2+ `JOIN`s, none of them `LEFT`. */
194
+ export function isReorderableChain(ast) {
195
+ const joins = ast.joins ?? [];
196
+ return joins.length >= 2 && joins.every((j) => j.joinType !== 'left');
197
+ }
198
+ /**
199
+ * Rebuilds a valid left-deep `{from, joins}` pair for `order` — normally the DP's own winning order — reusing
200
+ * every original `JoinClause` exactly as parsed (same `on` expression, same span), just reassigning which table
201
+ * each one is considered to *introduce*. Sound because the graph `joinGraphFromAst` builds is always a spanning
202
+ * tree, one edge per non-`FROM` table, drawn straight from that table's own `ON` clause: `buildJoinPlan` only
203
+ * ever needs, for each step, *some* earlier-known table on the other side of a real equality — never that it be
204
+ * the immediately preceding one (`leftJoinValue`/`leftJoinKeyOf` already resolve a qualified reference against
205
+ * the whole chain so far, plan.md §25.4 C3 slice b) — so swapping which side of an unchanged predicate counts as
206
+ * "newly introduced" is all a reordering ever needs to do. `edges[i]` is `(ast.joins ?? [])[i]`'s own edge,
207
+ * the exact parallel `joinGraphFromAst` builds them in.
208
+ *
209
+ * Returns `null` if some step in `order` has no real edge back to anything earlier in it — this engine's grammar
210
+ * has no way to express a `JOIN` with no `ON` clause, so an order that would need a cross join can never be built
211
+ * here. The DP's own cost-minimizing search should never actually choose one on a connected graph (routing
212
+ * through the real tree is always cheaper), so this is a defensive check, not an expected outcome.
213
+ */
214
+ export function rebuildJoinsForOrder(ast, edges, order) {
215
+ const joinList = ast.joins ?? [];
216
+ const edgeClause = new Map(edges.map((e, i) => [e, joinList[i]]));
217
+ const edgesByTable = new Map();
218
+ for (const e of edges) {
219
+ edgesByTable.set(e.a, [...(edgesByTable.get(e.a) ?? []), e]);
220
+ edgesByTable.set(e.b, [...(edgesByTable.get(e.b) ?? []), e]);
221
+ }
222
+ const refFor = (name) => (name === ast.from.name ? ast.from : edgeClause.get(edgesByTable.get(name)[0]).table);
223
+ // ^ any of `name`'s own edges names it the same way (every original clause's `table` is that same table's own
224
+ // ref, regardless of which edge produced it), so the first is as good as any for recovering its `TableRef`.
225
+ const visited = new Set([order[0]]);
226
+ const joins = [];
227
+ for (let k = 1; k < order.length; k++) {
228
+ const name = order[k];
229
+ const edge = (edgesByTable.get(name) ?? []).find((e) => visited.has(e.a === name ? e.b : e.a));
230
+ if (!edge)
231
+ return null;
232
+ joins.push({ ...edgeClause.get(edge), table: refFor(name) });
233
+ visited.add(name);
234
+ }
235
+ return { from: refFor(order[0]), joins };
236
+ }
237
+ /**
238
+ * The one entry point `runQuery.ts` needs: for an eligible chain (`isReorderableChain`) whose cost-optimal order
239
+ * differs from how it was written, an `{from, joins}` pair to run instead — `null` when reordering does not apply
240
+ * (a `LEFT JOIN` somewhere, fewer than two `JOIN`s) or would not actually change anything (the written order is
241
+ * already the DP's own answer, or — defensively — `rebuildJoinsForOrder` found no real edge for it).
242
+ */
243
+ export function reorderJoinsIfCheaper(ast, tables, rowsPerPage) {
244
+ if (!isReorderableChain(ast))
245
+ return null;
246
+ const { nodes, edges } = joinGraphFromAst(ast, tables, rowsPerPage);
247
+ const winner = runJoinOrderDP(nodes, edges).winner;
248
+ const written = writtenOrderFromAst(ast);
249
+ if (winner.tables.join('\u0000') === written.join('\u0000'))
250
+ return null;
251
+ return rebuildJoinsForOrder(ast, edges, winner.tables);
252
+ }