querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,906 @@
1
+ /**
2
+ * Rule-based rewrites. Most rules still consult no cost model at all: constant folding and predicate pushdown
3
+ * fire on a structural match alone, and never lose to a scan or another shape. Two kinds of rule are the
4
+ * exception, both plan.md §25.4 C5. `index-selection`/`range-index-selection` (C5 slice 1) pick a candidate by
5
+ * shape first, then decline it if its own estimated cost turns out not to beat the scan it would replace — a
6
+ * candidate that matches the predicate can still lose. `join-algorithm-selection` (C5 slice 2) has no shape step
7
+ * at all: every applicable algorithm is costed and the cheapest wins outright, the same as a real optimizer's
8
+ * plan enumeration. The cost *estimates* behind every one of these live in `cost.ts` (`rootEstimate`, the same
9
+ * formula `EXPLAIN` shows) — never a number invented for the comparison alone (§12).
10
+ *
11
+ * Rules run in order, each on the output of the last, and each produces one
12
+ * `optimize` event carrying the before and after trees.
13
+ */
14
+ import { indexSelectionPrompt, leftmostPrefixPrompt } from "../predict.js";
15
+ import { describeKeys as describeKeysText } from "../index/lookup.js";
16
+ import { columnsOfIndex, indexNameOf, indexSpecsOf, isClusteringIndex, joinIndexFor } from "../index/spec.js";
17
+ import { mapChildren, walkExpr } from "../parser/index.js";
18
+ import { arithmetic, negate } from "../value.js";
19
+ import { conjoin, conjuncts, equalityOn, exprToSql, formatRangeBounds, lookupText, rangeOn } from "./plan.js";
20
+ import { boundedRangeSelectivity, ndistinct, rootEstimate } from "./cost.js";
21
+ function unwrapLimit(plan) {
22
+ let rest = plan;
23
+ let limit = null;
24
+ let distinct = null;
25
+ if (rest.op === 'Limit') {
26
+ limit = { count: rest.count, offset: rest.offset ?? 0 };
27
+ rest = rest.child;
28
+ }
29
+ if (rest.op === 'HashDistinct' || rest.op === 'SortDistinct') {
30
+ distinct = rest;
31
+ rest = rest.child;
32
+ }
33
+ return { limit: limit === null && distinct === null ? null : { limit, distinct }, rest };
34
+ }
35
+ function rewrapLimit(top, plan) {
36
+ if (top === null)
37
+ return plan;
38
+ let out = plan;
39
+ if (top.distinct) {
40
+ out = top.distinct.op === 'HashDistinct'
41
+ ? { op: 'HashDistinct', child: out }
42
+ : { op: 'SortDistinct', ...(top.distinct.sortKey ? { sortKey: top.distinct.sortKey } : {}), child: out };
43
+ }
44
+ if (top.limit) {
45
+ out = { op: 'Limit', count: top.limit.count, ...(top.limit.offset > 0 ? { offset: top.limit.offset } : {}), child: out };
46
+ }
47
+ return out;
48
+ }
49
+ function unwrapSort(plan) {
50
+ return plan.op === 'Sort'
51
+ ? { sort: { column: plan.column, direction: plan.direction }, rest: plan.child }
52
+ : { sort: null, rest: plan };
53
+ }
54
+ function rewrapSort(sort, plan) {
55
+ return sort === null ? plan : { op: 'Sort', column: sort.column, direction: sort.direction, child: plan };
56
+ }
57
+ function unwrapAggregate(plan) {
58
+ if (plan.op === 'Having' && (plan.child.op === 'HashAggregate' || plan.child.op === 'SortAggregate')) {
59
+ return { aggregate: { node: plan.child, having: plan }, rest: plan.child.child };
60
+ }
61
+ return plan.op === 'HashAggregate' || plan.op === 'SortAggregate'
62
+ ? { aggregate: { node: plan, having: null }, rest: plan.child }
63
+ : { aggregate: null, rest: plan };
64
+ }
65
+ function rewrapAggregate(aggregate, plan) {
66
+ if (aggregate === null)
67
+ return plan;
68
+ const { node, having } = aggregate;
69
+ const rebuilt = node.op === 'HashAggregate'
70
+ ? { op: 'HashAggregate', groupBy: node.groupBy, aggregates: node.aggregates, child: plan }
71
+ : { op: 'SortAggregate', groupBy: node.groupBy, aggregates: node.aggregates, child: plan };
72
+ return having ? { op: 'Having', predicate: having.predicate, child: rebuilt } : rebuilt;
73
+ }
74
+ /**
75
+ * Statically compares two literals — the same three-valued logic
76
+ * `exec/evaluate.ts`'s `compare` uses at runtime (kept as a small separate
77
+ * copy rather than an import, since `planner/` never depends on `exec/`).
78
+ * NULL on either side is UNKNOWN, not foldable.
79
+ */
80
+ function foldLiteralCompare(op, left, right) {
81
+ if (left === null || right === null)
82
+ return null;
83
+ if (typeof left === 'number' && typeof right === 'number')
84
+ return orderedCompare(op, left, right);
85
+ const a = typeof left === 'boolean' ? String(Number(left)) : String(left);
86
+ const b = typeof right === 'boolean' ? String(Number(right)) : String(right);
87
+ return orderedCompare(op, a, b);
88
+ }
89
+ function orderedCompare(op, a, b) {
90
+ switch (op) {
91
+ case '=':
92
+ return a === b;
93
+ case '<>':
94
+ return a !== b;
95
+ case '<':
96
+ return a < b;
97
+ case '>':
98
+ return a > b;
99
+ case '<=':
100
+ return a <= b;
101
+ case '>=':
102
+ return a >= b;
103
+ }
104
+ }
105
+ /**
106
+ * A comparison whose truth value does not depend on the row at all: two
107
+ * literals (`5 > 3`). Anything else — a column against a literal, or two
108
+ * columns — depends on the data and cannot be folded.
109
+ *
110
+ * That includes a column against *itself*. `age = age` looks like a
111
+ * tautology, but under three-valued logic it is UNKNOWN, not TRUE, on a row
112
+ * where `age` is NULL — and only TRUE passes a filter. Folding it away
113
+ * returned (and, in a `DELETE`, destroyed) exactly the rows the predicate
114
+ * excludes. The SQLite oracle found this (plan.md §25.4 A1). The correct
115
+ * rewrite is `age IS NOT NULL`, which needs an `IS NULL` predicate the
116
+ * engine did not have; until it does, the comparison is simply left alone.
117
+ */
118
+ function foldComparison(expr) {
119
+ const { op, left, right } = expr;
120
+ if (left.kind === 'literal' && right.kind === 'literal') {
121
+ const result = foldLiteralCompare(op, left.value, right.value);
122
+ return result === null ? null : result ? 'true' : 'false';
123
+ }
124
+ return null;
125
+ }
126
+ /** Arithmetic built only from literals — `2 + 3`, `-(4 * 5)` — which needs no row to evaluate. */
127
+ function isConstantArithmetic(expr) {
128
+ if (expr.kind === 'literal')
129
+ return true;
130
+ if (expr.kind === 'arith')
131
+ return isConstantArithmetic(expr.left) && isConstantArithmetic(expr.right);
132
+ if (expr.kind === 'neg')
133
+ return isConstantArithmetic(expr.operand);
134
+ return false;
135
+ }
136
+ function literalFor(value, span) {
137
+ return { kind: 'literal', value, raw: value === null ? 'NULL' : String(value), span };
138
+ }
139
+ function evaluateConstant(expr) {
140
+ if (expr.kind === 'literal')
141
+ return expr.value;
142
+ if (expr.kind === 'neg')
143
+ return negate(evaluateConstant(expr.operand));
144
+ if (expr.kind === 'arith')
145
+ return arithmetic(expr.op, evaluateConstant(expr.left), evaluateConstant(expr.right));
146
+ return null;
147
+ }
148
+ /**
149
+ * Replaces every *maximal* constant-arithmetic subtree with its value
150
+ * (plan.md §25.4 B1b), recording what it folded. `WHERE id = 2 + 3` becomes
151
+ * `WHERE id = 5` — which is also what lets `index-selection` see an equality
152
+ * on a literal at all. Outermost only: `2 + 3 * 4` is one fold to 14, not two.
153
+ */
154
+ function foldArithmetic(expr, folds) {
155
+ if ((expr.kind === 'arith' || expr.kind === 'neg') && isConstantArithmetic(expr)) {
156
+ const folded = literalFor(evaluateConstant(expr), expr.span);
157
+ folds.push({ before: exprToSql(expr), after: folded.kind === 'literal' ? folded.raw : '' });
158
+ return folded;
159
+ }
160
+ return mapChildren(expr, (child) => foldArithmetic(child, folds));
161
+ }
162
+ /**
163
+ * Two simplifications, both about work that does not depend on the row:
164
+ *
165
+ * - constant arithmetic (`2 + 3`) is worked out once, here, instead of for
166
+ * every row the scan reads (and so a literal can reach an index rule);
167
+ * - a conjunct that is always true (`5 > 3`) is dropped — `X AND true` is `X`.
168
+ *
169
+ * Deliberately does not act on an always-false conjunct: `X AND false` is
170
+ * `false`, but v0's plan shape has no node for "this scan produces no rows"
171
+ * without inventing one, so a `1 = 2`-shaped clause is left exactly as
172
+ * written. The query still runs correctly either way — this rule only ever
173
+ * removes work that was already redundant, never changes what a query
174
+ * returns.
175
+ */
176
+ const constantFolding = (input) => {
177
+ const { limit, rest: plan } = unwrapLimit(input);
178
+ if (plan.op !== 'Project')
179
+ return null;
180
+ const { sort, rest: belowSort } = unwrapSort(plan.child);
181
+ const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
182
+ if (belowProject.op !== 'Filter')
183
+ return null;
184
+ const folds = [];
185
+ const parts = conjuncts(belowProject.predicate).map((part) => foldArithmetic(part, folds));
186
+ const alwaysTrue = parts
187
+ .map((p, i) => ({ i, expr: p }))
188
+ .filter(({ expr }) => expr.kind === 'compare' && foldComparison(expr) === 'true');
189
+ if (alwaysTrue.length === 0 && folds.length === 0)
190
+ return null;
191
+ const dropped = new Set(alwaysTrue.map((f) => f.i));
192
+ const residual = conjoin(parts.filter((_, i) => !dropped.has(i)));
193
+ const rewrittenFilter = residual
194
+ ? { op: 'Filter', predicate: residual, child: belowProject.child }
195
+ : belowProject.child;
196
+ const root = rewrapLimit(limit, {
197
+ op: 'Project',
198
+ columns: plan.columns,
199
+ child: rewrapSort(sort, rewrapAggregate(aggregate, rewrittenFilter)),
200
+ });
201
+ const foldSentence = folds.length === 0
202
+ ? ''
203
+ : `${folds.map((f) => `\`${f.before}\` is just ${f.after}`).join(', ')} — worked out once here instead of for every row`;
204
+ const clauses = alwaysTrue.map((f) => `\`${exprToSql(f.expr)}\``).join(', ');
205
+ const verb = alwaysTrue.length === 1 ? 'is' : 'are';
206
+ const dropSentence = alwaysTrue.length === 0
207
+ ? ''
208
+ : residual
209
+ ? `${clauses} ${verb} always true, whatever the row — it drops out of the predicate, leaving only the part that actually depends on data`
210
+ : `${clauses} ${verb} always true, whatever the row — with nothing else in the WHERE clause, the filter disappears entirely`;
211
+ return {
212
+ rule: 'constant-folding',
213
+ label: `Constant folding: ${[foldSentence, dropSentence].filter(Boolean).join('; and ')}.`,
214
+ plan: root,
215
+ changed: new Set([rewrittenFilter]),
216
+ };
217
+ };
218
+ /**
219
+ * Fold `Filter` into the scan beneath it.
220
+ *
221
+ * In a single-table query the filter already sits directly above the scan, so
222
+ * there is nothing to push it *past*. The real move is pushing it *into* the
223
+ * scan: the scan then discards non-matching rows as it reads them, instead of
224
+ * materialising every row for a separate operator to throw away.
225
+ */
226
+ const predicatePushdown = (input) => {
227
+ const { limit, rest: plan } = unwrapLimit(input);
228
+ if (plan.op !== 'Project')
229
+ return null;
230
+ const { sort, rest: belowSort } = unwrapSort(plan.child);
231
+ const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
232
+ if (belowProject.op !== 'Filter')
233
+ return null;
234
+ const scan = belowProject.child;
235
+ if (scan.op !== 'SeqScan' || scan.filter)
236
+ return null;
237
+ const pushed = { op: 'SeqScan', table: scan.table, filter: belowProject.predicate };
238
+ const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, pushed)) });
239
+ return {
240
+ rule: 'predicate-pushdown',
241
+ label: 'Predicate pushdown: the filter moves into the scan, so rows are discarded as they are read rather than materialised and then thrown away.',
242
+ plan: root,
243
+ changed: new Set([pushed]),
244
+ };
245
+ };
246
+ /** Every table a qualified column reference inside `expr` names. Empty for a literal. */
247
+ function tablesReferencedBy(expr) {
248
+ const tables = new Set();
249
+ walkExpr(expr, (e) => {
250
+ if (e.kind === 'column' && e.table)
251
+ tables.add(e.table);
252
+ });
253
+ return tables;
254
+ }
255
+ /**
256
+ * Clears every column's table qualifier. A conjunct pushed into one side's own
257
+ * `SeqScan` runs against that table's *raw* rows — `{ id: 1, name: 'ada' }`,
258
+ * never `{ 'authors.id': 1, ... }` — qualification only exists to route a
259
+ * `WHERE` clause to the right side before the rewrite, the same way it exists
260
+ * in the SQL text only to route the reference to the right table at parse
261
+ * time. Left qualified, `authors.name = 'ada'` would read `row['authors.name']`
262
+ * off a row that has no such key and silently match nothing.
263
+ */
264
+ function stripTableQualifiers(expr) {
265
+ if (expr.kind === 'column') {
266
+ if (!expr.table)
267
+ return expr;
268
+ const { table: _table, ...rest } = expr;
269
+ return rest;
270
+ }
271
+ return mapChildren(expr, stripTableQualifiers);
272
+ }
273
+ /**
274
+ * Pushes each `WHERE` conjunct that names only one side of a `JOIN` into that
275
+ * side's own `SeqScan` — `predicatePushdown`'s counterpart for a `Join`'s two
276
+ * children instead of a single scan. A conjunct naming both sides can only be
277
+ * evaluated once a row from each side is already paired, so it stays behind
278
+ * as a residual `Filter` above the `Join`; every reference is qualified once
279
+ * a query has a `JOIN` at all (the parser guarantees it), so there is no
280
+ * unqualified conjunct to worry about routing.
281
+ *
282
+ * Deliberately narrow: it does not also run `index-selection` on the side it
283
+ * just filtered — that rule only ever looks at the very top of the tree
284
+ * (`Project → … → SeqScan`), not inside a `Join`'s own children, so a pushed
285
+ * conjunct that happens to match an index still runs as a full scan. Reaching
286
+ * `index-selection` into a `Join`'s children is its own, larger step.
287
+ *
288
+ * Also deliberately narrow about *which* `Join`: `join.left.op !== 'SeqScan'` bails on a chain of two or more
289
+ * `JOIN`s (plan.md §25.4 C3 slice b) just as it would on any other shape this rule does not recognise — a chain's
290
+ * outermost `Join` has another `Join` as its `left`, never a bare scan. A 3-or-more-table query's `WHERE` clause
291
+ * therefore always stays a residual `Filter` above the whole chain for now, exactly like a predicate this rule
292
+ * already declines to reach *into* a single `Join`'s own children (the paragraph above) — a missed optimization,
293
+ * not a correctness gap; extending pushdown down a chain is left for later, alongside reaching it past a `Join` at all.
294
+ *
295
+ * A conjunct on the *inner* side of a `LEFT JOIN` (plan.md §25.4 C3) is never pushed, on purpose: filtering it into
296
+ * the inner scan removes rows *before* the join can even find out they would not have matched, so a left row that
297
+ * *would* have been NULL-padded instead vanishes with no trace of it — an inner join wearing a LEFT JOIN's syntax.
298
+ * Left as a residual `Filter` above the `Join` instead, it runs after the padding, exactly where SQL puts it. The
299
+ * outer (left) side has no such trap — filtering it first only ever removes rows the join would keep anyway.
300
+ */
301
+ const joinPredicatePushdown = (input) => {
302
+ const { limit, rest: plan } = unwrapLimit(input);
303
+ if (plan.op !== 'Project')
304
+ return null;
305
+ const { sort, rest: belowSort } = unwrapSort(plan.child);
306
+ const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
307
+ if (belowProject.op !== 'Filter')
308
+ return null;
309
+ const join = belowProject.child;
310
+ // The inner side is a bare `SeqScan` — or, for an index nested-loop join, an `IndexProbe`, which has no scan to
311
+ // filter: a conjunct naming only the inner table then stays above the Join as a residual, never pushed.
312
+ const probesInner = join.op === 'Join' && join.right.op === 'IndexProbe';
313
+ if (join.op !== 'Join' ||
314
+ join.left.op !== 'SeqScan' ||
315
+ join.left.filter ||
316
+ (join.right.op !== 'SeqScan' && join.right.op !== 'IndexProbe') ||
317
+ (join.right.op === 'SeqScan' && join.right.filter)) {
318
+ return null;
319
+ }
320
+ const parts = conjuncts(belowProject.predicate);
321
+ const leftParts = [];
322
+ const rightParts = [];
323
+ const residualParts = [];
324
+ for (const part of parts) {
325
+ const tables = tablesReferencedBy(part);
326
+ if (tables.size === 1 && tables.has(join.leftTable))
327
+ leftParts.push(part);
328
+ else if (tables.size === 1 && tables.has(join.rightTable) && !probesInner && join.joinType !== 'left')
329
+ rightParts.push(part);
330
+ else
331
+ residualParts.push(part);
332
+ }
333
+ if (leftParts.length === 0 && rightParts.length === 0)
334
+ return null;
335
+ const newLeft = leftParts.length > 0
336
+ ? { op: 'SeqScan', table: join.leftTable, filter: stripTableQualifiers(conjoin(leftParts)) }
337
+ : join.left;
338
+ const newRight = rightParts.length > 0
339
+ ? { op: 'SeqScan', table: join.rightTable, filter: stripTableQualifiers(conjoin(rightParts)) }
340
+ : join.right;
341
+ const newJoin = { ...join, left: newLeft, right: newRight };
342
+ const residual = conjoin(residualParts);
343
+ const filtered = residual ? { op: 'Filter', predicate: residual, child: newJoin } : newJoin;
344
+ const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, filtered)) });
345
+ const pushedSides = [
346
+ leftParts.length > 0 ? join.leftTable : null,
347
+ rightParts.length > 0 ? join.rightTable : null,
348
+ ].filter((t) => t !== null);
349
+ const changed = new Set([newJoin, ...(leftParts.length > 0 ? [newLeft] : []), ...(rightParts.length > 0 ? [newRight] : [])]);
350
+ return {
351
+ rule: 'join-predicate-pushdown',
352
+ label: residual
353
+ ? `Predicate pushdown: the part${pushedSides.length === 1 ? '' : 's'} of the WHERE clause naming only ${pushedSides.join(' or only ')} move${pushedSides.length === 1 ? 's' : ''} into that scan; the rest can only be checked once a row from each side is paired, so it stays a residual Filter above the Join.`
354
+ : `Predicate pushdown: every part of the WHERE clause names only one side, so the whole thing moves into ${pushedSides.length === 1 ? 'that scan' : 'the two scans'} — nothing is left to recheck after the Join.`,
355
+ plan: root,
356
+ changed,
357
+ };
358
+ };
359
+ /** A join algorithm's name for narration — `emit.ts`'s own `describeJoin` is phrase-shaped for a different sentence, so this stays a small local copy rather than an import that would need reshaping either way. */
360
+ function joinAlgorithmName(algorithm) {
361
+ switch (algorithm) {
362
+ case 'nested-loop':
363
+ return 'nested-loop';
364
+ case 'hash':
365
+ return 'hash join';
366
+ case 'sort-merge':
367
+ return 'sort-merge';
368
+ case 'index-nested-loop':
369
+ return 'index nested-loop';
370
+ }
371
+ }
372
+ /**
373
+ * Recomputes one `Join` node's own algorithm by cost, after first doing the same for whatever `Join` sits beneath
374
+ * it in a chain (plan.md §25.4 C3 slice b) — an earlier step's own chosen algorithm changes its own estimated page
375
+ * cost, which a later step's comparison then has to use, the same way a real cost-based join enumerator prices a
376
+ * whole left-deep chain bottom-up. `right`'s shape changes with the candidate: a bare `SeqScan` for nested-loop,
377
+ * hash and sort-merge, an `IndexProbe` only for index-nested-loop — the same swap `buildJoinPlan` itself makes.
378
+ */
379
+ function chooseJoinAlgorithms(join, table, rowsPerPage, otherTables) {
380
+ const tableNamed = (name) => (name === table.name ? table : otherTables[name]);
381
+ const below = join.left.op === 'Join' ? chooseJoinAlgorithms(join.left, table, rowsPerPage, otherTables) : null;
382
+ const newLeft = below ? below.plan : join.left;
383
+ const rightTableData = tableNamed(join.rightTable);
384
+ // Reused as-is, not rebuilt bare: `join-predicate-pushdown` (which already ran, earlier in `RULES`) may have
385
+ // already pushed a WHERE conjunct into this exact scan — losing that filter here would silently widen the
386
+ // join's own results, not just its cost estimate.
387
+ const plainRight = join.right.op === 'SeqScan' ? join.right : { op: 'SeqScan', table: join.rightTable };
388
+ const rightHasPushedFilter = plainRight.op === 'SeqScan' && plainRight.filter !== undefined;
389
+ // `IndexProbe` has nowhere to carry a residual filter, so a pushed-down conjunct on this side rules index
390
+ // nested-loop out entirely rather than risk silently dropping it — the same reason `join-predicate-pushdown`
391
+ // itself never pushes a conjunct into a side that already probes an index (its own `probesInner` check).
392
+ const idxSpec = rightHasPushedFilter ? undefined : joinIndexFor(rightTableData, join.rightColumn);
393
+ const algorithms = idxSpec
394
+ ? ['nested-loop', 'hash', 'sort-merge', 'index-nested-loop']
395
+ : ['nested-loop', 'hash', 'sort-merge'];
396
+ const candidates = algorithms.map((algorithm) => {
397
+ const right = algorithm === 'index-nested-loop'
398
+ ? { op: 'IndexProbe', table: join.rightTable, column: indexNameOf(idxSpec.columns), outerTable: join.leftTable, outerColumn: join.leftColumn }
399
+ : plainRight;
400
+ const plan = { ...join, left: newLeft, right, algorithm };
401
+ return { algorithm, plan, estPages: rootEstimate(plan, table, rowsPerPage, otherTables).estPages };
402
+ });
403
+ // Ties favour whatever is already there (nested-loop, first in `algorithms`) — never rewrite just because two
404
+ // candidates happen to cost the same.
405
+ const chosen = candidates.reduce((best, c) => (c.estPages < best.estPages ? c : best));
406
+ const runnersUp = candidates.filter((c) => c !== chosen);
407
+ const changedHere = chosen.algorithm !== join.algorithm;
408
+ return {
409
+ plan: chosen.plan,
410
+ changes: [...(below?.changes ?? []), ...(changedHere ? [{ leftTable: join.leftTable, rightTable: join.rightTable, chosen, runnersUp }] : [])],
411
+ changedNodes: [...(below?.changedNodes ?? []), ...(changedHere ? [chosen.plan] : [])],
412
+ };
413
+ }
414
+ /**
415
+ * Chooses each `JOIN`'s physical algorithm by cost (plan.md §25.4 C5 slice 2) — the first rule with no shape step
416
+ * at all. `buildPlan.ts` always starts a chain out as `nested-loop`, its own cost-oblivious default, the same way
417
+ * it always starts a predicate out as a bare `SeqScan`; this rule prices every algorithm that could actually run
418
+ * each step — nested-loop, hash, sort-merge, and index-nested-loop whenever the inner table has a usable index —
419
+ * with `cost.ts`'s own per-algorithm formulas (the same ones `EXPLAIN` already shows), and keeps whichever is
420
+ * cheapest. Unlike `index-selection` there is no "matches the shape but costs more, so decline" outcome: every
421
+ * applicable algorithm is a real, working candidate, so the rule always ends up choosing one — it may just be the
422
+ * one already there, in which case nothing changed and this reports no rewrite, like any other rule that finds
423
+ * nothing worth doing.
424
+ *
425
+ * A chain (plan.md §25.4 C3 slice b) is priced bottom-up, one `Join` at a time — see `chooseJoinAlgorithms`.
426
+ *
427
+ * Only runs when the caller never forced a specific algorithm: `emit.ts` leaves this rule out of `enabled`
428
+ * whenever `EngineOptions.joinStrategy` was set explicitly — `/compare`'s and `/tools/planner`'s whole point is
429
+ * forcing one algorithm to see its own effect, which a cost comparison would otherwise silently override.
430
+ */
431
+ const joinAlgorithmSelection = (input, table, rowsPerPage, otherTables) => {
432
+ const { limit, rest: plan } = unwrapLimit(input);
433
+ if (plan.op !== 'Project')
434
+ return null;
435
+ const belowProject = plan.child;
436
+ const outerJoin = belowProject.op === 'Filter' ? belowProject.child : belowProject;
437
+ if (outerJoin.op !== 'Join')
438
+ return null;
439
+ const result = chooseJoinAlgorithms(outerJoin, table, rowsPerPage, otherTables);
440
+ if (result.changes.length === 0)
441
+ return null;
442
+ const newBelowProject = belowProject.op === 'Filter' ? { ...belowProject, child: result.plan } : result.plan;
443
+ const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: newBelowProject });
444
+ const label = result.changes
445
+ .map(({ leftTable, rightTable, chosen, runnersUp }) => {
446
+ const others = runnersUp.map((c) => `${joinAlgorithmName(c.algorithm)}'s ${String(c.estPages)}`).join(', ');
447
+ return `\`${leftTable}\` × \`${rightTable}\`: ${joinAlgorithmName(chosen.algorithm)} chosen by cost — an estimated ${String(chosen.estPages)} pages against ${others}.`;
448
+ })
449
+ .join(' ');
450
+ // The "Why not?" panel (plan.md §25.4 C5, fourth slice) names one runner-up, not the whole candidate set — the
451
+ // first changed step's own cheapest alternative, since a `Rewrite` carries a single `whyNot`, not one per step.
452
+ // A chain with several changed steps still gets the full comparison in `label` above; only the panel is scoped
453
+ // to the first step.
454
+ const firstChange = result.changes[0];
455
+ const cheapestRunnerUp = firstChange.runnersUp.reduce((best, c) => (c.estPages < best.estPages ? c : best));
456
+ return {
457
+ rule: 'join-algorithm-selection',
458
+ label,
459
+ plan: root,
460
+ changed: new Set(result.changedNodes),
461
+ whyNot: {
462
+ chosen: { label: joinAlgorithmName(firstChange.chosen.algorithm), estPages: firstChange.chosen.estPages },
463
+ runnerUp: { label: joinAlgorithmName(cheapestRunnerUp.algorithm), estPages: cheapestRunnerUp.estPages },
464
+ },
465
+ };
466
+ };
467
+ /** The lower and upper bound `parts` put on `column`, and which conjuncts they are — at most one of each direction. */
468
+ function rangeBoundsOn(parts, column, taken) {
469
+ const matches = parts
470
+ .map((p, i) => ({ i, match: taken.has(i) ? undefined : rangeOn(p, column) }))
471
+ .filter((m) => m.match !== undefined);
472
+ const lower = matches.find((m) => m.match.op === '>' || m.match.op === '>=');
473
+ const upper = matches.find((m) => m.match.op === '<' || m.match.op === '<=');
474
+ if (!lower && !upper)
475
+ return undefined;
476
+ const low = lower ? { value: lower.match.value, inclusive: lower.match.op === '>=' } : undefined;
477
+ const high = upper ? { value: upper.match.value, inclusive: upper.match.op === '<=' } : undefined;
478
+ return { column, ...(low ? { low } : {}), ...(high ? { high } : {}), consumed: [lower?.i, upper?.i].filter((i) => i !== undefined) };
479
+ }
480
+ /** The composite indexes `parts` cannot enter: it constrains one of their columns, but not their leftmost one. */
481
+ function unusableComposites(table, parts) {
482
+ return indexSpecsOf(table).flatMap((spec) => {
483
+ if (spec.columns.length < 2)
484
+ return [];
485
+ const constrains = (column) => parts.some((p) => equalityOn(p, column) !== undefined || rangeOn(p, column) !== undefined);
486
+ if (constrains(spec.columns[0]))
487
+ return [];
488
+ const named = spec.columns.slice(1).filter(constrains);
489
+ return named.length > 0 ? [{ name: indexNameOf(spec.columns), columns: spec.columns, named }] : [];
490
+ });
491
+ }
492
+ /**
493
+ * The leftmost-prefix rule, said aloud for whoever is looking at a plan that did *not* use an index it might have
494
+ * (plan.md §25.4 B3): a composite index on `(a, b)` is ordered by `a` first, so an equality on `b` alone matches entries
495
+ * scattered across the whole index — there is no run to read, and the index cannot be entered. `null` when there is
496
+ * nothing to say.
497
+ */
498
+ export function leftmostPrefixNote(table, where) {
499
+ if (!where)
500
+ return null;
501
+ const skipped = unusableComposites(table, conjuncts(where));
502
+ if (skipped.length === 0)
503
+ return null;
504
+ return skipped
505
+ .map((c) => `The index on (${c.columns.join(', ')}) cannot help: the predicate constrains ${c.named.map((n) => `\`${n}\``).join(' and ')} but not \`${c.columns[0]}\`, and an index can only be entered through its leftmost column — its entries are ordered by \`${c.columns[0]}\` first, so ${c.named.length === 1 ? 'that value is' : 'those values are'} scattered across the whole index.`)
506
+ .join(' ');
507
+ }
508
+ /**
509
+ * Swap a `SeqScan` for an `IndexScan` when a conjunct is an equality on an
510
+ * indexed column. Conjuncts the index cannot answer stay behind as a
511
+ * recheck — the index narrows the search, it does not verify the whole
512
+ * predicate.
513
+ *
514
+ * A table can declare more than one index (plan.md §22.2's "second candidate
515
+ * index"). When two or more each have a matching equality conjunct, this
516
+ * picks the more selective one — higher `ndistinct`, so an equality on it
517
+ * matches fewer rows on average — the same fact the cost model's own
518
+ * equality-selectivity estimate already rests on, not a second, independent
519
+ * comparison. The conjunct that lost stays in the recheck, exactly like every
520
+ * other conjunct the chosen index can't answer — v0 has no way to intersect
521
+ * two index lookups.
522
+ *
523
+ * A **composite** index (plan.md §25.4 B3) is entered through a *leftmost
524
+ * prefix*: the run of its leading columns that each have an equality. An
525
+ * equality on `a` enters an index on `(a, b)`; one on `b` alone cannot, and the
526
+ * rule leaves that index unused — and says why. The more columns a candidate's
527
+ * prefix pins, the fewer rows it selects, so that is what the choice compares.
528
+ *
529
+ * **Chosen by cost, not just by shape (plan.md §25.4 C5)**: once the most
530
+ * selective *candidate* is picked, its own estimated page cost is compared
531
+ * against the scan it would replace — `rootEstimate` on each, the same
532
+ * formula `EXPLAIN` already shows. A candidate that matches the predicate's
533
+ * shape but would not actually be cheaper (a low-selectivity index — most of
534
+ * `logins.status`'s rows are `'ok'`, say — costs more one random heap fetch
535
+ * at a time than one sequential scan already would) is *declined*: the rule
536
+ * still reports what it considered and why it said no, but the plan stays a
537
+ * `SeqScan`. This is the one thing plan.md §12 always drew a line at (*"an
538
+ * invented figure with a decimal point is worse than an honest guess"*) —
539
+ * not that a rule could never consult a cost, only that doing so had to be
540
+ * real, not a rule of thumb pretending to be one. `rootEstimate` already is.
541
+ */
542
+ const indexSelection = (input, table, rowsPerPage) => {
543
+ const { limit, rest: plan } = unwrapLimit(input);
544
+ if (plan.op !== 'Project')
545
+ return null;
546
+ const { sort, rest: belowSort } = unwrapSort(plan.child);
547
+ const { aggregate, rest: scan } = unwrapAggregate(belowSort);
548
+ if (scan.op !== 'SeqScan' || !scan.filter)
549
+ return null;
550
+ const parts = conjuncts(scan.filter);
551
+ const candidates = indexSpecsOf(table).flatMap((spec) => {
552
+ const matched = [];
553
+ const used = new Set();
554
+ for (const column of spec.columns) {
555
+ // The run of matched columns stops at the first one with no equality — that is the leftmost-prefix rule.
556
+ const partIndex = parts.findIndex((p, i) => !used.has(i) && equalityOn(p, column) !== undefined);
557
+ if (partIndex === -1)
558
+ break;
559
+ used.add(partIndex);
560
+ matched.push({ column, partIndex, key: equalityOn(parts[partIndex], column) });
561
+ }
562
+ if (matched.length === 0)
563
+ return [];
564
+ const nextColumn = spec.columns[matched.length];
565
+ // A range *within* the prefix needs the entries' own row addresses to walk the leaf chain (`index/rangeLookup.ts`)
566
+ // — a clustered table's secondary index carries a clustering key instead (plan.md §25.4 B3e), so this narrowing
567
+ // is only offered for the clustering key's own index, or when the table isn't clustered at all.
568
+ const clusterSafe = !table.clusteredKey || isClusteringIndex(table.clusteredKey, spec.columns);
569
+ const next = nextColumn === undefined || !clusterSafe ? undefined : rangeBoundsOn(parts, nextColumn, used);
570
+ return [
571
+ {
572
+ name: indexNameOf(spec.columns),
573
+ columns: spec.columns,
574
+ unique: spec.unique === true,
575
+ matched,
576
+ fraction: matched.reduce((product, m) => product / ndistinct(table, m.column), 1),
577
+ ...(next ? { next } : {}),
578
+ },
579
+ ];
580
+ });
581
+ if (candidates.length === 0)
582
+ return null;
583
+ const distinctOf = (c) => Math.round(1 / c.fraction);
584
+ // Narrower wins: a smaller fraction, then more columns pinned, then a range on the next column to narrow the run further.
585
+ const better = (c, best) => c.fraction !== best.fraction
586
+ ? c.fraction < best.fraction
587
+ : c.matched.length !== best.matched.length
588
+ ? c.matched.length > best.matched.length
589
+ : c.next !== undefined && best.next === undefined;
590
+ const chosen = candidates.reduce((best, c) => (better(c, best) ? c : best));
591
+ const runnersUp = candidates.filter((c) => c !== chosen);
592
+ const consumed = new Set([...chosen.matched.map((m) => m.partIndex), ...(chosen.next?.consumed ?? [])]);
593
+ const residual = conjoin(parts.filter((_, i) => !consumed.has(i)));
594
+ const keys = chosen.matched.map((m) => m.key);
595
+ const composite = chosen.columns.length > 1;
596
+ // An equality prefix with a range on the next column becomes a range scan *within* the prefix's run; otherwise a lookup.
597
+ const indexScan = chosen.next
598
+ ? {
599
+ op: 'IndexRangeScan',
600
+ table: scan.table,
601
+ column: chosen.name,
602
+ prefix: keys,
603
+ ...(chosen.next.low ? { low: chosen.next.low } : {}),
604
+ ...(chosen.next.high ? { high: chosen.next.high } : {}),
605
+ ...(residual ? { residual } : {}),
606
+ }
607
+ : {
608
+ op: 'IndexScan',
609
+ table: scan.table,
610
+ column: chosen.name,
611
+ key: keys[0],
612
+ ...(composite ? { keys } : {}),
613
+ ...(residual ? { residual } : {}),
614
+ };
615
+ const nameOf = (c) => (c.columns.length > 1 ? `(${c.columns.join(', ')})` : `\`${c.name}\``);
616
+ // Chosen by cost, not just by shape (plan.md §25.4 C5): a candidate that matches the predicate can still lose to
617
+ // the scan it would replace — a low-selectivity index costs one random heap fetch per match, which adds up past
618
+ // whatever one sequential scan already costs. `rootEstimate` on each side is the same formula `EXPLAIN` shows,
619
+ // not a new heuristic invented for this comparison alone.
620
+ const scanPages = rootEstimate(scan, table, rowsPerPage).estPages;
621
+ const indexPages = rootEstimate(indexScan, table, rowsPerPage).estPages;
622
+ if (indexPages >= scanPages) {
623
+ return {
624
+ rule: 'index-selection',
625
+ label: `${nameOf(chosen)} matches the predicate, but the estimated cost disagrees: the lookup would read about ${String(indexPages)} page${indexPages === 1 ? '' : 's'} against the scan's ${String(scanPages)} — no cheaper, so the scan stays and the whole predicate is rechecked there instead.`,
626
+ plan: input,
627
+ changed: new Set(),
628
+ whyNot: {
629
+ chosen: { label: 'a sequential scan', estPages: scanPages },
630
+ runnerUp: { label: `${nameOf(chosen)} index lookup`, estPages: indexPages },
631
+ },
632
+ };
633
+ }
634
+ const root = rewrapLimit(limit, {
635
+ op: 'Project',
636
+ columns: plan.columns,
637
+ child: rewrapSort(sort, rewrapAggregate(aggregate, indexScan)),
638
+ });
639
+ const chosenDistinct = distinctOf(chosen);
640
+ const runnerUpNames = runnersUp.map(nameOf).join(', ');
641
+ const runnerUpDistinct = runnersUp.map(distinctOf);
642
+ const choiceClause = runnersUp.length === 0
643
+ ? ''
644
+ : runnerUpDistinct.every((d) => d === chosenDistinct)
645
+ ? ` (tied with ${runnerUpNames} at ${String(chosenDistinct)} distinct values each — ${nameOf(chosen)} wins only because it was declared first, not because it narrows the search any further)`
646
+ : ` (chosen over ${runnerUpNames} — ${String(chosenDistinct)} distinct values beats ${runnerUpDistinct.map(String).join('/')}, so it narrows the search more)`;
647
+ const named = chosen.matched.map((m) => m.column);
648
+ const prefixClause = composite
649
+ ? named.length === chosen.columns.length
650
+ ? `The predicate has an equality on every column of the composite index (${chosen.columns.join(', ')})`
651
+ : `The predicate has an equality on ${named.map((c) => `\`${c}\``).join(' and ')} — ${named.length === 1 ? 'the leftmost column' : 'the leftmost columns'} of the composite index (${chosen.columns.join(', ')}), which is all an index needs to be entered: \`${chosen.columns[named.length]}\` only orders the entries inside each run`
652
+ : '';
653
+ const unusable = leftmostPrefixNote(table, scan.filter);
654
+ const uniqueClause = chosen.unique && named.length === chosen.columns.length && !chosen.next
655
+ ? ' The index is UNIQUE, so the lookup returns at most one row and stops at the first entry it finds.'
656
+ : '';
657
+ const rangeClause = chosen.next
658
+ ? ` The next column, \`${chosen.next.column}\`, has a range (${formatRangeBounds(chosen.next.column, chosen.next.low, chosen.next.high)}), so the index narrows it within that run: one descent to where the range begins inside ${describeKeysText(keys)}, then along the leaf chain until the run ends.`
659
+ : '';
660
+ // Whether this costs a second descent (plan.md §25.4 B3e) depends on whether `index-only-scan` covers the query
661
+ // next — this rule runs before it and cannot know yet, so that fact is left to the execution-stage narration
662
+ // (`index/lookup.ts`'s `emitIndexLookup`), which always gets to see the real answer.
663
+ const clusteredClause = table.clusteredKey && !isClusteringIndex(table.clusteredKey, chosen.columns)
664
+ ? ` This table is clustered on \`${indexNameOf(table.clusteredKey)}\`, and ${composite ? `(${chosen.columns.join(', ')})` : `\`${chosen.name}\``} is a secondary index — it holds the clustering key, not the row's address.`
665
+ : '';
666
+ return {
667
+ rule: 'index-selection',
668
+ label: `${composite
669
+ ? `${prefixClause}${choiceClause}, so the ${named.length > 1 ? 'equalities become' : 'equality becomes'} an index lookup.${residual ? ' The rest of the predicate cannot be answered by the index and is rechecked on each row the index returns.' : ''}`
670
+ : residual
671
+ ? `\`${chosen.name}\` is indexed${choiceClause}, so the equality becomes an index lookup. The rest of the predicate cannot be answered by the index and is rechecked on each row the index returns.`
672
+ : `\`${chosen.name}\` is indexed${choiceClause} and the predicate is an equality on it, so the scan becomes a B+Tree lookup — a handful of pages instead of the whole table.`}${uniqueClause}${rangeClause}${clusteredClause}${unusable ? ` ${unusable}` : ''}`,
673
+ plan: root,
674
+ changed: new Set([indexScan]),
675
+ prompt: composite && named.length < chosen.columns.length
676
+ ? leftmostPrefixPrompt(chosen.name, chosen.columns, named)
677
+ : composite
678
+ ? indexSelectionPrompt(`(${chosen.columns.join(', ')})`, keys[0], lookupText(chosen.name, keys))
679
+ : indexSelectionPrompt(chosen.name, chosen.matched[0].key),
680
+ whyNot: {
681
+ chosen: { label: `${nameOf(chosen)} index lookup`, estPages: indexPages },
682
+ runnerUp: { label: 'a sequential scan', estPages: scanPages },
683
+ },
684
+ };
685
+ };
686
+ /**
687
+ * Swap a `SeqScan` for an `IndexRangeScan` when a range comparison (`<`,
688
+ * `>`, `<=`, `>=` — including a `BETWEEN`, already desugared to both a lower
689
+ * and an upper bound by the parser) sits on an indexed column. Descends the
690
+ * B+Tree once to the leaf the lower bound would start at (the leftmost leaf,
691
+ * with none), then the executor walks the leaf sibling chain horizontally
692
+ * rather than rescanning the table (plan.md §23.1).
693
+ *
694
+ * Deliberately narrower than `index-selection`: it only ever sees a plan
695
+ * `index-selection` left untouched as a `SeqScan` — no equality conjunct
696
+ * matched any indexed column at all, *or* `index-selection` found one but
697
+ * declined it on cost (plan.md §25.4 C5) — because an equality is always at
698
+ * least as selective a starting point as a range, and choosing between an
699
+ * equality's `1/ndistinct` and a range's interval fraction by comparing the
700
+ * two numbers directly would still be exactly the kind of made-up comparison
701
+ * §12 rules out; the real cost model, below, is a different thing. At most
702
+ * one lower bound and one upper bound per column are consumed; a second
703
+ * bound in the same direction (`age > 10 AND age > 20`) is left as a
704
+ * residual recheck rather than merged — correctness never depends on
705
+ * catching it, only how tightly the index narrows the search does.
706
+ *
707
+ * **Chosen by cost too**: the same gate `index-selection` applies against the scan it would replace — see its own
708
+ * doc comment for why.
709
+ */
710
+ const rangeIndexSelection = (input, table, rowsPerPage) => {
711
+ const { limit, rest: plan } = unwrapLimit(input);
712
+ if (plan.op !== 'Project')
713
+ return null;
714
+ const { sort, rest: belowSort } = unwrapSort(plan.child);
715
+ const { aggregate, rest: scan } = unwrapAggregate(belowSort);
716
+ if (scan.op !== 'SeqScan' || !scan.filter)
717
+ return null;
718
+ const parts = conjuncts(scan.filter);
719
+ // A range enters an index through its *leading* column, plain or composite — the entries are ordered by it first.
720
+ // A range walk reads the leaf chain by row address, which a clustered table's secondary index does not carry
721
+ // (plan.md §25.4 B3e) — so a range only ever chooses the clustering key's own index there, or any index at all
722
+ // when the table isn't clustered.
723
+ const clusterSafe = (spec) => !table.clusteredKey || isClusteringIndex(table.clusteredKey, spec.columns);
724
+ const candidates = indexSpecsOf(table)
725
+ .filter(clusterSafe)
726
+ .flatMap((spec) => {
727
+ const column = spec.columns[0];
728
+ const name = indexNameOf(spec.columns);
729
+ const matches = parts
730
+ .map((p, i) => ({ i, match: rangeOn(p, column) }))
731
+ .filter((m) => m.match !== undefined);
732
+ if (matches.length === 0)
733
+ return [];
734
+ const lower = matches.find((m) => m.match.op === '>' || m.match.op === '>=');
735
+ const upper = matches.find((m) => m.match.op === '<' || m.match.op === '<=');
736
+ const consumed = new Set([lower?.i, upper?.i].filter((i) => i !== undefined));
737
+ const low = lower
738
+ ? { value: lower.match.value, inclusive: lower.match.op === '>=' }
739
+ : undefined;
740
+ const high = upper
741
+ ? { value: upper.match.value, inclusive: upper.match.op === '<=' }
742
+ : undefined;
743
+ const numericLow = low && typeof low.value === 'number' ? low.value : null;
744
+ const numericHigh = high && typeof high.value === 'number' ? high.value : null;
745
+ const selectivity = boundedRangeSelectivity(column, numericLow, numericHigh, table);
746
+ return [{ column, name, consumed, low, high, selectivity }];
747
+ });
748
+ if (candidates.length === 0)
749
+ return null;
750
+ const chosen = candidates.reduce((best, c) => c.selectivity.fraction < best.selectivity.fraction ? c : best);
751
+ const runnersUp = candidates.filter((c) => c !== chosen);
752
+ const residual = conjoin(parts.filter((_, i) => !chosen.consumed.has(i)));
753
+ const rangeScan = {
754
+ op: 'IndexRangeScan',
755
+ table: scan.table,
756
+ column: chosen.name,
757
+ ...(chosen.low ? { low: chosen.low } : {}),
758
+ ...(chosen.high ? { high: chosen.high } : {}),
759
+ ...(residual ? { residual } : {}),
760
+ };
761
+ // Chosen by cost, not just by shape (plan.md §25.4 C5) — the same gate `index-selection` applies, see its own
762
+ // doc comment: a range that matches the predicate can still cost more than the scan it would replace once it
763
+ // spans enough of the table.
764
+ const scanPages = rootEstimate(scan, table, rowsPerPage).estPages;
765
+ const rangePages = rootEstimate(rangeScan, table, rowsPerPage).estPages;
766
+ if (rangePages >= scanPages) {
767
+ return {
768
+ rule: 'range-index-selection',
769
+ label: `\`${chosen.column}\` matches the predicate's range, but the estimated cost disagrees: the range walk would read about ${String(rangePages)} page${rangePages === 1 ? '' : 's'} against the scan's ${String(scanPages)} — no cheaper, so the scan stays and the whole predicate is rechecked there instead.`,
770
+ plan: input,
771
+ changed: new Set(),
772
+ whyNot: {
773
+ chosen: { label: 'a sequential scan', estPages: scanPages },
774
+ runnerUp: { label: `a range scan on \`${chosen.column}\``, estPages: rangePages },
775
+ },
776
+ };
777
+ }
778
+ const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, rangeScan)) });
779
+ const runnerUpNames = runnersUp.map((c) => `\`${c.column}\``).join(', ');
780
+ const choiceClause = runnersUp.length === 0
781
+ ? ''
782
+ : runnersUp.every((c) => c.selectivity.fraction === chosen.selectivity.fraction)
783
+ ? ` (tied with ${runnerUpNames} — \`${chosen.column}\` wins only because it was declared first, not because it narrows the search any further)`
784
+ : ` (chosen over ${runnerUpNames} — a narrower estimated range)`;
785
+ const boundsText = formatRangeBounds(chosen.column, chosen.low, chosen.high);
786
+ const indexedText = chosen.name === chosen.column
787
+ ? `\`${chosen.column}\` is indexed`
788
+ : `the composite index (${chosen.name}) is ordered by \`${chosen.column}\` first, so it can be entered there`;
789
+ return {
790
+ rule: 'range-index-selection',
791
+ label: residual
792
+ ? `${indexedText} and ${boundsText}${choiceClause}, so the scan becomes a single descent to the first matching leaf, then a walk along the leaf chain — no full table scan. The rest of the predicate is rechecked on each row the index returns.`
793
+ : `${indexedText} and the whole predicate is ${boundsText}${choiceClause}, so the scan becomes a single descent to the first matching leaf, then a walk along the leaf chain — no full table scan.`,
794
+ plan: root,
795
+ changed: new Set([rangeScan]),
796
+ whyNot: {
797
+ chosen: { label: `a range scan on \`${chosen.column}\``, estPages: rangePages },
798
+ runnerUp: { label: 'a sequential scan', estPages: scanPages },
799
+ },
800
+ };
801
+ };
802
+ /**
803
+ * When an `IndexScan`'s equality lookup already answers the query outright —
804
+ * every column the SELECT list needs is the indexed column itself, and
805
+ * there is no leftover conjunct to recheck against the row — the leaf's key
806
+ * *is* the answer. Runs after `index-selection`, since it rewrites the
807
+ * `IndexScan` that rule just produced; deliberately narrow (no `Sort` or
808
+ * aggregate node in between) rather than reaching through them the way the
809
+ * earlier rules do, since an index-only result is always exactly one row and
810
+ * there is nothing there yet worth the extra cases.
811
+ */
812
+ const indexOnlyScan = (input) => {
813
+ const { limit, rest: plan } = unwrapLimit(input);
814
+ if (plan.op !== 'Project' || plan.columns.kind !== 'columns')
815
+ return null;
816
+ if (plan.child.op !== 'IndexScan' || plan.child.residual)
817
+ return null;
818
+ const scan = plan.child;
819
+ // An index holds the values of every column it is on — one for a plain index, several for a composite one, which
820
+ // therefore *covers* any query that reads only those columns (plan.md §25.4 B3).
821
+ const covered = columnsOfIndex(scan.column);
822
+ const onlyIndexedColumns = plan.columns.names.every((name) => covered.includes(name));
823
+ if (!onlyIndexedColumns)
824
+ return null;
825
+ const indexOnly = {
826
+ op: 'IndexOnlyScan',
827
+ table: scan.table,
828
+ column: scan.column,
829
+ key: scan.key,
830
+ ...(scan.keys ? { keys: scan.keys } : {}),
831
+ };
832
+ const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: indexOnly });
833
+ const needs = plan.columns.names.map((n) => `\`${n}\``).join(', ');
834
+ return {
835
+ rule: 'index-only-scan',
836
+ label: covered.length > 1
837
+ ? `Every column this query needs — ${needs} — is one the composite index (${scan.column}) holds, so the index covers the query: the entries answer it without ever fetching the heap page the pointers name.`
838
+ : `Every column this query needs — \`${scan.column}\` — is already the index key, so the lookup answers it without ever fetching the heap page the pointer names.`,
839
+ plan: root,
840
+ changed: new Set([indexOnly]),
841
+ };
842
+ };
843
+ const RULES = [
844
+ { name: 'constant-folding', apply: constantFolding },
845
+ { name: 'predicate-pushdown', apply: predicatePushdown },
846
+ { name: 'join-predicate-pushdown', apply: joinPredicatePushdown },
847
+ { name: 'join-algorithm-selection', apply: joinAlgorithmSelection },
848
+ { name: 'index-selection', apply: indexSelection },
849
+ { name: 'range-index-selection', apply: rangeIndexSelection },
850
+ { name: 'index-only-scan', apply: indexOnlyScan },
851
+ ];
852
+ /** Every rewrite rule's name, in the order the optimizer runs them. */
853
+ export const OPTIMIZER_RULES = RULES.map((r) => r.name);
854
+ /**
855
+ * `DELETE`/`UPDATE` reuse this same optimizer (`emitDeletePlanEvents`, `emitUpdatePlanEvents`) to find the rows to touch,
856
+ * wrapped in a throwaway `Project *`. Until §25.4 B3b their executors could only run a `SeqScan` or an `IndexScan`,
857
+ * so `range-index-selection` was left out and a write could never use a range index; `exec/writeScan.ts` now runs an
858
+ * `IndexRangeScan` too, so every rule applies. (`index-only-scan` never fires on a write — the `Project *` wants every
859
+ * column — which is the point: a write reads the rows it changes.) Kept as its own export so the write path's rule set
860
+ * stays a named decision, not an accident of reusing `OPTIMIZER_RULES`.
861
+ */
862
+ export const WRITE_PATH_RULES = new Set(OPTIMIZER_RULES);
863
+ /**
864
+ * Applies every enabled rule that fires, in order. Returns each step for
865
+ * animation.
866
+ *
867
+ * `rowsPerPage` is threaded through to every rule (plan.md §25.4 C5): `index-selection`/`range-index-selection`
868
+ * need it to compare an index lookup's own estimated page cost against the scan it would replace, and
869
+ * `join-algorithm-selection` needs it (and `otherTables`) for the same reason, one table further out — see each
870
+ * rule's own doc comment. `otherTables` defaults to `{}` — every caller without a `JOIN` in the query omits it.
871
+ * `enabled` defaults to all rules — the pipeline (`emit.ts`) leaves it out except to keep
872
+ * `join-algorithm-selection` from overriding a `joinStrategy` explicitly requested. `/tools/planner` (plan.md
873
+ * §22.3) passes its own subset so any rule can be toggled off and its effect on the plan watched.
874
+ */
875
+ export function optimize(plan, table, rowsPerPage, otherTables = {}, enabled = new Set(OPTIMIZER_RULES)) {
876
+ const steps = [];
877
+ let current = plan;
878
+ for (const { name, apply } of RULES) {
879
+ if (!enabled.has(name))
880
+ continue;
881
+ const rewrite = apply(current, table, rowsPerPage, otherTables);
882
+ if (!rewrite)
883
+ continue;
884
+ steps.push(rewrite);
885
+ current = rewrite.plan;
886
+ }
887
+ return steps;
888
+ }
889
+ /** The predicate a scan will evaluate, whatever form it ended up in. */
890
+ export function scanPredicate(plan) {
891
+ if (plan.op === 'SeqScan')
892
+ return plan.filter;
893
+ if (plan.op === 'IndexScan' || plan.op === 'IndexRangeScan')
894
+ return plan.residual;
895
+ return plan.op === 'Filter' ||
896
+ plan.op === 'Having' ||
897
+ plan.op === 'HashDistinct' ||
898
+ plan.op === 'SortDistinct' ||
899
+ plan.op === 'HashAggregate' ||
900
+ plan.op === 'SortAggregate' ||
901
+ plan.op === 'Sort' ||
902
+ plan.op === 'Project' ||
903
+ plan.op === 'Limit'
904
+ ? scanPredicate(plan.child)
905
+ : undefined;
906
+ }