querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,906 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Rule-based rewrites. Most rules still consult no cost model at all: constant folding and predicate pushdown
|
|
3
|
+
* fire on a structural match alone, and never lose to a scan or another shape. Two kinds of rule are the
|
|
4
|
+
* exception, both plan.md §25.4 C5. `index-selection`/`range-index-selection` (C5 slice 1) pick a candidate by
|
|
5
|
+
* shape first, then decline it if its own estimated cost turns out not to beat the scan it would replace — a
|
|
6
|
+
* candidate that matches the predicate can still lose. `join-algorithm-selection` (C5 slice 2) has no shape step
|
|
7
|
+
* at all: every applicable algorithm is costed and the cheapest wins outright, the same as a real optimizer's
|
|
8
|
+
* plan enumeration. The cost *estimates* behind every one of these live in `cost.ts` (`rootEstimate`, the same
|
|
9
|
+
* formula `EXPLAIN` shows) — never a number invented for the comparison alone (§12).
|
|
10
|
+
*
|
|
11
|
+
* Rules run in order, each on the output of the last, and each produces one
|
|
12
|
+
* `optimize` event carrying the before and after trees.
|
|
13
|
+
*/
|
|
14
|
+
import { indexSelectionPrompt, leftmostPrefixPrompt } from "../predict.js";
|
|
15
|
+
import { describeKeys as describeKeysText } from "../index/lookup.js";
|
|
16
|
+
import { columnsOfIndex, indexNameOf, indexSpecsOf, isClusteringIndex, joinIndexFor } from "../index/spec.js";
|
|
17
|
+
import { mapChildren, walkExpr } from "../parser/index.js";
|
|
18
|
+
import { arithmetic, negate } from "../value.js";
|
|
19
|
+
import { conjoin, conjuncts, equalityOn, exprToSql, formatRangeBounds, lookupText, rangeOn } from "./plan.js";
|
|
20
|
+
import { boundedRangeSelectivity, ndistinct, rootEstimate } from "./cost.js";
|
|
21
|
+
function unwrapLimit(plan) {
|
|
22
|
+
let rest = plan;
|
|
23
|
+
let limit = null;
|
|
24
|
+
let distinct = null;
|
|
25
|
+
if (rest.op === 'Limit') {
|
|
26
|
+
limit = { count: rest.count, offset: rest.offset ?? 0 };
|
|
27
|
+
rest = rest.child;
|
|
28
|
+
}
|
|
29
|
+
if (rest.op === 'HashDistinct' || rest.op === 'SortDistinct') {
|
|
30
|
+
distinct = rest;
|
|
31
|
+
rest = rest.child;
|
|
32
|
+
}
|
|
33
|
+
return { limit: limit === null && distinct === null ? null : { limit, distinct }, rest };
|
|
34
|
+
}
|
|
35
|
+
function rewrapLimit(top, plan) {
|
|
36
|
+
if (top === null)
|
|
37
|
+
return plan;
|
|
38
|
+
let out = plan;
|
|
39
|
+
if (top.distinct) {
|
|
40
|
+
out = top.distinct.op === 'HashDistinct'
|
|
41
|
+
? { op: 'HashDistinct', child: out }
|
|
42
|
+
: { op: 'SortDistinct', ...(top.distinct.sortKey ? { sortKey: top.distinct.sortKey } : {}), child: out };
|
|
43
|
+
}
|
|
44
|
+
if (top.limit) {
|
|
45
|
+
out = { op: 'Limit', count: top.limit.count, ...(top.limit.offset > 0 ? { offset: top.limit.offset } : {}), child: out };
|
|
46
|
+
}
|
|
47
|
+
return out;
|
|
48
|
+
}
|
|
49
|
+
function unwrapSort(plan) {
|
|
50
|
+
return plan.op === 'Sort'
|
|
51
|
+
? { sort: { column: plan.column, direction: plan.direction }, rest: plan.child }
|
|
52
|
+
: { sort: null, rest: plan };
|
|
53
|
+
}
|
|
54
|
+
function rewrapSort(sort, plan) {
|
|
55
|
+
return sort === null ? plan : { op: 'Sort', column: sort.column, direction: sort.direction, child: plan };
|
|
56
|
+
}
|
|
57
|
+
function unwrapAggregate(plan) {
|
|
58
|
+
if (plan.op === 'Having' && (plan.child.op === 'HashAggregate' || plan.child.op === 'SortAggregate')) {
|
|
59
|
+
return { aggregate: { node: plan.child, having: plan }, rest: plan.child.child };
|
|
60
|
+
}
|
|
61
|
+
return plan.op === 'HashAggregate' || plan.op === 'SortAggregate'
|
|
62
|
+
? { aggregate: { node: plan, having: null }, rest: plan.child }
|
|
63
|
+
: { aggregate: null, rest: plan };
|
|
64
|
+
}
|
|
65
|
+
function rewrapAggregate(aggregate, plan) {
|
|
66
|
+
if (aggregate === null)
|
|
67
|
+
return plan;
|
|
68
|
+
const { node, having } = aggregate;
|
|
69
|
+
const rebuilt = node.op === 'HashAggregate'
|
|
70
|
+
? { op: 'HashAggregate', groupBy: node.groupBy, aggregates: node.aggregates, child: plan }
|
|
71
|
+
: { op: 'SortAggregate', groupBy: node.groupBy, aggregates: node.aggregates, child: plan };
|
|
72
|
+
return having ? { op: 'Having', predicate: having.predicate, child: rebuilt } : rebuilt;
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* Statically compares two literals — the same three-valued logic
|
|
76
|
+
* `exec/evaluate.ts`'s `compare` uses at runtime (kept as a small separate
|
|
77
|
+
* copy rather than an import, since `planner/` never depends on `exec/`).
|
|
78
|
+
* NULL on either side is UNKNOWN, not foldable.
|
|
79
|
+
*/
|
|
80
|
+
function foldLiteralCompare(op, left, right) {
|
|
81
|
+
if (left === null || right === null)
|
|
82
|
+
return null;
|
|
83
|
+
if (typeof left === 'number' && typeof right === 'number')
|
|
84
|
+
return orderedCompare(op, left, right);
|
|
85
|
+
const a = typeof left === 'boolean' ? String(Number(left)) : String(left);
|
|
86
|
+
const b = typeof right === 'boolean' ? String(Number(right)) : String(right);
|
|
87
|
+
return orderedCompare(op, a, b);
|
|
88
|
+
}
|
|
89
|
+
function orderedCompare(op, a, b) {
|
|
90
|
+
switch (op) {
|
|
91
|
+
case '=':
|
|
92
|
+
return a === b;
|
|
93
|
+
case '<>':
|
|
94
|
+
return a !== b;
|
|
95
|
+
case '<':
|
|
96
|
+
return a < b;
|
|
97
|
+
case '>':
|
|
98
|
+
return a > b;
|
|
99
|
+
case '<=':
|
|
100
|
+
return a <= b;
|
|
101
|
+
case '>=':
|
|
102
|
+
return a >= b;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* A comparison whose truth value does not depend on the row at all: two
|
|
107
|
+
* literals (`5 > 3`). Anything else — a column against a literal, or two
|
|
108
|
+
* columns — depends on the data and cannot be folded.
|
|
109
|
+
*
|
|
110
|
+
* That includes a column against *itself*. `age = age` looks like a
|
|
111
|
+
* tautology, but under three-valued logic it is UNKNOWN, not TRUE, on a row
|
|
112
|
+
* where `age` is NULL — and only TRUE passes a filter. Folding it away
|
|
113
|
+
* returned (and, in a `DELETE`, destroyed) exactly the rows the predicate
|
|
114
|
+
* excludes. The SQLite oracle found this (plan.md §25.4 A1). The correct
|
|
115
|
+
* rewrite is `age IS NOT NULL`, which needs an `IS NULL` predicate the
|
|
116
|
+
* engine did not have; until it does, the comparison is simply left alone.
|
|
117
|
+
*/
|
|
118
|
+
function foldComparison(expr) {
|
|
119
|
+
const { op, left, right } = expr;
|
|
120
|
+
if (left.kind === 'literal' && right.kind === 'literal') {
|
|
121
|
+
const result = foldLiteralCompare(op, left.value, right.value);
|
|
122
|
+
return result === null ? null : result ? 'true' : 'false';
|
|
123
|
+
}
|
|
124
|
+
return null;
|
|
125
|
+
}
|
|
126
|
+
/** Arithmetic built only from literals — `2 + 3`, `-(4 * 5)` — which needs no row to evaluate. */
|
|
127
|
+
function isConstantArithmetic(expr) {
|
|
128
|
+
if (expr.kind === 'literal')
|
|
129
|
+
return true;
|
|
130
|
+
if (expr.kind === 'arith')
|
|
131
|
+
return isConstantArithmetic(expr.left) && isConstantArithmetic(expr.right);
|
|
132
|
+
if (expr.kind === 'neg')
|
|
133
|
+
return isConstantArithmetic(expr.operand);
|
|
134
|
+
return false;
|
|
135
|
+
}
|
|
136
|
+
function literalFor(value, span) {
|
|
137
|
+
return { kind: 'literal', value, raw: value === null ? 'NULL' : String(value), span };
|
|
138
|
+
}
|
|
139
|
+
function evaluateConstant(expr) {
|
|
140
|
+
if (expr.kind === 'literal')
|
|
141
|
+
return expr.value;
|
|
142
|
+
if (expr.kind === 'neg')
|
|
143
|
+
return negate(evaluateConstant(expr.operand));
|
|
144
|
+
if (expr.kind === 'arith')
|
|
145
|
+
return arithmetic(expr.op, evaluateConstant(expr.left), evaluateConstant(expr.right));
|
|
146
|
+
return null;
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* Replaces every *maximal* constant-arithmetic subtree with its value
|
|
150
|
+
* (plan.md §25.4 B1b), recording what it folded. `WHERE id = 2 + 3` becomes
|
|
151
|
+
* `WHERE id = 5` — which is also what lets `index-selection` see an equality
|
|
152
|
+
* on a literal at all. Outermost only: `2 + 3 * 4` is one fold to 14, not two.
|
|
153
|
+
*/
|
|
154
|
+
function foldArithmetic(expr, folds) {
|
|
155
|
+
if ((expr.kind === 'arith' || expr.kind === 'neg') && isConstantArithmetic(expr)) {
|
|
156
|
+
const folded = literalFor(evaluateConstant(expr), expr.span);
|
|
157
|
+
folds.push({ before: exprToSql(expr), after: folded.kind === 'literal' ? folded.raw : '' });
|
|
158
|
+
return folded;
|
|
159
|
+
}
|
|
160
|
+
return mapChildren(expr, (child) => foldArithmetic(child, folds));
|
|
161
|
+
}
|
|
162
|
+
/**
|
|
163
|
+
* Two simplifications, both about work that does not depend on the row:
|
|
164
|
+
*
|
|
165
|
+
* - constant arithmetic (`2 + 3`) is worked out once, here, instead of for
|
|
166
|
+
* every row the scan reads (and so a literal can reach an index rule);
|
|
167
|
+
* - a conjunct that is always true (`5 > 3`) is dropped — `X AND true` is `X`.
|
|
168
|
+
*
|
|
169
|
+
* Deliberately does not act on an always-false conjunct: `X AND false` is
|
|
170
|
+
* `false`, but v0's plan shape has no node for "this scan produces no rows"
|
|
171
|
+
* without inventing one, so a `1 = 2`-shaped clause is left exactly as
|
|
172
|
+
* written. The query still runs correctly either way — this rule only ever
|
|
173
|
+
* removes work that was already redundant, never changes what a query
|
|
174
|
+
* returns.
|
|
175
|
+
*/
|
|
176
|
+
const constantFolding = (input) => {
|
|
177
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
178
|
+
if (plan.op !== 'Project')
|
|
179
|
+
return null;
|
|
180
|
+
const { sort, rest: belowSort } = unwrapSort(plan.child);
|
|
181
|
+
const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
|
|
182
|
+
if (belowProject.op !== 'Filter')
|
|
183
|
+
return null;
|
|
184
|
+
const folds = [];
|
|
185
|
+
const parts = conjuncts(belowProject.predicate).map((part) => foldArithmetic(part, folds));
|
|
186
|
+
const alwaysTrue = parts
|
|
187
|
+
.map((p, i) => ({ i, expr: p }))
|
|
188
|
+
.filter(({ expr }) => expr.kind === 'compare' && foldComparison(expr) === 'true');
|
|
189
|
+
if (alwaysTrue.length === 0 && folds.length === 0)
|
|
190
|
+
return null;
|
|
191
|
+
const dropped = new Set(alwaysTrue.map((f) => f.i));
|
|
192
|
+
const residual = conjoin(parts.filter((_, i) => !dropped.has(i)));
|
|
193
|
+
const rewrittenFilter = residual
|
|
194
|
+
? { op: 'Filter', predicate: residual, child: belowProject.child }
|
|
195
|
+
: belowProject.child;
|
|
196
|
+
const root = rewrapLimit(limit, {
|
|
197
|
+
op: 'Project',
|
|
198
|
+
columns: plan.columns,
|
|
199
|
+
child: rewrapSort(sort, rewrapAggregate(aggregate, rewrittenFilter)),
|
|
200
|
+
});
|
|
201
|
+
const foldSentence = folds.length === 0
|
|
202
|
+
? ''
|
|
203
|
+
: `${folds.map((f) => `\`${f.before}\` is just ${f.after}`).join(', ')} — worked out once here instead of for every row`;
|
|
204
|
+
const clauses = alwaysTrue.map((f) => `\`${exprToSql(f.expr)}\``).join(', ');
|
|
205
|
+
const verb = alwaysTrue.length === 1 ? 'is' : 'are';
|
|
206
|
+
const dropSentence = alwaysTrue.length === 0
|
|
207
|
+
? ''
|
|
208
|
+
: residual
|
|
209
|
+
? `${clauses} ${verb} always true, whatever the row — it drops out of the predicate, leaving only the part that actually depends on data`
|
|
210
|
+
: `${clauses} ${verb} always true, whatever the row — with nothing else in the WHERE clause, the filter disappears entirely`;
|
|
211
|
+
return {
|
|
212
|
+
rule: 'constant-folding',
|
|
213
|
+
label: `Constant folding: ${[foldSentence, dropSentence].filter(Boolean).join('; and ')}.`,
|
|
214
|
+
plan: root,
|
|
215
|
+
changed: new Set([rewrittenFilter]),
|
|
216
|
+
};
|
|
217
|
+
};
|
|
218
|
+
/**
|
|
219
|
+
* Fold `Filter` into the scan beneath it.
|
|
220
|
+
*
|
|
221
|
+
* In a single-table query the filter already sits directly above the scan, so
|
|
222
|
+
* there is nothing to push it *past*. The real move is pushing it *into* the
|
|
223
|
+
* scan: the scan then discards non-matching rows as it reads them, instead of
|
|
224
|
+
* materialising every row for a separate operator to throw away.
|
|
225
|
+
*/
|
|
226
|
+
const predicatePushdown = (input) => {
|
|
227
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
228
|
+
if (plan.op !== 'Project')
|
|
229
|
+
return null;
|
|
230
|
+
const { sort, rest: belowSort } = unwrapSort(plan.child);
|
|
231
|
+
const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
|
|
232
|
+
if (belowProject.op !== 'Filter')
|
|
233
|
+
return null;
|
|
234
|
+
const scan = belowProject.child;
|
|
235
|
+
if (scan.op !== 'SeqScan' || scan.filter)
|
|
236
|
+
return null;
|
|
237
|
+
const pushed = { op: 'SeqScan', table: scan.table, filter: belowProject.predicate };
|
|
238
|
+
const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, pushed)) });
|
|
239
|
+
return {
|
|
240
|
+
rule: 'predicate-pushdown',
|
|
241
|
+
label: 'Predicate pushdown: the filter moves into the scan, so rows are discarded as they are read rather than materialised and then thrown away.',
|
|
242
|
+
plan: root,
|
|
243
|
+
changed: new Set([pushed]),
|
|
244
|
+
};
|
|
245
|
+
};
|
|
246
|
+
/** Every table a qualified column reference inside `expr` names. Empty for a literal. */
|
|
247
|
+
function tablesReferencedBy(expr) {
|
|
248
|
+
const tables = new Set();
|
|
249
|
+
walkExpr(expr, (e) => {
|
|
250
|
+
if (e.kind === 'column' && e.table)
|
|
251
|
+
tables.add(e.table);
|
|
252
|
+
});
|
|
253
|
+
return tables;
|
|
254
|
+
}
|
|
255
|
+
/**
|
|
256
|
+
* Clears every column's table qualifier. A conjunct pushed into one side's own
|
|
257
|
+
* `SeqScan` runs against that table's *raw* rows — `{ id: 1, name: 'ada' }`,
|
|
258
|
+
* never `{ 'authors.id': 1, ... }` — qualification only exists to route a
|
|
259
|
+
* `WHERE` clause to the right side before the rewrite, the same way it exists
|
|
260
|
+
* in the SQL text only to route the reference to the right table at parse
|
|
261
|
+
* time. Left qualified, `authors.name = 'ada'` would read `row['authors.name']`
|
|
262
|
+
* off a row that has no such key and silently match nothing.
|
|
263
|
+
*/
|
|
264
|
+
function stripTableQualifiers(expr) {
|
|
265
|
+
if (expr.kind === 'column') {
|
|
266
|
+
if (!expr.table)
|
|
267
|
+
return expr;
|
|
268
|
+
const { table: _table, ...rest } = expr;
|
|
269
|
+
return rest;
|
|
270
|
+
}
|
|
271
|
+
return mapChildren(expr, stripTableQualifiers);
|
|
272
|
+
}
|
|
273
|
+
/**
|
|
274
|
+
* Pushes each `WHERE` conjunct that names only one side of a `JOIN` into that
|
|
275
|
+
* side's own `SeqScan` — `predicatePushdown`'s counterpart for a `Join`'s two
|
|
276
|
+
* children instead of a single scan. A conjunct naming both sides can only be
|
|
277
|
+
* evaluated once a row from each side is already paired, so it stays behind
|
|
278
|
+
* as a residual `Filter` above the `Join`; every reference is qualified once
|
|
279
|
+
* a query has a `JOIN` at all (the parser guarantees it), so there is no
|
|
280
|
+
* unqualified conjunct to worry about routing.
|
|
281
|
+
*
|
|
282
|
+
* Deliberately narrow: it does not also run `index-selection` on the side it
|
|
283
|
+
* just filtered — that rule only ever looks at the very top of the tree
|
|
284
|
+
* (`Project → … → SeqScan`), not inside a `Join`'s own children, so a pushed
|
|
285
|
+
* conjunct that happens to match an index still runs as a full scan. Reaching
|
|
286
|
+
* `index-selection` into a `Join`'s children is its own, larger step.
|
|
287
|
+
*
|
|
288
|
+
* Also deliberately narrow about *which* `Join`: `join.left.op !== 'SeqScan'` bails on a chain of two or more
|
|
289
|
+
* `JOIN`s (plan.md §25.4 C3 slice b) just as it would on any other shape this rule does not recognise — a chain's
|
|
290
|
+
* outermost `Join` has another `Join` as its `left`, never a bare scan. A 3-or-more-table query's `WHERE` clause
|
|
291
|
+
* therefore always stays a residual `Filter` above the whole chain for now, exactly like a predicate this rule
|
|
292
|
+
* already declines to reach *into* a single `Join`'s own children (the paragraph above) — a missed optimization,
|
|
293
|
+
* not a correctness gap; extending pushdown down a chain is left for later, alongside reaching it past a `Join` at all.
|
|
294
|
+
*
|
|
295
|
+
* A conjunct on the *inner* side of a `LEFT JOIN` (plan.md §25.4 C3) is never pushed, on purpose: filtering it into
|
|
296
|
+
* the inner scan removes rows *before* the join can even find out they would not have matched, so a left row that
|
|
297
|
+
* *would* have been NULL-padded instead vanishes with no trace of it — an inner join wearing a LEFT JOIN's syntax.
|
|
298
|
+
* Left as a residual `Filter` above the `Join` instead, it runs after the padding, exactly where SQL puts it. The
|
|
299
|
+
* outer (left) side has no such trap — filtering it first only ever removes rows the join would keep anyway.
|
|
300
|
+
*/
|
|
301
|
+
const joinPredicatePushdown = (input) => {
|
|
302
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
303
|
+
if (plan.op !== 'Project')
|
|
304
|
+
return null;
|
|
305
|
+
const { sort, rest: belowSort } = unwrapSort(plan.child);
|
|
306
|
+
const { aggregate, rest: belowProject } = unwrapAggregate(belowSort);
|
|
307
|
+
if (belowProject.op !== 'Filter')
|
|
308
|
+
return null;
|
|
309
|
+
const join = belowProject.child;
|
|
310
|
+
// The inner side is a bare `SeqScan` — or, for an index nested-loop join, an `IndexProbe`, which has no scan to
|
|
311
|
+
// filter: a conjunct naming only the inner table then stays above the Join as a residual, never pushed.
|
|
312
|
+
const probesInner = join.op === 'Join' && join.right.op === 'IndexProbe';
|
|
313
|
+
if (join.op !== 'Join' ||
|
|
314
|
+
join.left.op !== 'SeqScan' ||
|
|
315
|
+
join.left.filter ||
|
|
316
|
+
(join.right.op !== 'SeqScan' && join.right.op !== 'IndexProbe') ||
|
|
317
|
+
(join.right.op === 'SeqScan' && join.right.filter)) {
|
|
318
|
+
return null;
|
|
319
|
+
}
|
|
320
|
+
const parts = conjuncts(belowProject.predicate);
|
|
321
|
+
const leftParts = [];
|
|
322
|
+
const rightParts = [];
|
|
323
|
+
const residualParts = [];
|
|
324
|
+
for (const part of parts) {
|
|
325
|
+
const tables = tablesReferencedBy(part);
|
|
326
|
+
if (tables.size === 1 && tables.has(join.leftTable))
|
|
327
|
+
leftParts.push(part);
|
|
328
|
+
else if (tables.size === 1 && tables.has(join.rightTable) && !probesInner && join.joinType !== 'left')
|
|
329
|
+
rightParts.push(part);
|
|
330
|
+
else
|
|
331
|
+
residualParts.push(part);
|
|
332
|
+
}
|
|
333
|
+
if (leftParts.length === 0 && rightParts.length === 0)
|
|
334
|
+
return null;
|
|
335
|
+
const newLeft = leftParts.length > 0
|
|
336
|
+
? { op: 'SeqScan', table: join.leftTable, filter: stripTableQualifiers(conjoin(leftParts)) }
|
|
337
|
+
: join.left;
|
|
338
|
+
const newRight = rightParts.length > 0
|
|
339
|
+
? { op: 'SeqScan', table: join.rightTable, filter: stripTableQualifiers(conjoin(rightParts)) }
|
|
340
|
+
: join.right;
|
|
341
|
+
const newJoin = { ...join, left: newLeft, right: newRight };
|
|
342
|
+
const residual = conjoin(residualParts);
|
|
343
|
+
const filtered = residual ? { op: 'Filter', predicate: residual, child: newJoin } : newJoin;
|
|
344
|
+
const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, filtered)) });
|
|
345
|
+
const pushedSides = [
|
|
346
|
+
leftParts.length > 0 ? join.leftTable : null,
|
|
347
|
+
rightParts.length > 0 ? join.rightTable : null,
|
|
348
|
+
].filter((t) => t !== null);
|
|
349
|
+
const changed = new Set([newJoin, ...(leftParts.length > 0 ? [newLeft] : []), ...(rightParts.length > 0 ? [newRight] : [])]);
|
|
350
|
+
return {
|
|
351
|
+
rule: 'join-predicate-pushdown',
|
|
352
|
+
label: residual
|
|
353
|
+
? `Predicate pushdown: the part${pushedSides.length === 1 ? '' : 's'} of the WHERE clause naming only ${pushedSides.join(' or only ')} move${pushedSides.length === 1 ? 's' : ''} into that scan; the rest can only be checked once a row from each side is paired, so it stays a residual Filter above the Join.`
|
|
354
|
+
: `Predicate pushdown: every part of the WHERE clause names only one side, so the whole thing moves into ${pushedSides.length === 1 ? 'that scan' : 'the two scans'} — nothing is left to recheck after the Join.`,
|
|
355
|
+
plan: root,
|
|
356
|
+
changed,
|
|
357
|
+
};
|
|
358
|
+
};
|
|
359
|
+
/** A join algorithm's name for narration — `emit.ts`'s own `describeJoin` is phrase-shaped for a different sentence, so this stays a small local copy rather than an import that would need reshaping either way. */
|
|
360
|
+
function joinAlgorithmName(algorithm) {
|
|
361
|
+
switch (algorithm) {
|
|
362
|
+
case 'nested-loop':
|
|
363
|
+
return 'nested-loop';
|
|
364
|
+
case 'hash':
|
|
365
|
+
return 'hash join';
|
|
366
|
+
case 'sort-merge':
|
|
367
|
+
return 'sort-merge';
|
|
368
|
+
case 'index-nested-loop':
|
|
369
|
+
return 'index nested-loop';
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
/**
|
|
373
|
+
* Recomputes one `Join` node's own algorithm by cost, after first doing the same for whatever `Join` sits beneath
|
|
374
|
+
* it in a chain (plan.md §25.4 C3 slice b) — an earlier step's own chosen algorithm changes its own estimated page
|
|
375
|
+
* cost, which a later step's comparison then has to use, the same way a real cost-based join enumerator prices a
|
|
376
|
+
* whole left-deep chain bottom-up. `right`'s shape changes with the candidate: a bare `SeqScan` for nested-loop,
|
|
377
|
+
* hash and sort-merge, an `IndexProbe` only for index-nested-loop — the same swap `buildJoinPlan` itself makes.
|
|
378
|
+
*/
|
|
379
|
+
function chooseJoinAlgorithms(join, table, rowsPerPage, otherTables) {
|
|
380
|
+
const tableNamed = (name) => (name === table.name ? table : otherTables[name]);
|
|
381
|
+
const below = join.left.op === 'Join' ? chooseJoinAlgorithms(join.left, table, rowsPerPage, otherTables) : null;
|
|
382
|
+
const newLeft = below ? below.plan : join.left;
|
|
383
|
+
const rightTableData = tableNamed(join.rightTable);
|
|
384
|
+
// Reused as-is, not rebuilt bare: `join-predicate-pushdown` (which already ran, earlier in `RULES`) may have
|
|
385
|
+
// already pushed a WHERE conjunct into this exact scan — losing that filter here would silently widen the
|
|
386
|
+
// join's own results, not just its cost estimate.
|
|
387
|
+
const plainRight = join.right.op === 'SeqScan' ? join.right : { op: 'SeqScan', table: join.rightTable };
|
|
388
|
+
const rightHasPushedFilter = plainRight.op === 'SeqScan' && plainRight.filter !== undefined;
|
|
389
|
+
// `IndexProbe` has nowhere to carry a residual filter, so a pushed-down conjunct on this side rules index
|
|
390
|
+
// nested-loop out entirely rather than risk silently dropping it — the same reason `join-predicate-pushdown`
|
|
391
|
+
// itself never pushes a conjunct into a side that already probes an index (its own `probesInner` check).
|
|
392
|
+
const idxSpec = rightHasPushedFilter ? undefined : joinIndexFor(rightTableData, join.rightColumn);
|
|
393
|
+
const algorithms = idxSpec
|
|
394
|
+
? ['nested-loop', 'hash', 'sort-merge', 'index-nested-loop']
|
|
395
|
+
: ['nested-loop', 'hash', 'sort-merge'];
|
|
396
|
+
const candidates = algorithms.map((algorithm) => {
|
|
397
|
+
const right = algorithm === 'index-nested-loop'
|
|
398
|
+
? { op: 'IndexProbe', table: join.rightTable, column: indexNameOf(idxSpec.columns), outerTable: join.leftTable, outerColumn: join.leftColumn }
|
|
399
|
+
: plainRight;
|
|
400
|
+
const plan = { ...join, left: newLeft, right, algorithm };
|
|
401
|
+
return { algorithm, plan, estPages: rootEstimate(plan, table, rowsPerPage, otherTables).estPages };
|
|
402
|
+
});
|
|
403
|
+
// Ties favour whatever is already there (nested-loop, first in `algorithms`) — never rewrite just because two
|
|
404
|
+
// candidates happen to cost the same.
|
|
405
|
+
const chosen = candidates.reduce((best, c) => (c.estPages < best.estPages ? c : best));
|
|
406
|
+
const runnersUp = candidates.filter((c) => c !== chosen);
|
|
407
|
+
const changedHere = chosen.algorithm !== join.algorithm;
|
|
408
|
+
return {
|
|
409
|
+
plan: chosen.plan,
|
|
410
|
+
changes: [...(below?.changes ?? []), ...(changedHere ? [{ leftTable: join.leftTable, rightTable: join.rightTable, chosen, runnersUp }] : [])],
|
|
411
|
+
changedNodes: [...(below?.changedNodes ?? []), ...(changedHere ? [chosen.plan] : [])],
|
|
412
|
+
};
|
|
413
|
+
}
|
|
414
|
+
/**
|
|
415
|
+
* Chooses each `JOIN`'s physical algorithm by cost (plan.md §25.4 C5 slice 2) — the first rule with no shape step
|
|
416
|
+
* at all. `buildPlan.ts` always starts a chain out as `nested-loop`, its own cost-oblivious default, the same way
|
|
417
|
+
* it always starts a predicate out as a bare `SeqScan`; this rule prices every algorithm that could actually run
|
|
418
|
+
* each step — nested-loop, hash, sort-merge, and index-nested-loop whenever the inner table has a usable index —
|
|
419
|
+
* with `cost.ts`'s own per-algorithm formulas (the same ones `EXPLAIN` already shows), and keeps whichever is
|
|
420
|
+
* cheapest. Unlike `index-selection` there is no "matches the shape but costs more, so decline" outcome: every
|
|
421
|
+
* applicable algorithm is a real, working candidate, so the rule always ends up choosing one — it may just be the
|
|
422
|
+
* one already there, in which case nothing changed and this reports no rewrite, like any other rule that finds
|
|
423
|
+
* nothing worth doing.
|
|
424
|
+
*
|
|
425
|
+
* A chain (plan.md §25.4 C3 slice b) is priced bottom-up, one `Join` at a time — see `chooseJoinAlgorithms`.
|
|
426
|
+
*
|
|
427
|
+
* Only runs when the caller never forced a specific algorithm: `emit.ts` leaves this rule out of `enabled`
|
|
428
|
+
* whenever `EngineOptions.joinStrategy` was set explicitly — `/compare`'s and `/tools/planner`'s whole point is
|
|
429
|
+
* forcing one algorithm to see its own effect, which a cost comparison would otherwise silently override.
|
|
430
|
+
*/
|
|
431
|
+
const joinAlgorithmSelection = (input, table, rowsPerPage, otherTables) => {
|
|
432
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
433
|
+
if (plan.op !== 'Project')
|
|
434
|
+
return null;
|
|
435
|
+
const belowProject = plan.child;
|
|
436
|
+
const outerJoin = belowProject.op === 'Filter' ? belowProject.child : belowProject;
|
|
437
|
+
if (outerJoin.op !== 'Join')
|
|
438
|
+
return null;
|
|
439
|
+
const result = chooseJoinAlgorithms(outerJoin, table, rowsPerPage, otherTables);
|
|
440
|
+
if (result.changes.length === 0)
|
|
441
|
+
return null;
|
|
442
|
+
const newBelowProject = belowProject.op === 'Filter' ? { ...belowProject, child: result.plan } : result.plan;
|
|
443
|
+
const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: newBelowProject });
|
|
444
|
+
const label = result.changes
|
|
445
|
+
.map(({ leftTable, rightTable, chosen, runnersUp }) => {
|
|
446
|
+
const others = runnersUp.map((c) => `${joinAlgorithmName(c.algorithm)}'s ${String(c.estPages)}`).join(', ');
|
|
447
|
+
return `\`${leftTable}\` × \`${rightTable}\`: ${joinAlgorithmName(chosen.algorithm)} chosen by cost — an estimated ${String(chosen.estPages)} pages against ${others}.`;
|
|
448
|
+
})
|
|
449
|
+
.join(' ');
|
|
450
|
+
// The "Why not?" panel (plan.md §25.4 C5, fourth slice) names one runner-up, not the whole candidate set — the
|
|
451
|
+
// first changed step's own cheapest alternative, since a `Rewrite` carries a single `whyNot`, not one per step.
|
|
452
|
+
// A chain with several changed steps still gets the full comparison in `label` above; only the panel is scoped
|
|
453
|
+
// to the first step.
|
|
454
|
+
const firstChange = result.changes[0];
|
|
455
|
+
const cheapestRunnerUp = firstChange.runnersUp.reduce((best, c) => (c.estPages < best.estPages ? c : best));
|
|
456
|
+
return {
|
|
457
|
+
rule: 'join-algorithm-selection',
|
|
458
|
+
label,
|
|
459
|
+
plan: root,
|
|
460
|
+
changed: new Set(result.changedNodes),
|
|
461
|
+
whyNot: {
|
|
462
|
+
chosen: { label: joinAlgorithmName(firstChange.chosen.algorithm), estPages: firstChange.chosen.estPages },
|
|
463
|
+
runnerUp: { label: joinAlgorithmName(cheapestRunnerUp.algorithm), estPages: cheapestRunnerUp.estPages },
|
|
464
|
+
},
|
|
465
|
+
};
|
|
466
|
+
};
|
|
467
|
+
/** The lower and upper bound `parts` put on `column`, and which conjuncts they are — at most one of each direction. */
|
|
468
|
+
function rangeBoundsOn(parts, column, taken) {
|
|
469
|
+
const matches = parts
|
|
470
|
+
.map((p, i) => ({ i, match: taken.has(i) ? undefined : rangeOn(p, column) }))
|
|
471
|
+
.filter((m) => m.match !== undefined);
|
|
472
|
+
const lower = matches.find((m) => m.match.op === '>' || m.match.op === '>=');
|
|
473
|
+
const upper = matches.find((m) => m.match.op === '<' || m.match.op === '<=');
|
|
474
|
+
if (!lower && !upper)
|
|
475
|
+
return undefined;
|
|
476
|
+
const low = lower ? { value: lower.match.value, inclusive: lower.match.op === '>=' } : undefined;
|
|
477
|
+
const high = upper ? { value: upper.match.value, inclusive: upper.match.op === '<=' } : undefined;
|
|
478
|
+
return { column, ...(low ? { low } : {}), ...(high ? { high } : {}), consumed: [lower?.i, upper?.i].filter((i) => i !== undefined) };
|
|
479
|
+
}
|
|
480
|
+
/** The composite indexes `parts` cannot enter: it constrains one of their columns, but not their leftmost one. */
|
|
481
|
+
function unusableComposites(table, parts) {
|
|
482
|
+
return indexSpecsOf(table).flatMap((spec) => {
|
|
483
|
+
if (spec.columns.length < 2)
|
|
484
|
+
return [];
|
|
485
|
+
const constrains = (column) => parts.some((p) => equalityOn(p, column) !== undefined || rangeOn(p, column) !== undefined);
|
|
486
|
+
if (constrains(spec.columns[0]))
|
|
487
|
+
return [];
|
|
488
|
+
const named = spec.columns.slice(1).filter(constrains);
|
|
489
|
+
return named.length > 0 ? [{ name: indexNameOf(spec.columns), columns: spec.columns, named }] : [];
|
|
490
|
+
});
|
|
491
|
+
}
|
|
492
|
+
/**
|
|
493
|
+
* The leftmost-prefix rule, said aloud for whoever is looking at a plan that did *not* use an index it might have
|
|
494
|
+
* (plan.md §25.4 B3): a composite index on `(a, b)` is ordered by `a` first, so an equality on `b` alone matches entries
|
|
495
|
+
* scattered across the whole index — there is no run to read, and the index cannot be entered. `null` when there is
|
|
496
|
+
* nothing to say.
|
|
497
|
+
*/
|
|
498
|
+
export function leftmostPrefixNote(table, where) {
|
|
499
|
+
if (!where)
|
|
500
|
+
return null;
|
|
501
|
+
const skipped = unusableComposites(table, conjuncts(where));
|
|
502
|
+
if (skipped.length === 0)
|
|
503
|
+
return null;
|
|
504
|
+
return skipped
|
|
505
|
+
.map((c) => `The index on (${c.columns.join(', ')}) cannot help: the predicate constrains ${c.named.map((n) => `\`${n}\``).join(' and ')} but not \`${c.columns[0]}\`, and an index can only be entered through its leftmost column — its entries are ordered by \`${c.columns[0]}\` first, so ${c.named.length === 1 ? 'that value is' : 'those values are'} scattered across the whole index.`)
|
|
506
|
+
.join(' ');
|
|
507
|
+
}
|
|
508
|
+
/**
|
|
509
|
+
* Swap a `SeqScan` for an `IndexScan` when a conjunct is an equality on an
|
|
510
|
+
* indexed column. Conjuncts the index cannot answer stay behind as a
|
|
511
|
+
* recheck — the index narrows the search, it does not verify the whole
|
|
512
|
+
* predicate.
|
|
513
|
+
*
|
|
514
|
+
* A table can declare more than one index (plan.md §22.2's "second candidate
|
|
515
|
+
* index"). When two or more each have a matching equality conjunct, this
|
|
516
|
+
* picks the more selective one — higher `ndistinct`, so an equality on it
|
|
517
|
+
* matches fewer rows on average — the same fact the cost model's own
|
|
518
|
+
* equality-selectivity estimate already rests on, not a second, independent
|
|
519
|
+
* comparison. The conjunct that lost stays in the recheck, exactly like every
|
|
520
|
+
* other conjunct the chosen index can't answer — v0 has no way to intersect
|
|
521
|
+
* two index lookups.
|
|
522
|
+
*
|
|
523
|
+
* A **composite** index (plan.md §25.4 B3) is entered through a *leftmost
|
|
524
|
+
* prefix*: the run of its leading columns that each have an equality. An
|
|
525
|
+
* equality on `a` enters an index on `(a, b)`; one on `b` alone cannot, and the
|
|
526
|
+
* rule leaves that index unused — and says why. The more columns a candidate's
|
|
527
|
+
* prefix pins, the fewer rows it selects, so that is what the choice compares.
|
|
528
|
+
*
|
|
529
|
+
* **Chosen by cost, not just by shape (plan.md §25.4 C5)**: once the most
|
|
530
|
+
* selective *candidate* is picked, its own estimated page cost is compared
|
|
531
|
+
* against the scan it would replace — `rootEstimate` on each, the same
|
|
532
|
+
* formula `EXPLAIN` already shows. A candidate that matches the predicate's
|
|
533
|
+
* shape but would not actually be cheaper (a low-selectivity index — most of
|
|
534
|
+
* `logins.status`'s rows are `'ok'`, say — costs more one random heap fetch
|
|
535
|
+
* at a time than one sequential scan already would) is *declined*: the rule
|
|
536
|
+
* still reports what it considered and why it said no, but the plan stays a
|
|
537
|
+
* `SeqScan`. This is the one thing plan.md §12 always drew a line at (*"an
|
|
538
|
+
* invented figure with a decimal point is worse than an honest guess"*) —
|
|
539
|
+
* not that a rule could never consult a cost, only that doing so had to be
|
|
540
|
+
* real, not a rule of thumb pretending to be one. `rootEstimate` already is.
|
|
541
|
+
*/
|
|
542
|
+
const indexSelection = (input, table, rowsPerPage) => {
|
|
543
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
544
|
+
if (plan.op !== 'Project')
|
|
545
|
+
return null;
|
|
546
|
+
const { sort, rest: belowSort } = unwrapSort(plan.child);
|
|
547
|
+
const { aggregate, rest: scan } = unwrapAggregate(belowSort);
|
|
548
|
+
if (scan.op !== 'SeqScan' || !scan.filter)
|
|
549
|
+
return null;
|
|
550
|
+
const parts = conjuncts(scan.filter);
|
|
551
|
+
const candidates = indexSpecsOf(table).flatMap((spec) => {
|
|
552
|
+
const matched = [];
|
|
553
|
+
const used = new Set();
|
|
554
|
+
for (const column of spec.columns) {
|
|
555
|
+
// The run of matched columns stops at the first one with no equality — that is the leftmost-prefix rule.
|
|
556
|
+
const partIndex = parts.findIndex((p, i) => !used.has(i) && equalityOn(p, column) !== undefined);
|
|
557
|
+
if (partIndex === -1)
|
|
558
|
+
break;
|
|
559
|
+
used.add(partIndex);
|
|
560
|
+
matched.push({ column, partIndex, key: equalityOn(parts[partIndex], column) });
|
|
561
|
+
}
|
|
562
|
+
if (matched.length === 0)
|
|
563
|
+
return [];
|
|
564
|
+
const nextColumn = spec.columns[matched.length];
|
|
565
|
+
// A range *within* the prefix needs the entries' own row addresses to walk the leaf chain (`index/rangeLookup.ts`)
|
|
566
|
+
// — a clustered table's secondary index carries a clustering key instead (plan.md §25.4 B3e), so this narrowing
|
|
567
|
+
// is only offered for the clustering key's own index, or when the table isn't clustered at all.
|
|
568
|
+
const clusterSafe = !table.clusteredKey || isClusteringIndex(table.clusteredKey, spec.columns);
|
|
569
|
+
const next = nextColumn === undefined || !clusterSafe ? undefined : rangeBoundsOn(parts, nextColumn, used);
|
|
570
|
+
return [
|
|
571
|
+
{
|
|
572
|
+
name: indexNameOf(spec.columns),
|
|
573
|
+
columns: spec.columns,
|
|
574
|
+
unique: spec.unique === true,
|
|
575
|
+
matched,
|
|
576
|
+
fraction: matched.reduce((product, m) => product / ndistinct(table, m.column), 1),
|
|
577
|
+
...(next ? { next } : {}),
|
|
578
|
+
},
|
|
579
|
+
];
|
|
580
|
+
});
|
|
581
|
+
if (candidates.length === 0)
|
|
582
|
+
return null;
|
|
583
|
+
const distinctOf = (c) => Math.round(1 / c.fraction);
|
|
584
|
+
// Narrower wins: a smaller fraction, then more columns pinned, then a range on the next column to narrow the run further.
|
|
585
|
+
const better = (c, best) => c.fraction !== best.fraction
|
|
586
|
+
? c.fraction < best.fraction
|
|
587
|
+
: c.matched.length !== best.matched.length
|
|
588
|
+
? c.matched.length > best.matched.length
|
|
589
|
+
: c.next !== undefined && best.next === undefined;
|
|
590
|
+
const chosen = candidates.reduce((best, c) => (better(c, best) ? c : best));
|
|
591
|
+
const runnersUp = candidates.filter((c) => c !== chosen);
|
|
592
|
+
const consumed = new Set([...chosen.matched.map((m) => m.partIndex), ...(chosen.next?.consumed ?? [])]);
|
|
593
|
+
const residual = conjoin(parts.filter((_, i) => !consumed.has(i)));
|
|
594
|
+
const keys = chosen.matched.map((m) => m.key);
|
|
595
|
+
const composite = chosen.columns.length > 1;
|
|
596
|
+
// An equality prefix with a range on the next column becomes a range scan *within* the prefix's run; otherwise a lookup.
|
|
597
|
+
const indexScan = chosen.next
|
|
598
|
+
? {
|
|
599
|
+
op: 'IndexRangeScan',
|
|
600
|
+
table: scan.table,
|
|
601
|
+
column: chosen.name,
|
|
602
|
+
prefix: keys,
|
|
603
|
+
...(chosen.next.low ? { low: chosen.next.low } : {}),
|
|
604
|
+
...(chosen.next.high ? { high: chosen.next.high } : {}),
|
|
605
|
+
...(residual ? { residual } : {}),
|
|
606
|
+
}
|
|
607
|
+
: {
|
|
608
|
+
op: 'IndexScan',
|
|
609
|
+
table: scan.table,
|
|
610
|
+
column: chosen.name,
|
|
611
|
+
key: keys[0],
|
|
612
|
+
...(composite ? { keys } : {}),
|
|
613
|
+
...(residual ? { residual } : {}),
|
|
614
|
+
};
|
|
615
|
+
const nameOf = (c) => (c.columns.length > 1 ? `(${c.columns.join(', ')})` : `\`${c.name}\``);
|
|
616
|
+
// Chosen by cost, not just by shape (plan.md §25.4 C5): a candidate that matches the predicate can still lose to
|
|
617
|
+
// the scan it would replace — a low-selectivity index costs one random heap fetch per match, which adds up past
|
|
618
|
+
// whatever one sequential scan already costs. `rootEstimate` on each side is the same formula `EXPLAIN` shows,
|
|
619
|
+
// not a new heuristic invented for this comparison alone.
|
|
620
|
+
const scanPages = rootEstimate(scan, table, rowsPerPage).estPages;
|
|
621
|
+
const indexPages = rootEstimate(indexScan, table, rowsPerPage).estPages;
|
|
622
|
+
if (indexPages >= scanPages) {
|
|
623
|
+
return {
|
|
624
|
+
rule: 'index-selection',
|
|
625
|
+
label: `${nameOf(chosen)} matches the predicate, but the estimated cost disagrees: the lookup would read about ${String(indexPages)} page${indexPages === 1 ? '' : 's'} against the scan's ${String(scanPages)} — no cheaper, so the scan stays and the whole predicate is rechecked there instead.`,
|
|
626
|
+
plan: input,
|
|
627
|
+
changed: new Set(),
|
|
628
|
+
whyNot: {
|
|
629
|
+
chosen: { label: 'a sequential scan', estPages: scanPages },
|
|
630
|
+
runnerUp: { label: `${nameOf(chosen)} index lookup`, estPages: indexPages },
|
|
631
|
+
},
|
|
632
|
+
};
|
|
633
|
+
}
|
|
634
|
+
const root = rewrapLimit(limit, {
|
|
635
|
+
op: 'Project',
|
|
636
|
+
columns: plan.columns,
|
|
637
|
+
child: rewrapSort(sort, rewrapAggregate(aggregate, indexScan)),
|
|
638
|
+
});
|
|
639
|
+
const chosenDistinct = distinctOf(chosen);
|
|
640
|
+
const runnerUpNames = runnersUp.map(nameOf).join(', ');
|
|
641
|
+
const runnerUpDistinct = runnersUp.map(distinctOf);
|
|
642
|
+
const choiceClause = runnersUp.length === 0
|
|
643
|
+
? ''
|
|
644
|
+
: runnerUpDistinct.every((d) => d === chosenDistinct)
|
|
645
|
+
? ` (tied with ${runnerUpNames} at ${String(chosenDistinct)} distinct values each — ${nameOf(chosen)} wins only because it was declared first, not because it narrows the search any further)`
|
|
646
|
+
: ` (chosen over ${runnerUpNames} — ${String(chosenDistinct)} distinct values beats ${runnerUpDistinct.map(String).join('/')}, so it narrows the search more)`;
|
|
647
|
+
const named = chosen.matched.map((m) => m.column);
|
|
648
|
+
const prefixClause = composite
|
|
649
|
+
? named.length === chosen.columns.length
|
|
650
|
+
? `The predicate has an equality on every column of the composite index (${chosen.columns.join(', ')})`
|
|
651
|
+
: `The predicate has an equality on ${named.map((c) => `\`${c}\``).join(' and ')} — ${named.length === 1 ? 'the leftmost column' : 'the leftmost columns'} of the composite index (${chosen.columns.join(', ')}), which is all an index needs to be entered: \`${chosen.columns[named.length]}\` only orders the entries inside each run`
|
|
652
|
+
: '';
|
|
653
|
+
const unusable = leftmostPrefixNote(table, scan.filter);
|
|
654
|
+
const uniqueClause = chosen.unique && named.length === chosen.columns.length && !chosen.next
|
|
655
|
+
? ' The index is UNIQUE, so the lookup returns at most one row and stops at the first entry it finds.'
|
|
656
|
+
: '';
|
|
657
|
+
const rangeClause = chosen.next
|
|
658
|
+
? ` The next column, \`${chosen.next.column}\`, has a range (${formatRangeBounds(chosen.next.column, chosen.next.low, chosen.next.high)}), so the index narrows it within that run: one descent to where the range begins inside ${describeKeysText(keys)}, then along the leaf chain until the run ends.`
|
|
659
|
+
: '';
|
|
660
|
+
// Whether this costs a second descent (plan.md §25.4 B3e) depends on whether `index-only-scan` covers the query
|
|
661
|
+
// next — this rule runs before it and cannot know yet, so that fact is left to the execution-stage narration
|
|
662
|
+
// (`index/lookup.ts`'s `emitIndexLookup`), which always gets to see the real answer.
|
|
663
|
+
const clusteredClause = table.clusteredKey && !isClusteringIndex(table.clusteredKey, chosen.columns)
|
|
664
|
+
? ` This table is clustered on \`${indexNameOf(table.clusteredKey)}\`, and ${composite ? `(${chosen.columns.join(', ')})` : `\`${chosen.name}\``} is a secondary index — it holds the clustering key, not the row's address.`
|
|
665
|
+
: '';
|
|
666
|
+
return {
|
|
667
|
+
rule: 'index-selection',
|
|
668
|
+
label: `${composite
|
|
669
|
+
? `${prefixClause}${choiceClause}, so the ${named.length > 1 ? 'equalities become' : 'equality becomes'} an index lookup.${residual ? ' The rest of the predicate cannot be answered by the index and is rechecked on each row the index returns.' : ''}`
|
|
670
|
+
: residual
|
|
671
|
+
? `\`${chosen.name}\` is indexed${choiceClause}, so the equality becomes an index lookup. The rest of the predicate cannot be answered by the index and is rechecked on each row the index returns.`
|
|
672
|
+
: `\`${chosen.name}\` is indexed${choiceClause} and the predicate is an equality on it, so the scan becomes a B+Tree lookup — a handful of pages instead of the whole table.`}${uniqueClause}${rangeClause}${clusteredClause}${unusable ? ` ${unusable}` : ''}`,
|
|
673
|
+
plan: root,
|
|
674
|
+
changed: new Set([indexScan]),
|
|
675
|
+
prompt: composite && named.length < chosen.columns.length
|
|
676
|
+
? leftmostPrefixPrompt(chosen.name, chosen.columns, named)
|
|
677
|
+
: composite
|
|
678
|
+
? indexSelectionPrompt(`(${chosen.columns.join(', ')})`, keys[0], lookupText(chosen.name, keys))
|
|
679
|
+
: indexSelectionPrompt(chosen.name, chosen.matched[0].key),
|
|
680
|
+
whyNot: {
|
|
681
|
+
chosen: { label: `${nameOf(chosen)} index lookup`, estPages: indexPages },
|
|
682
|
+
runnerUp: { label: 'a sequential scan', estPages: scanPages },
|
|
683
|
+
},
|
|
684
|
+
};
|
|
685
|
+
};
|
|
686
|
+
/**
|
|
687
|
+
* Swap a `SeqScan` for an `IndexRangeScan` when a range comparison (`<`,
|
|
688
|
+
* `>`, `<=`, `>=` — including a `BETWEEN`, already desugared to both a lower
|
|
689
|
+
* and an upper bound by the parser) sits on an indexed column. Descends the
|
|
690
|
+
* B+Tree once to the leaf the lower bound would start at (the leftmost leaf,
|
|
691
|
+
* with none), then the executor walks the leaf sibling chain horizontally
|
|
692
|
+
* rather than rescanning the table (plan.md §23.1).
|
|
693
|
+
*
|
|
694
|
+
* Deliberately narrower than `index-selection`: it only ever sees a plan
|
|
695
|
+
* `index-selection` left untouched as a `SeqScan` — no equality conjunct
|
|
696
|
+
* matched any indexed column at all, *or* `index-selection` found one but
|
|
697
|
+
* declined it on cost (plan.md §25.4 C5) — because an equality is always at
|
|
698
|
+
* least as selective a starting point as a range, and choosing between an
|
|
699
|
+
* equality's `1/ndistinct` and a range's interval fraction by comparing the
|
|
700
|
+
* two numbers directly would still be exactly the kind of made-up comparison
|
|
701
|
+
* §12 rules out; the real cost model, below, is a different thing. At most
|
|
702
|
+
* one lower bound and one upper bound per column are consumed; a second
|
|
703
|
+
* bound in the same direction (`age > 10 AND age > 20`) is left as a
|
|
704
|
+
* residual recheck rather than merged — correctness never depends on
|
|
705
|
+
* catching it, only how tightly the index narrows the search does.
|
|
706
|
+
*
|
|
707
|
+
* **Chosen by cost too**: the same gate `index-selection` applies against the scan it would replace — see its own
|
|
708
|
+
* doc comment for why.
|
|
709
|
+
*/
|
|
710
|
+
const rangeIndexSelection = (input, table, rowsPerPage) => {
|
|
711
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
712
|
+
if (plan.op !== 'Project')
|
|
713
|
+
return null;
|
|
714
|
+
const { sort, rest: belowSort } = unwrapSort(plan.child);
|
|
715
|
+
const { aggregate, rest: scan } = unwrapAggregate(belowSort);
|
|
716
|
+
if (scan.op !== 'SeqScan' || !scan.filter)
|
|
717
|
+
return null;
|
|
718
|
+
const parts = conjuncts(scan.filter);
|
|
719
|
+
// A range enters an index through its *leading* column, plain or composite — the entries are ordered by it first.
|
|
720
|
+
// A range walk reads the leaf chain by row address, which a clustered table's secondary index does not carry
|
|
721
|
+
// (plan.md §25.4 B3e) — so a range only ever chooses the clustering key's own index there, or any index at all
|
|
722
|
+
// when the table isn't clustered.
|
|
723
|
+
const clusterSafe = (spec) => !table.clusteredKey || isClusteringIndex(table.clusteredKey, spec.columns);
|
|
724
|
+
const candidates = indexSpecsOf(table)
|
|
725
|
+
.filter(clusterSafe)
|
|
726
|
+
.flatMap((spec) => {
|
|
727
|
+
const column = spec.columns[0];
|
|
728
|
+
const name = indexNameOf(spec.columns);
|
|
729
|
+
const matches = parts
|
|
730
|
+
.map((p, i) => ({ i, match: rangeOn(p, column) }))
|
|
731
|
+
.filter((m) => m.match !== undefined);
|
|
732
|
+
if (matches.length === 0)
|
|
733
|
+
return [];
|
|
734
|
+
const lower = matches.find((m) => m.match.op === '>' || m.match.op === '>=');
|
|
735
|
+
const upper = matches.find((m) => m.match.op === '<' || m.match.op === '<=');
|
|
736
|
+
const consumed = new Set([lower?.i, upper?.i].filter((i) => i !== undefined));
|
|
737
|
+
const low = lower
|
|
738
|
+
? { value: lower.match.value, inclusive: lower.match.op === '>=' }
|
|
739
|
+
: undefined;
|
|
740
|
+
const high = upper
|
|
741
|
+
? { value: upper.match.value, inclusive: upper.match.op === '<=' }
|
|
742
|
+
: undefined;
|
|
743
|
+
const numericLow = low && typeof low.value === 'number' ? low.value : null;
|
|
744
|
+
const numericHigh = high && typeof high.value === 'number' ? high.value : null;
|
|
745
|
+
const selectivity = boundedRangeSelectivity(column, numericLow, numericHigh, table);
|
|
746
|
+
return [{ column, name, consumed, low, high, selectivity }];
|
|
747
|
+
});
|
|
748
|
+
if (candidates.length === 0)
|
|
749
|
+
return null;
|
|
750
|
+
const chosen = candidates.reduce((best, c) => c.selectivity.fraction < best.selectivity.fraction ? c : best);
|
|
751
|
+
const runnersUp = candidates.filter((c) => c !== chosen);
|
|
752
|
+
const residual = conjoin(parts.filter((_, i) => !chosen.consumed.has(i)));
|
|
753
|
+
const rangeScan = {
|
|
754
|
+
op: 'IndexRangeScan',
|
|
755
|
+
table: scan.table,
|
|
756
|
+
column: chosen.name,
|
|
757
|
+
...(chosen.low ? { low: chosen.low } : {}),
|
|
758
|
+
...(chosen.high ? { high: chosen.high } : {}),
|
|
759
|
+
...(residual ? { residual } : {}),
|
|
760
|
+
};
|
|
761
|
+
// Chosen by cost, not just by shape (plan.md §25.4 C5) — the same gate `index-selection` applies, see its own
|
|
762
|
+
// doc comment: a range that matches the predicate can still cost more than the scan it would replace once it
|
|
763
|
+
// spans enough of the table.
|
|
764
|
+
const scanPages = rootEstimate(scan, table, rowsPerPage).estPages;
|
|
765
|
+
const rangePages = rootEstimate(rangeScan, table, rowsPerPage).estPages;
|
|
766
|
+
if (rangePages >= scanPages) {
|
|
767
|
+
return {
|
|
768
|
+
rule: 'range-index-selection',
|
|
769
|
+
label: `\`${chosen.column}\` matches the predicate's range, but the estimated cost disagrees: the range walk would read about ${String(rangePages)} page${rangePages === 1 ? '' : 's'} against the scan's ${String(scanPages)} — no cheaper, so the scan stays and the whole predicate is rechecked there instead.`,
|
|
770
|
+
plan: input,
|
|
771
|
+
changed: new Set(),
|
|
772
|
+
whyNot: {
|
|
773
|
+
chosen: { label: 'a sequential scan', estPages: scanPages },
|
|
774
|
+
runnerUp: { label: `a range scan on \`${chosen.column}\``, estPages: rangePages },
|
|
775
|
+
},
|
|
776
|
+
};
|
|
777
|
+
}
|
|
778
|
+
const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: rewrapSort(sort, rewrapAggregate(aggregate, rangeScan)) });
|
|
779
|
+
const runnerUpNames = runnersUp.map((c) => `\`${c.column}\``).join(', ');
|
|
780
|
+
const choiceClause = runnersUp.length === 0
|
|
781
|
+
? ''
|
|
782
|
+
: runnersUp.every((c) => c.selectivity.fraction === chosen.selectivity.fraction)
|
|
783
|
+
? ` (tied with ${runnerUpNames} — \`${chosen.column}\` wins only because it was declared first, not because it narrows the search any further)`
|
|
784
|
+
: ` (chosen over ${runnerUpNames} — a narrower estimated range)`;
|
|
785
|
+
const boundsText = formatRangeBounds(chosen.column, chosen.low, chosen.high);
|
|
786
|
+
const indexedText = chosen.name === chosen.column
|
|
787
|
+
? `\`${chosen.column}\` is indexed`
|
|
788
|
+
: `the composite index (${chosen.name}) is ordered by \`${chosen.column}\` first, so it can be entered there`;
|
|
789
|
+
return {
|
|
790
|
+
rule: 'range-index-selection',
|
|
791
|
+
label: residual
|
|
792
|
+
? `${indexedText} and ${boundsText}${choiceClause}, so the scan becomes a single descent to the first matching leaf, then a walk along the leaf chain — no full table scan. The rest of the predicate is rechecked on each row the index returns.`
|
|
793
|
+
: `${indexedText} and the whole predicate is ${boundsText}${choiceClause}, so the scan becomes a single descent to the first matching leaf, then a walk along the leaf chain — no full table scan.`,
|
|
794
|
+
plan: root,
|
|
795
|
+
changed: new Set([rangeScan]),
|
|
796
|
+
whyNot: {
|
|
797
|
+
chosen: { label: `a range scan on \`${chosen.column}\``, estPages: rangePages },
|
|
798
|
+
runnerUp: { label: 'a sequential scan', estPages: scanPages },
|
|
799
|
+
},
|
|
800
|
+
};
|
|
801
|
+
};
|
|
802
|
+
/**
|
|
803
|
+
* When an `IndexScan`'s equality lookup already answers the query outright —
|
|
804
|
+
* every column the SELECT list needs is the indexed column itself, and
|
|
805
|
+
* there is no leftover conjunct to recheck against the row — the leaf's key
|
|
806
|
+
* *is* the answer. Runs after `index-selection`, since it rewrites the
|
|
807
|
+
* `IndexScan` that rule just produced; deliberately narrow (no `Sort` or
|
|
808
|
+
* aggregate node in between) rather than reaching through them the way the
|
|
809
|
+
* earlier rules do, since an index-only result is always exactly one row and
|
|
810
|
+
* there is nothing there yet worth the extra cases.
|
|
811
|
+
*/
|
|
812
|
+
const indexOnlyScan = (input) => {
|
|
813
|
+
const { limit, rest: plan } = unwrapLimit(input);
|
|
814
|
+
if (plan.op !== 'Project' || plan.columns.kind !== 'columns')
|
|
815
|
+
return null;
|
|
816
|
+
if (plan.child.op !== 'IndexScan' || plan.child.residual)
|
|
817
|
+
return null;
|
|
818
|
+
const scan = plan.child;
|
|
819
|
+
// An index holds the values of every column it is on — one for a plain index, several for a composite one, which
|
|
820
|
+
// therefore *covers* any query that reads only those columns (plan.md §25.4 B3).
|
|
821
|
+
const covered = columnsOfIndex(scan.column);
|
|
822
|
+
const onlyIndexedColumns = plan.columns.names.every((name) => covered.includes(name));
|
|
823
|
+
if (!onlyIndexedColumns)
|
|
824
|
+
return null;
|
|
825
|
+
const indexOnly = {
|
|
826
|
+
op: 'IndexOnlyScan',
|
|
827
|
+
table: scan.table,
|
|
828
|
+
column: scan.column,
|
|
829
|
+
key: scan.key,
|
|
830
|
+
...(scan.keys ? { keys: scan.keys } : {}),
|
|
831
|
+
};
|
|
832
|
+
const root = rewrapLimit(limit, { op: 'Project', columns: plan.columns, child: indexOnly });
|
|
833
|
+
const needs = plan.columns.names.map((n) => `\`${n}\``).join(', ');
|
|
834
|
+
return {
|
|
835
|
+
rule: 'index-only-scan',
|
|
836
|
+
label: covered.length > 1
|
|
837
|
+
? `Every column this query needs — ${needs} — is one the composite index (${scan.column}) holds, so the index covers the query: the entries answer it without ever fetching the heap page the pointers name.`
|
|
838
|
+
: `Every column this query needs — \`${scan.column}\` — is already the index key, so the lookup answers it without ever fetching the heap page the pointer names.`,
|
|
839
|
+
plan: root,
|
|
840
|
+
changed: new Set([indexOnly]),
|
|
841
|
+
};
|
|
842
|
+
};
|
|
843
|
+
const RULES = [
|
|
844
|
+
{ name: 'constant-folding', apply: constantFolding },
|
|
845
|
+
{ name: 'predicate-pushdown', apply: predicatePushdown },
|
|
846
|
+
{ name: 'join-predicate-pushdown', apply: joinPredicatePushdown },
|
|
847
|
+
{ name: 'join-algorithm-selection', apply: joinAlgorithmSelection },
|
|
848
|
+
{ name: 'index-selection', apply: indexSelection },
|
|
849
|
+
{ name: 'range-index-selection', apply: rangeIndexSelection },
|
|
850
|
+
{ name: 'index-only-scan', apply: indexOnlyScan },
|
|
851
|
+
];
|
|
852
|
+
/** Every rewrite rule's name, in the order the optimizer runs them. */
|
|
853
|
+
export const OPTIMIZER_RULES = RULES.map((r) => r.name);
|
|
854
|
+
/**
|
|
855
|
+
* `DELETE`/`UPDATE` reuse this same optimizer (`emitDeletePlanEvents`, `emitUpdatePlanEvents`) to find the rows to touch,
|
|
856
|
+
* wrapped in a throwaway `Project *`. Until §25.4 B3b their executors could only run a `SeqScan` or an `IndexScan`,
|
|
857
|
+
* so `range-index-selection` was left out and a write could never use a range index; `exec/writeScan.ts` now runs an
|
|
858
|
+
* `IndexRangeScan` too, so every rule applies. (`index-only-scan` never fires on a write — the `Project *` wants every
|
|
859
|
+
* column — which is the point: a write reads the rows it changes.) Kept as its own export so the write path's rule set
|
|
860
|
+
* stays a named decision, not an accident of reusing `OPTIMIZER_RULES`.
|
|
861
|
+
*/
|
|
862
|
+
export const WRITE_PATH_RULES = new Set(OPTIMIZER_RULES);
|
|
863
|
+
/**
|
|
864
|
+
* Applies every enabled rule that fires, in order. Returns each step for
|
|
865
|
+
* animation.
|
|
866
|
+
*
|
|
867
|
+
* `rowsPerPage` is threaded through to every rule (plan.md §25.4 C5): `index-selection`/`range-index-selection`
|
|
868
|
+
* need it to compare an index lookup's own estimated page cost against the scan it would replace, and
|
|
869
|
+
* `join-algorithm-selection` needs it (and `otherTables`) for the same reason, one table further out — see each
|
|
870
|
+
* rule's own doc comment. `otherTables` defaults to `{}` — every caller without a `JOIN` in the query omits it.
|
|
871
|
+
* `enabled` defaults to all rules — the pipeline (`emit.ts`) leaves it out except to keep
|
|
872
|
+
* `join-algorithm-selection` from overriding a `joinStrategy` explicitly requested. `/tools/planner` (plan.md
|
|
873
|
+
* §22.3) passes its own subset so any rule can be toggled off and its effect on the plan watched.
|
|
874
|
+
*/
|
|
875
|
+
export function optimize(plan, table, rowsPerPage, otherTables = {}, enabled = new Set(OPTIMIZER_RULES)) {
|
|
876
|
+
const steps = [];
|
|
877
|
+
let current = plan;
|
|
878
|
+
for (const { name, apply } of RULES) {
|
|
879
|
+
if (!enabled.has(name))
|
|
880
|
+
continue;
|
|
881
|
+
const rewrite = apply(current, table, rowsPerPage, otherTables);
|
|
882
|
+
if (!rewrite)
|
|
883
|
+
continue;
|
|
884
|
+
steps.push(rewrite);
|
|
885
|
+
current = rewrite.plan;
|
|
886
|
+
}
|
|
887
|
+
return steps;
|
|
888
|
+
}
|
|
889
|
+
/** The predicate a scan will evaluate, whatever form it ended up in. */
|
|
890
|
+
export function scanPredicate(plan) {
|
|
891
|
+
if (plan.op === 'SeqScan')
|
|
892
|
+
return plan.filter;
|
|
893
|
+
if (plan.op === 'IndexScan' || plan.op === 'IndexRangeScan')
|
|
894
|
+
return plan.residual;
|
|
895
|
+
return plan.op === 'Filter' ||
|
|
896
|
+
plan.op === 'Having' ||
|
|
897
|
+
plan.op === 'HashDistinct' ||
|
|
898
|
+
plan.op === 'SortDistinct' ||
|
|
899
|
+
plan.op === 'HashAggregate' ||
|
|
900
|
+
plan.op === 'SortAggregate' ||
|
|
901
|
+
plan.op === 'Sort' ||
|
|
902
|
+
plan.op === 'Project' ||
|
|
903
|
+
plan.op === 'Limit'
|
|
904
|
+
? scanPredicate(plan.child)
|
|
905
|
+
: undefined;
|
|
906
|
+
}
|