querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,582 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A first cost model: cardinality and page-count *estimates*, annotated onto
|
|
3
|
+
* plan nodes (plan.md §7.3, §22.2).
|
|
4
|
+
*
|
|
5
|
+
* Every number here is a textbook heuristic, and each carries a `basis` string
|
|
6
|
+
* naming the rule of thumb it came from — an equality is `1 / ndistinct`, an
|
|
7
|
+
* open range is linear over the column's min–max, an `AND` multiplies its parts
|
|
8
|
+
* and assumes they are independent. plan.md §12 is explicit that an invented
|
|
9
|
+
* figure with a decimal point is worse than an honest guess, so the estimate is
|
|
10
|
+
* always shown next to the count the executor actually produced: the optimiser
|
|
11
|
+
* is guessing, and the UI says so.
|
|
12
|
+
*
|
|
13
|
+
* Pure and synchronous, on the same rules as the rest of `src/engine`.
|
|
14
|
+
*/
|
|
15
|
+
import { columnsOfIndex } from "../index/spec.js";
|
|
16
|
+
import { lookupKeysOf } from "./plan.js";
|
|
17
|
+
/** Default selectivity for an open range with no usable min/max (Selinger et al., 1979). */
|
|
18
|
+
export const OPEN_RANGE_SELECTIVITY = 1 / 3;
|
|
19
|
+
/** Default equality selectivity when a column has no distinct-value stats. */
|
|
20
|
+
export const NO_STATS_EQ_SELECTIVITY = 0.1;
|
|
21
|
+
/**
|
|
22
|
+
* `LIKE` has no honest closed form without a histogram of prefixes — a guess of
|
|
23
|
+
* one row in ten, the classic optimizer default. Named as a default on the node.
|
|
24
|
+
*/
|
|
25
|
+
export const LIKE_SELECTIVITY = 0.1;
|
|
26
|
+
/** B+Tree fan-out — `DEFAULT_MAX_KEYS` in src/engine/index/btree.ts. */
|
|
27
|
+
const FANOUT = 4;
|
|
28
|
+
/**
|
|
29
|
+
* How selective an equality lookup on the first `width` columns of the index `name` is: `distinct` is the number of
|
|
30
|
+
* distinct keys it assumes (each column's distinct count multiplied — independence, the same assumption an `AND`
|
|
31
|
+
* makes), so `rows / distinct` rows match on average. For a plain index that is `ndistinct` of its one column.
|
|
32
|
+
*/
|
|
33
|
+
function lookupSelectivity(table, name, width) {
|
|
34
|
+
const columns = columnsOfIndex(name).slice(0, width);
|
|
35
|
+
const counts = columns.map((column) => ndistinct(table, column));
|
|
36
|
+
const distinct = counts.reduce((product, n) => product * n, 1);
|
|
37
|
+
return {
|
|
38
|
+
distinct,
|
|
39
|
+
basis: columns.length === 1 ? `1/${String(distinct)} on ${name}` : `${counts.map((n) => `1/${String(n)}`).join(' × ')} on ${columns.join(', ')}`,
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
function liveNdistinct(rows, column) {
|
|
43
|
+
const seen = new Set();
|
|
44
|
+
for (const row of rows)
|
|
45
|
+
seen.add(row[column] ?? null);
|
|
46
|
+
return Math.max(seen.size, 1);
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Exported for `optimize.ts`'s `index-selection` rule — the same selectivity fact, reused to pick the more
|
|
50
|
+
* selective of two candidate indexes rather than re-deriving it. Reads `table.stats` when `ANALYZE` has set it
|
|
51
|
+
* (plan.md §25.4 C1) — a real count, but possibly a stale one, exactly the way a real optimizer's is — and
|
|
52
|
+
* otherwise scans the live rows, today's (unconditionally accurate, unconditionally *unrealistic*) behaviour.
|
|
53
|
+
*/
|
|
54
|
+
export function ndistinct(table, column) {
|
|
55
|
+
const stat = table.stats?.columns[column];
|
|
56
|
+
return stat ? Math.max(stat.nDistinct, 1) : liveNdistinct(table.rows, column);
|
|
57
|
+
}
|
|
58
|
+
function columnName(expr) {
|
|
59
|
+
if (expr.kind !== 'compare')
|
|
60
|
+
return null;
|
|
61
|
+
if (expr.left.kind === 'column')
|
|
62
|
+
return expr.left.name;
|
|
63
|
+
if (expr.right.kind === 'column')
|
|
64
|
+
return expr.right.name;
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
function literalValue(expr) {
|
|
68
|
+
if (expr.kind !== 'compare')
|
|
69
|
+
return null;
|
|
70
|
+
if (expr.left.kind === 'literal')
|
|
71
|
+
return expr.left.value;
|
|
72
|
+
if (expr.right.kind === 'literal')
|
|
73
|
+
return expr.right.value;
|
|
74
|
+
return null;
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* `column = literal`'s selectivity. Without statistics this is the textbook `1/ndistinct` — every distinct value
|
|
78
|
+
* assumed equally likely. With them (plan.md §25.4 C1), a `literal` that is itself one of the column's own
|
|
79
|
+
* most-common values gets its *real* frequency instead of the average; any other value gets the average
|
|
80
|
+
* frequency of everything the most-common list does *not* already cover, which is closer to `1/ndistinct` the
|
|
81
|
+
* less skewed the column turns out to be and further from it the more skewed.
|
|
82
|
+
*/
|
|
83
|
+
function equalitySelectivity(column, table, literal = null) {
|
|
84
|
+
if (!column || !table.columns.includes(column)) {
|
|
85
|
+
return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
|
|
86
|
+
}
|
|
87
|
+
const stat = table.stats?.columns[column];
|
|
88
|
+
const rowCount = table.stats?.rowCount ?? 0;
|
|
89
|
+
if (stat && rowCount > 0) {
|
|
90
|
+
const mcv = literal !== null ? stat.mostCommonValues.find((m) => m.value === literal) : undefined;
|
|
91
|
+
if (mcv) {
|
|
92
|
+
return { fraction: mcv.frequency / rowCount, basis: `${String(mcv.frequency)}/${String(rowCount)} on ${column} (its most common value, ANALYZE)` };
|
|
93
|
+
}
|
|
94
|
+
const coveredFraction = stat.mostCommonValues.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
|
|
95
|
+
const remainingDistinct = Math.max(1, stat.nDistinct - stat.mostCommonValues.length);
|
|
96
|
+
return {
|
|
97
|
+
fraction: Math.max(0, 1 - coveredFraction) / remainingDistinct,
|
|
98
|
+
basis: `1/${String(remainingDistinct)} of the rest on ${column} (ANALYZE)`,
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
const distinct = ndistinct(table, column);
|
|
102
|
+
return { fraction: 1 / distinct, basis: `1/${String(distinct)} on ${column}` };
|
|
103
|
+
}
|
|
104
|
+
function rangeSelectivity(column, op, literal, table) {
|
|
105
|
+
const fallback = { fraction: OPEN_RANGE_SELECTIVITY, basis: '⅓ default' };
|
|
106
|
+
if (!column || typeof literal !== 'number')
|
|
107
|
+
return fallback;
|
|
108
|
+
// A single bound is a half-open BETWEEN — `boundedRangeSelectivity` already has the ANALYZE-aware histogram
|
|
109
|
+
// logic (plan.md §25.4 C1), so a bound with statistics behind it is handed straight to it rather than
|
|
110
|
+
// duplicating that logic here. One without statistics falls through to this function's own min/max fallback,
|
|
111
|
+
// unchanged, so every query a table without `ANALYZE` ever ran keeps exactly the estimate it always got.
|
|
112
|
+
if (table.stats?.columns[column]) {
|
|
113
|
+
return op === '>' || op === '>=' ? boundedRangeSelectivity(column, literal, null, table) : boundedRangeSelectivity(column, null, literal, table);
|
|
114
|
+
}
|
|
115
|
+
const values = table.rows
|
|
116
|
+
.map((row) => row[column])
|
|
117
|
+
.filter((value) => typeof value === 'number');
|
|
118
|
+
if (values.length === 0)
|
|
119
|
+
return fallback;
|
|
120
|
+
const min = Math.min(...values);
|
|
121
|
+
const max = Math.max(...values);
|
|
122
|
+
if (max === min)
|
|
123
|
+
return fallback;
|
|
124
|
+
const above = (max - literal) / (max - min);
|
|
125
|
+
const fraction = Math.min(1, Math.max(0, op === '>' ? above : 1 - above));
|
|
126
|
+
return { fraction, basis: `~${String(Math.round(fraction * 100))}% by range` };
|
|
127
|
+
}
|
|
128
|
+
/**
|
|
129
|
+
* Fraction of an equi-depth histogram's own rows (plan.md §25.4 C1) that fall within `[low, high]` — each bucket
|
|
130
|
+
* assumed to hold an equal share of them, and each bucket's own share assumed spread evenly between its two
|
|
131
|
+
* boundaries. Numeric boundaries only; a text histogram (still useful for `mostCommonValues`, never for this)
|
|
132
|
+
* falls back to the classic ⅓ default, same as no histogram at all.
|
|
133
|
+
*/
|
|
134
|
+
function histogramFraction(bounds, low, high) {
|
|
135
|
+
if (bounds.length < 2 || !bounds.every((b) => typeof b === 'number'))
|
|
136
|
+
return OPEN_RANGE_SELECTIVITY;
|
|
137
|
+
const numeric = bounds;
|
|
138
|
+
const buckets = numeric.length - 1;
|
|
139
|
+
let covered = 0;
|
|
140
|
+
for (let i = 0; i < buckets; i++) {
|
|
141
|
+
const bucketLow = numeric[i];
|
|
142
|
+
const bucketHigh = numeric[i + 1];
|
|
143
|
+
const width = bucketHigh - bucketLow;
|
|
144
|
+
const effLow = low === null ? bucketLow : Math.max(bucketLow, low);
|
|
145
|
+
const effHigh = high === null ? bucketHigh : Math.min(bucketHigh, high);
|
|
146
|
+
const overlap = width <= 0 ? (effHigh >= effLow ? 1 : 0) : Math.max(0, effHigh - effLow) / width;
|
|
147
|
+
covered += overlap / buckets;
|
|
148
|
+
}
|
|
149
|
+
return Math.min(1, Math.max(0, covered));
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Fraction of `table`'s rows whose `column` falls within `[low, high]` —
|
|
153
|
+
* either bound may be `null` for a half-open range. A dedicated formula
|
|
154
|
+
* rather than reusing `selectivityOf`'s `AND` path (which multiplies two
|
|
155
|
+
* independent fractions): a range's own lower and upper bound are *not*
|
|
156
|
+
* independent of each other — the same column, not two different ones — so
|
|
157
|
+
* multiplying their separate `rangeSelectivity` fractions would
|
|
158
|
+
* double-discount a `BETWEEN` (plan.md §12: no plausible-but-false numbers).
|
|
159
|
+
* When both bounds and the column's numeric min/max are known, this instead
|
|
160
|
+
* computes the true linear-density overlap directly, which is provably the
|
|
161
|
+
* same fraction `rangeSelectivity` already gives for a single bound (set the
|
|
162
|
+
* absent bound to the column's own min or max and the two formulas agree).
|
|
163
|
+
*/
|
|
164
|
+
export function boundedRangeSelectivity(column, low, high, table) {
|
|
165
|
+
const bounded = low !== null && high !== null;
|
|
166
|
+
const fallback = {
|
|
167
|
+
fraction: bounded ? OPEN_RANGE_SELECTIVITY * OPEN_RANGE_SELECTIVITY : OPEN_RANGE_SELECTIVITY,
|
|
168
|
+
basis: bounded ? '⅓ × ⅓ default (no numeric stats)' : '⅓ default (no numeric stats)',
|
|
169
|
+
};
|
|
170
|
+
if (!column)
|
|
171
|
+
return fallback;
|
|
172
|
+
const stat = table.stats?.columns[column];
|
|
173
|
+
if (stat && table.stats.rowCount > 0) {
|
|
174
|
+
const rowCount = table.stats.rowCount;
|
|
175
|
+
const mcvInRange = stat.mostCommonValues.filter((m) => typeof m.value === 'number' && (low === null || m.value >= low) && (high === null || m.value <= high));
|
|
176
|
+
const mcvInRangeFraction = mcvInRange.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
|
|
177
|
+
const mcvTotalFraction = stat.mostCommonValues.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
|
|
178
|
+
const histFraction = histogramFraction(stat.histogramBounds, low, high);
|
|
179
|
+
const fraction = clampFraction(mcvInRangeFraction + (1 - mcvTotalFraction) * histFraction);
|
|
180
|
+
const boundsLabel = `${low === null ? 'no lower bound' : `≥${String(low)}`}, ${high === null ? 'no upper bound' : `≤${String(high)}`}`;
|
|
181
|
+
return { fraction, basis: `~${String(Math.round(fraction * 100))}% by histogram (${boundsLabel}, ANALYZE)` };
|
|
182
|
+
}
|
|
183
|
+
const values = table.rows
|
|
184
|
+
.map((row) => row[column])
|
|
185
|
+
.filter((value) => typeof value === 'number');
|
|
186
|
+
if (values.length === 0)
|
|
187
|
+
return fallback;
|
|
188
|
+
const min = Math.min(...values);
|
|
189
|
+
const max = Math.max(...values);
|
|
190
|
+
if (max === min)
|
|
191
|
+
return fallback;
|
|
192
|
+
const effectiveLow = low === null ? min : Math.max(min, low);
|
|
193
|
+
const effectiveHigh = high === null ? max : Math.min(max, high);
|
|
194
|
+
const fraction = Math.min(1, Math.max(0, (effectiveHigh - effectiveLow) / (max - min)));
|
|
195
|
+
const boundsLabel = `${low === null ? 'no lower bound' : `≥${String(low)}`}, ${high === null ? 'no upper bound' : `≤${String(high)}`}`;
|
|
196
|
+
return { fraction, basis: `~${String(Math.round(fraction * 100))}% by range (${boundsLabel})` };
|
|
197
|
+
}
|
|
198
|
+
/** The fraction of `column`'s values that are NULL — from `ANALYZE`'s stats when it has run (plan.md §25.4 C1;
|
|
199
|
+
* possibly stale), otherwise counted from the live data. */
|
|
200
|
+
function nullSelectivity(column, table) {
|
|
201
|
+
if (!column || !table.columns.includes(column)) {
|
|
202
|
+
return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
|
|
203
|
+
}
|
|
204
|
+
const stat = table.stats?.columns[column];
|
|
205
|
+
if (stat) {
|
|
206
|
+
const rowCount = table.stats.rowCount;
|
|
207
|
+
const nulls = Math.round(stat.nullFraction * rowCount);
|
|
208
|
+
return { fraction: stat.nullFraction, basis: `${String(nulls)}/${String(rowCount)} NULL in ${column} (ANALYZE)` };
|
|
209
|
+
}
|
|
210
|
+
if (table.rows.length === 0)
|
|
211
|
+
return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
|
|
212
|
+
const nulls = table.rows.filter((row) => (row[column] ?? null) === null).length;
|
|
213
|
+
return { fraction: nulls / table.rows.length, basis: `${String(nulls)}/${String(table.rows.length)} NULL in ${column}` };
|
|
214
|
+
}
|
|
215
|
+
/** The column an operand names, when it is a bare column reference. */
|
|
216
|
+
const columnOf = (expr) => (expr.kind === 'column' ? expr.name : null);
|
|
217
|
+
const clampFraction = (f) => Math.min(1, Math.max(0, f));
|
|
218
|
+
/**
|
|
219
|
+
* Selectivity of a predicate, with the heuristic that produced it.
|
|
220
|
+
*
|
|
221
|
+
* `OR` uses inclusion–exclusion (`a + b − a·b`) and, like `AND`'s product,
|
|
222
|
+
* assumes the two parts are independent; `NOT` is the complement. Both are
|
|
223
|
+
* exactly the assumptions plan.md §25.4 C2 later sets out to break on purpose.
|
|
224
|
+
*/
|
|
225
|
+
export function selectivityOf(expr, table) {
|
|
226
|
+
switch (expr.kind) {
|
|
227
|
+
case 'and': {
|
|
228
|
+
const left = selectivityOf(expr.left, table);
|
|
229
|
+
const right = selectivityOf(expr.right, table);
|
|
230
|
+
return {
|
|
231
|
+
fraction: left.fraction * right.fraction,
|
|
232
|
+
basis: `${left.basis} × ${right.basis}`,
|
|
233
|
+
};
|
|
234
|
+
}
|
|
235
|
+
case 'or': {
|
|
236
|
+
const left = selectivityOf(expr.left, table);
|
|
237
|
+
const right = selectivityOf(expr.right, table);
|
|
238
|
+
return {
|
|
239
|
+
fraction: clampFraction(left.fraction + right.fraction - left.fraction * right.fraction),
|
|
240
|
+
basis: `${left.basis} ∪ ${right.basis}`,
|
|
241
|
+
};
|
|
242
|
+
}
|
|
243
|
+
case 'not': {
|
|
244
|
+
const inner = selectivityOf(expr.operand, table);
|
|
245
|
+
return { fraction: clampFraction(1 - inner.fraction), basis: `1 − (${inner.basis})` };
|
|
246
|
+
}
|
|
247
|
+
case 'isNull': {
|
|
248
|
+
const nulls = nullSelectivity(columnOf(expr.operand), table);
|
|
249
|
+
return expr.negated
|
|
250
|
+
? { fraction: clampFraction(1 - nulls.fraction), basis: `1 − ${nulls.basis}` }
|
|
251
|
+
: nulls;
|
|
252
|
+
}
|
|
253
|
+
case 'in': {
|
|
254
|
+
const one = equalitySelectivity(columnOf(expr.operand), table);
|
|
255
|
+
const k = expr.items.length;
|
|
256
|
+
const fraction = clampFraction(k * one.fraction);
|
|
257
|
+
return {
|
|
258
|
+
fraction: expr.negated ? clampFraction(1 - fraction) : fraction,
|
|
259
|
+
basis: `${expr.negated ? '1 − ' : ''}${String(k)} × ${one.basis}`,
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
case 'like':
|
|
263
|
+
return {
|
|
264
|
+
fraction: expr.negated ? 1 - LIKE_SELECTIVITY : LIKE_SELECTIVITY,
|
|
265
|
+
basis: expr.negated ? '0.9 default (NOT LIKE)' : '0.1 default (LIKE)',
|
|
266
|
+
};
|
|
267
|
+
case 'compare': {
|
|
268
|
+
if (expr.op === '=')
|
|
269
|
+
return equalitySelectivity(columnName(expr), table, literalValue(expr));
|
|
270
|
+
if (expr.op === '<>') {
|
|
271
|
+
const eq = equalitySelectivity(columnName(expr), table, literalValue(expr));
|
|
272
|
+
return { fraction: clampFraction(1 - eq.fraction), basis: `1 − ${eq.basis}` };
|
|
273
|
+
}
|
|
274
|
+
return rangeSelectivity(columnName(expr), expr.op, literalValue(expr), table);
|
|
275
|
+
}
|
|
276
|
+
default:
|
|
277
|
+
return { fraction: 1, basis: 'no predicate' };
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
const clampRows = (value) => Math.max(0, Math.round(value));
|
|
281
|
+
/** Estimated B+Tree levels for a bulk-loaded index over `rowCount` rows. */
|
|
282
|
+
export function estimateIndexLevels(rowCount) {
|
|
283
|
+
let nodes = Math.max(1, Math.ceil(rowCount / FANOUT));
|
|
284
|
+
let levels = 1;
|
|
285
|
+
while (nodes > 1) {
|
|
286
|
+
nodes = Math.ceil(nodes / (FANOUT + 1));
|
|
287
|
+
levels += 1;
|
|
288
|
+
}
|
|
289
|
+
return levels;
|
|
290
|
+
}
|
|
291
|
+
/**
|
|
292
|
+
* `otherTables` is consulted only by the `Join` case, for whichever side of
|
|
293
|
+
* the join is not `table` itself — every other case estimates against the
|
|
294
|
+
* one table a v0 query without a JOIN ever has. A `Join`'s own two children
|
|
295
|
+
* are always a bare `SeqScan` in v1 (plan.md §22.2: no predicate pushdown
|
|
296
|
+
* into either side yet), so each side's own row/column data is exactly one
|
|
297
|
+
* table's, found by name in `table`/`otherTables` rather than threaded down
|
|
298
|
+
* through every other case that never needs a second table at all.
|
|
299
|
+
*/
|
|
300
|
+
function estimateNode(plan, table, rowsPerPage, into, otherTables = {}) {
|
|
301
|
+
const rowCount = table.rows.length;
|
|
302
|
+
const heapPages = Math.max(1, Math.ceil(rowCount / rowsPerPage));
|
|
303
|
+
const tableNamed = (name) => (name === table.name ? table : otherTables[name]);
|
|
304
|
+
let estimate;
|
|
305
|
+
switch (plan.op) {
|
|
306
|
+
case 'Join': {
|
|
307
|
+
const left = estimateNode(plan.left, tableNamed(plan.leftTable), rowsPerPage, into, otherTables);
|
|
308
|
+
const right = estimateNode(plan.right, tableNamed(plan.rightTable), rowsPerPage, into, otherTables);
|
|
309
|
+
// Foreign-key assumption: each left row matches about 1/ndistinct(right
|
|
310
|
+
// key) of the right table — the same independence heuristic an
|
|
311
|
+
// equality WHERE already makes, just applied to a join key instead of a
|
|
312
|
+
// literal. Row count never depends on the algorithm — only the page
|
|
313
|
+
// cost does.
|
|
314
|
+
const rightDistinct = ndistinct(tableNamed(plan.rightTable), plan.rightColumn);
|
|
315
|
+
// An `IndexProbe`'s own estimate is *per outer row*, so the full inner size comes from the table itself.
|
|
316
|
+
const rightRows = plan.algorithm === 'index-nested-loop' ? tableNamed(plan.rightTable).rows.length : right.estRows;
|
|
317
|
+
const innerEstimate = clampRows((left.estRows * rightRows) / rightDistinct);
|
|
318
|
+
// A LEFT JOIN (plan.md §25.4 C3) never drops a left row — the equi-join estimate is a floor, not the answer,
|
|
319
|
+
// whenever it would otherwise guess fewer rows than the outer side alone already has.
|
|
320
|
+
const estRows = plan.joinType === 'left' ? Math.max(left.estRows, innerEstimate) : innerEstimate;
|
|
321
|
+
const rowsBasis = plan.joinType === 'left' && estRows === left.estRows && left.estRows > innerEstimate
|
|
322
|
+
? `${String(left.estRows)} (every ${plan.leftTable} row survives a LEFT JOIN; the equi-join estimate alone was only ${String(innerEstimate)})`
|
|
323
|
+
: `${String(left.estRows)} × ${String(rightRows)} / ndistinct(${plan.rightTable}.${plan.rightColumn})`;
|
|
324
|
+
switch (plan.algorithm) {
|
|
325
|
+
case 'index-nested-loop':
|
|
326
|
+
estimate = {
|
|
327
|
+
// The outer is read once; then each outer row pays one probe — the descent, any extra leaves the run of
|
|
328
|
+
// equal keys spans, and a heap page per match (`right` is that per-probe estimate). This is the number
|
|
329
|
+
// that beats a rescan whenever the outer is small and the inner is big.
|
|
330
|
+
estRows,
|
|
331
|
+
estPages: left.estPages + left.estRows * right.estPages,
|
|
332
|
+
basis: `${rowsBasis}; index nested-loop reads ${plan.leftTable} once, then probes ${plan.rightTable}'s index on ${plan.rightColumn} once per row`,
|
|
333
|
+
};
|
|
334
|
+
break;
|
|
335
|
+
case 'hash':
|
|
336
|
+
estimate = {
|
|
337
|
+
// Build the right side into a hash table once, then probe with
|
|
338
|
+
// the left streamed — each side is read exactly once, the
|
|
339
|
+
// honest saving over nested-loop's repeated rescans.
|
|
340
|
+
estRows,
|
|
341
|
+
estPages: left.estPages + right.estPages,
|
|
342
|
+
basis: `${rowsBasis}; hash join reads each side once — build a table on ${plan.rightTable}, probe with ${plan.leftTable}`,
|
|
343
|
+
};
|
|
344
|
+
break;
|
|
345
|
+
case 'sort-merge': {
|
|
346
|
+
// Each side is read once, same as hash — but sorting it first is
|
|
347
|
+
// real spill I/O (mirroring `Sort`'s own `2 * spillPages` "at
|
|
348
|
+
// least run generation" estimate), which hash join's in-memory
|
|
349
|
+
// table never pays. Honest, not flattering: sort-merge is not
|
|
350
|
+
// free just because it avoids nested-loop's rescans.
|
|
351
|
+
const leftSpill = Math.max(1, Math.ceil(left.estRows / rowsPerPage));
|
|
352
|
+
const rightSpill = Math.max(1, Math.ceil(right.estRows / rowsPerPage));
|
|
353
|
+
estimate = {
|
|
354
|
+
estRows,
|
|
355
|
+
estPages: left.estPages + right.estPages + 2 * leftSpill + 2 * rightSpill,
|
|
356
|
+
basis: `${rowsBasis}; sort-merge reads each side once, then sorts both — at least ${String(2 * leftSpill)} spill pages on ${plan.leftTable}, ${String(2 * rightSpill)} on ${plan.rightTable}`,
|
|
357
|
+
};
|
|
358
|
+
break;
|
|
359
|
+
}
|
|
360
|
+
case 'nested-loop':
|
|
361
|
+
estimate = {
|
|
362
|
+
// Nested-loop's own cost is the honest, unflattering number:
|
|
363
|
+
// the outer read once, plus a full rescan of the inner for
|
|
364
|
+
// every outer row — no index taken advantage of yet, since
|
|
365
|
+
// `left`/`right` are always a bare SeqScan in v1.
|
|
366
|
+
estRows,
|
|
367
|
+
estPages: left.estPages + left.estRows * right.estPages,
|
|
368
|
+
basis: `${rowsBasis}; nested-loop rescans ${plan.rightTable} once per ${plan.leftTable} row`,
|
|
369
|
+
};
|
|
370
|
+
break;
|
|
371
|
+
}
|
|
372
|
+
break;
|
|
373
|
+
}
|
|
374
|
+
case 'SeqScan': {
|
|
375
|
+
if (plan.filter) {
|
|
376
|
+
const { fraction, basis } = selectivityOf(plan.filter, table);
|
|
377
|
+
estimate = { estRows: clampRows(rowCount * fraction), estPages: heapPages, basis };
|
|
378
|
+
}
|
|
379
|
+
else {
|
|
380
|
+
estimate = { estRows: rowCount, estPages: heapPages, basis: 'every row' };
|
|
381
|
+
}
|
|
382
|
+
break;
|
|
383
|
+
}
|
|
384
|
+
case 'IndexProbe': {
|
|
385
|
+
// One probe of `table`'s index for one outer row: the per-probe cost, which the Join multiplies by its outer rows.
|
|
386
|
+
const inner = tableNamed(plan.table);
|
|
387
|
+
const distinct = ndistinct(inner, plan.column);
|
|
388
|
+
const matches = Math.max(clampRows(inner.rows.length / distinct), 1);
|
|
389
|
+
const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
|
|
390
|
+
estimate = {
|
|
391
|
+
estRows: matches,
|
|
392
|
+
estPages: estimateIndexLevels(inner.rows.length) + extraLeaves + matches,
|
|
393
|
+
basis: `per outer row: 1/${String(distinct)} on ${plan.column}; ${String(estimateIndexLevels(inner.rows.length))} to descend, ${String(matches)} heap page${matches === 1 ? '' : 's'}`,
|
|
394
|
+
};
|
|
395
|
+
break;
|
|
396
|
+
}
|
|
397
|
+
case 'IndexScan': {
|
|
398
|
+
const { distinct, basis: lookupBasis } = lookupSelectivity(table, plan.column, lookupKeysOf(plan).length);
|
|
399
|
+
let fraction = 1 / distinct;
|
|
400
|
+
let basis = lookupBasis;
|
|
401
|
+
if (plan.residual) {
|
|
402
|
+
const residual = selectivityOf(plan.residual, table);
|
|
403
|
+
fraction *= residual.fraction;
|
|
404
|
+
basis += ` + recheck ${residual.basis}`;
|
|
405
|
+
}
|
|
406
|
+
const levels = estimateIndexLevels(rowCount);
|
|
407
|
+
// A repeated value is one entry per row: the lookup may read a few more leaves along the run, and each match
|
|
408
|
+
// costs its own heap page. For a unique value that is zero extra leaves and one page — the +1 it always was.
|
|
409
|
+
const matches = Math.max(clampRows(rowCount / distinct), 1);
|
|
410
|
+
const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
|
|
411
|
+
estimate = {
|
|
412
|
+
estRows: Math.max(clampRows(rowCount * fraction), plan.residual ? 0 : 1),
|
|
413
|
+
estPages: levels + extraLeaves + matches,
|
|
414
|
+
basis,
|
|
415
|
+
};
|
|
416
|
+
break;
|
|
417
|
+
}
|
|
418
|
+
case 'IndexRangeScan': {
|
|
419
|
+
const low = plan.low && typeof plan.low.value === 'number' ? plan.low.value : null;
|
|
420
|
+
const high = plan.high && typeof plan.high.value === 'number' ? plan.high.value : null;
|
|
421
|
+
const prefixLength = plan.prefix?.length ?? 0;
|
|
422
|
+
const rangeColumn = columnsOfIndex(plan.column)[prefixLength] ?? plan.column;
|
|
423
|
+
const { fraction: rangeFraction, basis: rangeBasis } = boundedRangeSelectivity(rangeColumn, low, high, table);
|
|
424
|
+
// Within a composite index's equality prefix the range narrows only that run.
|
|
425
|
+
const within = prefixLength > 0 ? lookupSelectivity(table, plan.column, prefixLength) : null;
|
|
426
|
+
let fraction = within ? rangeFraction / within.distinct : rangeFraction;
|
|
427
|
+
let basis = within ? `${within.basis} × ${rangeBasis}` : rangeBasis;
|
|
428
|
+
if (plan.residual) {
|
|
429
|
+
const residual = selectivityOf(plan.residual, table);
|
|
430
|
+
fraction *= residual.fraction;
|
|
431
|
+
basis += ` + recheck ${residual.basis}`;
|
|
432
|
+
}
|
|
433
|
+
const estRows = clampRows(rowCount * fraction);
|
|
434
|
+
const levels = estimateIndexLevels(rowCount);
|
|
435
|
+
// Descend to the first leaf (`levels` node reads), walk however many
|
|
436
|
+
// more leaves the estimated rows span via sibling pointers, then one
|
|
437
|
+
// heap page per matching row — the same one-touch-per-row granularity
|
|
438
|
+
// IndexScan already prices, just repeated `estRows` times instead of
|
|
439
|
+
// once.
|
|
440
|
+
const leafSpan = Math.max(1, Math.ceil(estRows / FANOUT));
|
|
441
|
+
const extraLeaves = Math.max(0, leafSpan - 1);
|
|
442
|
+
estimate = {
|
|
443
|
+
estRows,
|
|
444
|
+
estPages: levels + extraLeaves + estRows,
|
|
445
|
+
basis: `${basis}; ${String(levels)} to descend, ${String(extraLeaves)} more leaf${extraLeaves === 1 ? '' : 's'} via sibling pointers, ${String(estRows)} heap page${estRows === 1 ? '' : 's'}`,
|
|
446
|
+
};
|
|
447
|
+
break;
|
|
448
|
+
}
|
|
449
|
+
case 'IndexOnlyScan': {
|
|
450
|
+
// The tree levels alone — no heap page, the honest difference from IndexScan — plus any further leaves a
|
|
451
|
+
// repeated value's run spans.
|
|
452
|
+
const { distinct, basis: lookupBasis } = lookupSelectivity(table, plan.column, lookupKeysOf(plan).length);
|
|
453
|
+
const matches = Math.max(clampRows(rowCount / distinct), 1);
|
|
454
|
+
const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
|
|
455
|
+
estimate = {
|
|
456
|
+
estRows: matches,
|
|
457
|
+
estPages: estimateIndexLevels(rowCount) + extraLeaves,
|
|
458
|
+
basis: `${lookupBasis} — index-only, no heap page`,
|
|
459
|
+
};
|
|
460
|
+
break;
|
|
461
|
+
}
|
|
462
|
+
case 'Filter': {
|
|
463
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
464
|
+
const { fraction, basis } = selectivityOf(plan.predicate, table);
|
|
465
|
+
estimate = { estRows: clampRows(child.estRows * fraction), estPages: child.estPages, basis };
|
|
466
|
+
break;
|
|
467
|
+
}
|
|
468
|
+
case 'HashDistinct':
|
|
469
|
+
case 'SortDistinct': {
|
|
470
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
471
|
+
// How many distinct rows: bounded by the input, and — when the projection is plain columns — by the product
|
|
472
|
+
// of their distinct-value counts (the same independence assumption an AND makes). An expression, `*` or a
|
|
473
|
+
// column the table does not know has no statistic, so the input size stands.
|
|
474
|
+
const project = plan.child.op === 'Project' ? plan.child.columns : null;
|
|
475
|
+
let bound = child.estRows;
|
|
476
|
+
let basis = 'no distinct-value statistic for this list';
|
|
477
|
+
if (project?.kind === 'columns' && project.names.every((n) => table.columns.includes(n))) {
|
|
478
|
+
const distinct = project.names.reduce((product, n) => product * ndistinct(table, n), 1);
|
|
479
|
+
bound = Math.min(child.estRows, distinct);
|
|
480
|
+
basis = `min(${String(child.estRows)}, ${project.names.map((n) => `${String(ndistinct(table, n))} distinct ${n}`).join(' × ')})`;
|
|
481
|
+
}
|
|
482
|
+
// A sort-based DISTINCT spills exactly like `Sort`: every row written once and read back once.
|
|
483
|
+
const spill = plan.op === 'SortDistinct' ? 2 * Math.max(1, Math.ceil(child.estRows / rowsPerPage)) : 0;
|
|
484
|
+
estimate = {
|
|
485
|
+
estRows: clampRows(bound),
|
|
486
|
+
estPages: child.estPages + spill,
|
|
487
|
+
basis: plan.op === 'SortDistinct' ? `${basis}; sorted first (${String(spill)} spill pages)` : basis,
|
|
488
|
+
};
|
|
489
|
+
break;
|
|
490
|
+
}
|
|
491
|
+
case 'Having': {
|
|
492
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
493
|
+
// Aggregate values have no column statistics to consult, so every comparison against one falls to the
|
|
494
|
+
// range/equality *defaults* (`selectivityOf` finds no column) — honest, and labelled as such.
|
|
495
|
+
const { fraction, basis } = selectivityOf(plan.predicate, table);
|
|
496
|
+
estimate = { estRows: clampRows(child.estRows * fraction), estPages: child.estPages, basis: `HAVING: ${basis}` };
|
|
497
|
+
break;
|
|
498
|
+
}
|
|
499
|
+
case 'HashAggregate':
|
|
500
|
+
case 'SortAggregate': {
|
|
501
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
502
|
+
// A hash table adds no I/O of its own — everything happens in memory —
|
|
503
|
+
// but a sort aggregate spills exactly like `Sort` does, on top of
|
|
504
|
+
// whatever the child already costs to read; that extra spill, priced
|
|
505
|
+
// the same way `Sort`'s own case below prices it, is the entire point
|
|
506
|
+
// of comparison `/compare`'s third mode shows (plan.md §22.2).
|
|
507
|
+
const spillPages = plan.op === 'SortAggregate' ? 2 * Math.max(1, Math.ceil(child.estRows / rowsPerPage)) : 0;
|
|
508
|
+
if (plan.groupBy.length === 0) {
|
|
509
|
+
// No GROUP BY: the whole input is one group — even zero input rows
|
|
510
|
+
// still produce one, since COUNT(*) of nothing is 0, not absent.
|
|
511
|
+
estimate = {
|
|
512
|
+
estRows: 1,
|
|
513
|
+
estPages: child.estPages + spillPages,
|
|
514
|
+
basis: plan.op === 'SortAggregate'
|
|
515
|
+
? `no GROUP BY — one group, but sorted first anyway (${String(spillPages)} spill pages)`
|
|
516
|
+
: 'no GROUP BY — the whole input is one group',
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
else {
|
|
520
|
+
// Textbook independence assumption, same as an AND's selectivity:
|
|
521
|
+
// the product of each column's distinct-value count, capped at the
|
|
522
|
+
// rows actually going in — there cannot be more groups than rows.
|
|
523
|
+
const groups = plan.groupBy.reduce((product, col) => product * ndistinct(table, col), 1);
|
|
524
|
+
const groupBasis = `${plan.groupBy.map((c) => `ndistinct(${c})`).join(' × ')}, capped at the input`;
|
|
525
|
+
estimate = {
|
|
526
|
+
estRows: Math.max(1, Math.min(child.estRows, clampRows(groups))),
|
|
527
|
+
estPages: child.estPages + spillPages,
|
|
528
|
+
basis: plan.op === 'SortAggregate'
|
|
529
|
+
? `${groupBasis}; sorts ${String(child.estRows)} rows first (${String(spillPages)} spill pages)`
|
|
530
|
+
: groupBasis,
|
|
531
|
+
};
|
|
532
|
+
}
|
|
533
|
+
break;
|
|
534
|
+
}
|
|
535
|
+
case 'Sort': {
|
|
536
|
+
// Row order changes, not row count — but Sort's own I/O is the honest
|
|
537
|
+
// number to surface here: a spill writes and re-reads the data, so its
|
|
538
|
+
// page cost is real disk activity a Filter/Project never adds.
|
|
539
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
540
|
+
const spillPages = Math.max(1, Math.ceil(child.estRows / rowsPerPage));
|
|
541
|
+
estimate = {
|
|
542
|
+
estRows: child.estRows,
|
|
543
|
+
estPages: 2 * spillPages,
|
|
544
|
+
basis: `sorts ${String(child.estRows)} rows; at least ${String(2 * spillPages)} spill pages (run generation alone)`,
|
|
545
|
+
};
|
|
546
|
+
break;
|
|
547
|
+
}
|
|
548
|
+
case 'Project': {
|
|
549
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
550
|
+
estimate = { estRows: child.estRows, estPages: child.estPages, basis: child.basis };
|
|
551
|
+
break;
|
|
552
|
+
}
|
|
553
|
+
case 'Limit': {
|
|
554
|
+
const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
|
|
555
|
+
const offset = plan.offset ?? 0;
|
|
556
|
+
estimate = {
|
|
557
|
+
estRows: Math.min(Math.max(0, child.estRows - offset), plan.count),
|
|
558
|
+
estPages: child.estPages,
|
|
559
|
+
basis: offset > 0
|
|
560
|
+
? `min(${String(child.estRows)} − ${String(offset)} skipped, LIMIT ${String(plan.count)})`
|
|
561
|
+
: `min(${String(child.estRows)}, LIMIT ${String(plan.count)})`,
|
|
562
|
+
};
|
|
563
|
+
break;
|
|
564
|
+
}
|
|
565
|
+
}
|
|
566
|
+
into.set(plan, estimate);
|
|
567
|
+
return estimate;
|
|
568
|
+
}
|
|
569
|
+
/**
|
|
570
|
+
* Per-node estimates for a plan tree. `otherTables` is the join partner's
|
|
571
|
+
* `Table`, keyed by name — needed only when `plan` contains a `Join`; every
|
|
572
|
+
* other caller (no JOIN in the query) omits it.
|
|
573
|
+
*/
|
|
574
|
+
export function estimatePlan(plan, table, rowsPerPage, otherTables) {
|
|
575
|
+
const cost = new Map();
|
|
576
|
+
estimateNode(plan, table, rowsPerPage, cost, otherTables);
|
|
577
|
+
return cost;
|
|
578
|
+
}
|
|
579
|
+
/** The estimate for the plan's root node — the query's estimated output. */
|
|
580
|
+
export function rootEstimate(plan, table, rowsPerPage, otherTables) {
|
|
581
|
+
return estimateNode(plan, table, rowsPerPage, new Map(), otherTables);
|
|
582
|
+
}
|