querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,582 @@
1
+ /**
2
+ * A first cost model: cardinality and page-count *estimates*, annotated onto
3
+ * plan nodes (plan.md §7.3, §22.2).
4
+ *
5
+ * Every number here is a textbook heuristic, and each carries a `basis` string
6
+ * naming the rule of thumb it came from — an equality is `1 / ndistinct`, an
7
+ * open range is linear over the column's min–max, an `AND` multiplies its parts
8
+ * and assumes they are independent. plan.md §12 is explicit that an invented
9
+ * figure with a decimal point is worse than an honest guess, so the estimate is
10
+ * always shown next to the count the executor actually produced: the optimiser
11
+ * is guessing, and the UI says so.
12
+ *
13
+ * Pure and synchronous, on the same rules as the rest of `src/engine`.
14
+ */
15
+ import { columnsOfIndex } from "../index/spec.js";
16
+ import { lookupKeysOf } from "./plan.js";
17
+ /** Default selectivity for an open range with no usable min/max (Selinger et al., 1979). */
18
+ export const OPEN_RANGE_SELECTIVITY = 1 / 3;
19
+ /** Default equality selectivity when a column has no distinct-value stats. */
20
+ export const NO_STATS_EQ_SELECTIVITY = 0.1;
21
+ /**
22
+ * `LIKE` has no honest closed form without a histogram of prefixes — a guess of
23
+ * one row in ten, the classic optimizer default. Named as a default on the node.
24
+ */
25
+ export const LIKE_SELECTIVITY = 0.1;
26
+ /** B+Tree fan-out — `DEFAULT_MAX_KEYS` in src/engine/index/btree.ts. */
27
+ const FANOUT = 4;
28
+ /**
29
+ * How selective an equality lookup on the first `width` columns of the index `name` is: `distinct` is the number of
30
+ * distinct keys it assumes (each column's distinct count multiplied — independence, the same assumption an `AND`
31
+ * makes), so `rows / distinct` rows match on average. For a plain index that is `ndistinct` of its one column.
32
+ */
33
+ function lookupSelectivity(table, name, width) {
34
+ const columns = columnsOfIndex(name).slice(0, width);
35
+ const counts = columns.map((column) => ndistinct(table, column));
36
+ const distinct = counts.reduce((product, n) => product * n, 1);
37
+ return {
38
+ distinct,
39
+ basis: columns.length === 1 ? `1/${String(distinct)} on ${name}` : `${counts.map((n) => `1/${String(n)}`).join(' × ')} on ${columns.join(', ')}`,
40
+ };
41
+ }
42
+ function liveNdistinct(rows, column) {
43
+ const seen = new Set();
44
+ for (const row of rows)
45
+ seen.add(row[column] ?? null);
46
+ return Math.max(seen.size, 1);
47
+ }
48
+ /**
49
+ * Exported for `optimize.ts`'s `index-selection` rule — the same selectivity fact, reused to pick the more
50
+ * selective of two candidate indexes rather than re-deriving it. Reads `table.stats` when `ANALYZE` has set it
51
+ * (plan.md §25.4 C1) — a real count, but possibly a stale one, exactly the way a real optimizer's is — and
52
+ * otherwise scans the live rows, today's (unconditionally accurate, unconditionally *unrealistic*) behaviour.
53
+ */
54
+ export function ndistinct(table, column) {
55
+ const stat = table.stats?.columns[column];
56
+ return stat ? Math.max(stat.nDistinct, 1) : liveNdistinct(table.rows, column);
57
+ }
58
+ function columnName(expr) {
59
+ if (expr.kind !== 'compare')
60
+ return null;
61
+ if (expr.left.kind === 'column')
62
+ return expr.left.name;
63
+ if (expr.right.kind === 'column')
64
+ return expr.right.name;
65
+ return null;
66
+ }
67
+ function literalValue(expr) {
68
+ if (expr.kind !== 'compare')
69
+ return null;
70
+ if (expr.left.kind === 'literal')
71
+ return expr.left.value;
72
+ if (expr.right.kind === 'literal')
73
+ return expr.right.value;
74
+ return null;
75
+ }
76
+ /**
77
+ * `column = literal`'s selectivity. Without statistics this is the textbook `1/ndistinct` — every distinct value
78
+ * assumed equally likely. With them (plan.md §25.4 C1), a `literal` that is itself one of the column's own
79
+ * most-common values gets its *real* frequency instead of the average; any other value gets the average
80
+ * frequency of everything the most-common list does *not* already cover, which is closer to `1/ndistinct` the
81
+ * less skewed the column turns out to be and further from it the more skewed.
82
+ */
83
+ function equalitySelectivity(column, table, literal = null) {
84
+ if (!column || !table.columns.includes(column)) {
85
+ return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
86
+ }
87
+ const stat = table.stats?.columns[column];
88
+ const rowCount = table.stats?.rowCount ?? 0;
89
+ if (stat && rowCount > 0) {
90
+ const mcv = literal !== null ? stat.mostCommonValues.find((m) => m.value === literal) : undefined;
91
+ if (mcv) {
92
+ return { fraction: mcv.frequency / rowCount, basis: `${String(mcv.frequency)}/${String(rowCount)} on ${column} (its most common value, ANALYZE)` };
93
+ }
94
+ const coveredFraction = stat.mostCommonValues.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
95
+ const remainingDistinct = Math.max(1, stat.nDistinct - stat.mostCommonValues.length);
96
+ return {
97
+ fraction: Math.max(0, 1 - coveredFraction) / remainingDistinct,
98
+ basis: `1/${String(remainingDistinct)} of the rest on ${column} (ANALYZE)`,
99
+ };
100
+ }
101
+ const distinct = ndistinct(table, column);
102
+ return { fraction: 1 / distinct, basis: `1/${String(distinct)} on ${column}` };
103
+ }
104
+ function rangeSelectivity(column, op, literal, table) {
105
+ const fallback = { fraction: OPEN_RANGE_SELECTIVITY, basis: '⅓ default' };
106
+ if (!column || typeof literal !== 'number')
107
+ return fallback;
108
+ // A single bound is a half-open BETWEEN — `boundedRangeSelectivity` already has the ANALYZE-aware histogram
109
+ // logic (plan.md §25.4 C1), so a bound with statistics behind it is handed straight to it rather than
110
+ // duplicating that logic here. One without statistics falls through to this function's own min/max fallback,
111
+ // unchanged, so every query a table without `ANALYZE` ever ran keeps exactly the estimate it always got.
112
+ if (table.stats?.columns[column]) {
113
+ return op === '>' || op === '>=' ? boundedRangeSelectivity(column, literal, null, table) : boundedRangeSelectivity(column, null, literal, table);
114
+ }
115
+ const values = table.rows
116
+ .map((row) => row[column])
117
+ .filter((value) => typeof value === 'number');
118
+ if (values.length === 0)
119
+ return fallback;
120
+ const min = Math.min(...values);
121
+ const max = Math.max(...values);
122
+ if (max === min)
123
+ return fallback;
124
+ const above = (max - literal) / (max - min);
125
+ const fraction = Math.min(1, Math.max(0, op === '>' ? above : 1 - above));
126
+ return { fraction, basis: `~${String(Math.round(fraction * 100))}% by range` };
127
+ }
128
+ /**
129
+ * Fraction of an equi-depth histogram's own rows (plan.md §25.4 C1) that fall within `[low, high]` — each bucket
130
+ * assumed to hold an equal share of them, and each bucket's own share assumed spread evenly between its two
131
+ * boundaries. Numeric boundaries only; a text histogram (still useful for `mostCommonValues`, never for this)
132
+ * falls back to the classic ⅓ default, same as no histogram at all.
133
+ */
134
+ function histogramFraction(bounds, low, high) {
135
+ if (bounds.length < 2 || !bounds.every((b) => typeof b === 'number'))
136
+ return OPEN_RANGE_SELECTIVITY;
137
+ const numeric = bounds;
138
+ const buckets = numeric.length - 1;
139
+ let covered = 0;
140
+ for (let i = 0; i < buckets; i++) {
141
+ const bucketLow = numeric[i];
142
+ const bucketHigh = numeric[i + 1];
143
+ const width = bucketHigh - bucketLow;
144
+ const effLow = low === null ? bucketLow : Math.max(bucketLow, low);
145
+ const effHigh = high === null ? bucketHigh : Math.min(bucketHigh, high);
146
+ const overlap = width <= 0 ? (effHigh >= effLow ? 1 : 0) : Math.max(0, effHigh - effLow) / width;
147
+ covered += overlap / buckets;
148
+ }
149
+ return Math.min(1, Math.max(0, covered));
150
+ }
151
+ /**
152
+ * Fraction of `table`'s rows whose `column` falls within `[low, high]` —
153
+ * either bound may be `null` for a half-open range. A dedicated formula
154
+ * rather than reusing `selectivityOf`'s `AND` path (which multiplies two
155
+ * independent fractions): a range's own lower and upper bound are *not*
156
+ * independent of each other — the same column, not two different ones — so
157
+ * multiplying their separate `rangeSelectivity` fractions would
158
+ * double-discount a `BETWEEN` (plan.md §12: no plausible-but-false numbers).
159
+ * When both bounds and the column's numeric min/max are known, this instead
160
+ * computes the true linear-density overlap directly, which is provably the
161
+ * same fraction `rangeSelectivity` already gives for a single bound (set the
162
+ * absent bound to the column's own min or max and the two formulas agree).
163
+ */
164
+ export function boundedRangeSelectivity(column, low, high, table) {
165
+ const bounded = low !== null && high !== null;
166
+ const fallback = {
167
+ fraction: bounded ? OPEN_RANGE_SELECTIVITY * OPEN_RANGE_SELECTIVITY : OPEN_RANGE_SELECTIVITY,
168
+ basis: bounded ? '⅓ × ⅓ default (no numeric stats)' : '⅓ default (no numeric stats)',
169
+ };
170
+ if (!column)
171
+ return fallback;
172
+ const stat = table.stats?.columns[column];
173
+ if (stat && table.stats.rowCount > 0) {
174
+ const rowCount = table.stats.rowCount;
175
+ const mcvInRange = stat.mostCommonValues.filter((m) => typeof m.value === 'number' && (low === null || m.value >= low) && (high === null || m.value <= high));
176
+ const mcvInRangeFraction = mcvInRange.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
177
+ const mcvTotalFraction = stat.mostCommonValues.reduce((sum, m) => sum + m.frequency, 0) / rowCount;
178
+ const histFraction = histogramFraction(stat.histogramBounds, low, high);
179
+ const fraction = clampFraction(mcvInRangeFraction + (1 - mcvTotalFraction) * histFraction);
180
+ const boundsLabel = `${low === null ? 'no lower bound' : `≥${String(low)}`}, ${high === null ? 'no upper bound' : `≤${String(high)}`}`;
181
+ return { fraction, basis: `~${String(Math.round(fraction * 100))}% by histogram (${boundsLabel}, ANALYZE)` };
182
+ }
183
+ const values = table.rows
184
+ .map((row) => row[column])
185
+ .filter((value) => typeof value === 'number');
186
+ if (values.length === 0)
187
+ return fallback;
188
+ const min = Math.min(...values);
189
+ const max = Math.max(...values);
190
+ if (max === min)
191
+ return fallback;
192
+ const effectiveLow = low === null ? min : Math.max(min, low);
193
+ const effectiveHigh = high === null ? max : Math.min(max, high);
194
+ const fraction = Math.min(1, Math.max(0, (effectiveHigh - effectiveLow) / (max - min)));
195
+ const boundsLabel = `${low === null ? 'no lower bound' : `≥${String(low)}`}, ${high === null ? 'no upper bound' : `≤${String(high)}`}`;
196
+ return { fraction, basis: `~${String(Math.round(fraction * 100))}% by range (${boundsLabel})` };
197
+ }
198
+ /** The fraction of `column`'s values that are NULL — from `ANALYZE`'s stats when it has run (plan.md §25.4 C1;
199
+ * possibly stale), otherwise counted from the live data. */
200
+ function nullSelectivity(column, table) {
201
+ if (!column || !table.columns.includes(column)) {
202
+ return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
203
+ }
204
+ const stat = table.stats?.columns[column];
205
+ if (stat) {
206
+ const rowCount = table.stats.rowCount;
207
+ const nulls = Math.round(stat.nullFraction * rowCount);
208
+ return { fraction: stat.nullFraction, basis: `${String(nulls)}/${String(rowCount)} NULL in ${column} (ANALYZE)` };
209
+ }
210
+ if (table.rows.length === 0)
211
+ return { fraction: NO_STATS_EQ_SELECTIVITY, basis: '0.1 default' };
212
+ const nulls = table.rows.filter((row) => (row[column] ?? null) === null).length;
213
+ return { fraction: nulls / table.rows.length, basis: `${String(nulls)}/${String(table.rows.length)} NULL in ${column}` };
214
+ }
215
+ /** The column an operand names, when it is a bare column reference. */
216
+ const columnOf = (expr) => (expr.kind === 'column' ? expr.name : null);
217
+ const clampFraction = (f) => Math.min(1, Math.max(0, f));
218
+ /**
219
+ * Selectivity of a predicate, with the heuristic that produced it.
220
+ *
221
+ * `OR` uses inclusion–exclusion (`a + b − a·b`) and, like `AND`'s product,
222
+ * assumes the two parts are independent; `NOT` is the complement. Both are
223
+ * exactly the assumptions plan.md §25.4 C2 later sets out to break on purpose.
224
+ */
225
+ export function selectivityOf(expr, table) {
226
+ switch (expr.kind) {
227
+ case 'and': {
228
+ const left = selectivityOf(expr.left, table);
229
+ const right = selectivityOf(expr.right, table);
230
+ return {
231
+ fraction: left.fraction * right.fraction,
232
+ basis: `${left.basis} × ${right.basis}`,
233
+ };
234
+ }
235
+ case 'or': {
236
+ const left = selectivityOf(expr.left, table);
237
+ const right = selectivityOf(expr.right, table);
238
+ return {
239
+ fraction: clampFraction(left.fraction + right.fraction - left.fraction * right.fraction),
240
+ basis: `${left.basis} ∪ ${right.basis}`,
241
+ };
242
+ }
243
+ case 'not': {
244
+ const inner = selectivityOf(expr.operand, table);
245
+ return { fraction: clampFraction(1 - inner.fraction), basis: `1 − (${inner.basis})` };
246
+ }
247
+ case 'isNull': {
248
+ const nulls = nullSelectivity(columnOf(expr.operand), table);
249
+ return expr.negated
250
+ ? { fraction: clampFraction(1 - nulls.fraction), basis: `1 − ${nulls.basis}` }
251
+ : nulls;
252
+ }
253
+ case 'in': {
254
+ const one = equalitySelectivity(columnOf(expr.operand), table);
255
+ const k = expr.items.length;
256
+ const fraction = clampFraction(k * one.fraction);
257
+ return {
258
+ fraction: expr.negated ? clampFraction(1 - fraction) : fraction,
259
+ basis: `${expr.negated ? '1 − ' : ''}${String(k)} × ${one.basis}`,
260
+ };
261
+ }
262
+ case 'like':
263
+ return {
264
+ fraction: expr.negated ? 1 - LIKE_SELECTIVITY : LIKE_SELECTIVITY,
265
+ basis: expr.negated ? '0.9 default (NOT LIKE)' : '0.1 default (LIKE)',
266
+ };
267
+ case 'compare': {
268
+ if (expr.op === '=')
269
+ return equalitySelectivity(columnName(expr), table, literalValue(expr));
270
+ if (expr.op === '<>') {
271
+ const eq = equalitySelectivity(columnName(expr), table, literalValue(expr));
272
+ return { fraction: clampFraction(1 - eq.fraction), basis: `1 − ${eq.basis}` };
273
+ }
274
+ return rangeSelectivity(columnName(expr), expr.op, literalValue(expr), table);
275
+ }
276
+ default:
277
+ return { fraction: 1, basis: 'no predicate' };
278
+ }
279
+ }
280
+ const clampRows = (value) => Math.max(0, Math.round(value));
281
+ /** Estimated B+Tree levels for a bulk-loaded index over `rowCount` rows. */
282
+ export function estimateIndexLevels(rowCount) {
283
+ let nodes = Math.max(1, Math.ceil(rowCount / FANOUT));
284
+ let levels = 1;
285
+ while (nodes > 1) {
286
+ nodes = Math.ceil(nodes / (FANOUT + 1));
287
+ levels += 1;
288
+ }
289
+ return levels;
290
+ }
291
+ /**
292
+ * `otherTables` is consulted only by the `Join` case, for whichever side of
293
+ * the join is not `table` itself — every other case estimates against the
294
+ * one table a v0 query without a JOIN ever has. A `Join`'s own two children
295
+ * are always a bare `SeqScan` in v1 (plan.md §22.2: no predicate pushdown
296
+ * into either side yet), so each side's own row/column data is exactly one
297
+ * table's, found by name in `table`/`otherTables` rather than threaded down
298
+ * through every other case that never needs a second table at all.
299
+ */
300
+ function estimateNode(plan, table, rowsPerPage, into, otherTables = {}) {
301
+ const rowCount = table.rows.length;
302
+ const heapPages = Math.max(1, Math.ceil(rowCount / rowsPerPage));
303
+ const tableNamed = (name) => (name === table.name ? table : otherTables[name]);
304
+ let estimate;
305
+ switch (plan.op) {
306
+ case 'Join': {
307
+ const left = estimateNode(plan.left, tableNamed(plan.leftTable), rowsPerPage, into, otherTables);
308
+ const right = estimateNode(plan.right, tableNamed(plan.rightTable), rowsPerPage, into, otherTables);
309
+ // Foreign-key assumption: each left row matches about 1/ndistinct(right
310
+ // key) of the right table — the same independence heuristic an
311
+ // equality WHERE already makes, just applied to a join key instead of a
312
+ // literal. Row count never depends on the algorithm — only the page
313
+ // cost does.
314
+ const rightDistinct = ndistinct(tableNamed(plan.rightTable), plan.rightColumn);
315
+ // An `IndexProbe`'s own estimate is *per outer row*, so the full inner size comes from the table itself.
316
+ const rightRows = plan.algorithm === 'index-nested-loop' ? tableNamed(plan.rightTable).rows.length : right.estRows;
317
+ const innerEstimate = clampRows((left.estRows * rightRows) / rightDistinct);
318
+ // A LEFT JOIN (plan.md §25.4 C3) never drops a left row — the equi-join estimate is a floor, not the answer,
319
+ // whenever it would otherwise guess fewer rows than the outer side alone already has.
320
+ const estRows = plan.joinType === 'left' ? Math.max(left.estRows, innerEstimate) : innerEstimate;
321
+ const rowsBasis = plan.joinType === 'left' && estRows === left.estRows && left.estRows > innerEstimate
322
+ ? `${String(left.estRows)} (every ${plan.leftTable} row survives a LEFT JOIN; the equi-join estimate alone was only ${String(innerEstimate)})`
323
+ : `${String(left.estRows)} × ${String(rightRows)} / ndistinct(${plan.rightTable}.${plan.rightColumn})`;
324
+ switch (plan.algorithm) {
325
+ case 'index-nested-loop':
326
+ estimate = {
327
+ // The outer is read once; then each outer row pays one probe — the descent, any extra leaves the run of
328
+ // equal keys spans, and a heap page per match (`right` is that per-probe estimate). This is the number
329
+ // that beats a rescan whenever the outer is small and the inner is big.
330
+ estRows,
331
+ estPages: left.estPages + left.estRows * right.estPages,
332
+ basis: `${rowsBasis}; index nested-loop reads ${plan.leftTable} once, then probes ${plan.rightTable}'s index on ${plan.rightColumn} once per row`,
333
+ };
334
+ break;
335
+ case 'hash':
336
+ estimate = {
337
+ // Build the right side into a hash table once, then probe with
338
+ // the left streamed — each side is read exactly once, the
339
+ // honest saving over nested-loop's repeated rescans.
340
+ estRows,
341
+ estPages: left.estPages + right.estPages,
342
+ basis: `${rowsBasis}; hash join reads each side once — build a table on ${plan.rightTable}, probe with ${plan.leftTable}`,
343
+ };
344
+ break;
345
+ case 'sort-merge': {
346
+ // Each side is read once, same as hash — but sorting it first is
347
+ // real spill I/O (mirroring `Sort`'s own `2 * spillPages` "at
348
+ // least run generation" estimate), which hash join's in-memory
349
+ // table never pays. Honest, not flattering: sort-merge is not
350
+ // free just because it avoids nested-loop's rescans.
351
+ const leftSpill = Math.max(1, Math.ceil(left.estRows / rowsPerPage));
352
+ const rightSpill = Math.max(1, Math.ceil(right.estRows / rowsPerPage));
353
+ estimate = {
354
+ estRows,
355
+ estPages: left.estPages + right.estPages + 2 * leftSpill + 2 * rightSpill,
356
+ basis: `${rowsBasis}; sort-merge reads each side once, then sorts both — at least ${String(2 * leftSpill)} spill pages on ${plan.leftTable}, ${String(2 * rightSpill)} on ${plan.rightTable}`,
357
+ };
358
+ break;
359
+ }
360
+ case 'nested-loop':
361
+ estimate = {
362
+ // Nested-loop's own cost is the honest, unflattering number:
363
+ // the outer read once, plus a full rescan of the inner for
364
+ // every outer row — no index taken advantage of yet, since
365
+ // `left`/`right` are always a bare SeqScan in v1.
366
+ estRows,
367
+ estPages: left.estPages + left.estRows * right.estPages,
368
+ basis: `${rowsBasis}; nested-loop rescans ${plan.rightTable} once per ${plan.leftTable} row`,
369
+ };
370
+ break;
371
+ }
372
+ break;
373
+ }
374
+ case 'SeqScan': {
375
+ if (plan.filter) {
376
+ const { fraction, basis } = selectivityOf(plan.filter, table);
377
+ estimate = { estRows: clampRows(rowCount * fraction), estPages: heapPages, basis };
378
+ }
379
+ else {
380
+ estimate = { estRows: rowCount, estPages: heapPages, basis: 'every row' };
381
+ }
382
+ break;
383
+ }
384
+ case 'IndexProbe': {
385
+ // One probe of `table`'s index for one outer row: the per-probe cost, which the Join multiplies by its outer rows.
386
+ const inner = tableNamed(plan.table);
387
+ const distinct = ndistinct(inner, plan.column);
388
+ const matches = Math.max(clampRows(inner.rows.length / distinct), 1);
389
+ const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
390
+ estimate = {
391
+ estRows: matches,
392
+ estPages: estimateIndexLevels(inner.rows.length) + extraLeaves + matches,
393
+ basis: `per outer row: 1/${String(distinct)} on ${plan.column}; ${String(estimateIndexLevels(inner.rows.length))} to descend, ${String(matches)} heap page${matches === 1 ? '' : 's'}`,
394
+ };
395
+ break;
396
+ }
397
+ case 'IndexScan': {
398
+ const { distinct, basis: lookupBasis } = lookupSelectivity(table, plan.column, lookupKeysOf(plan).length);
399
+ let fraction = 1 / distinct;
400
+ let basis = lookupBasis;
401
+ if (plan.residual) {
402
+ const residual = selectivityOf(plan.residual, table);
403
+ fraction *= residual.fraction;
404
+ basis += ` + recheck ${residual.basis}`;
405
+ }
406
+ const levels = estimateIndexLevels(rowCount);
407
+ // A repeated value is one entry per row: the lookup may read a few more leaves along the run, and each match
408
+ // costs its own heap page. For a unique value that is zero extra leaves and one page — the +1 it always was.
409
+ const matches = Math.max(clampRows(rowCount / distinct), 1);
410
+ const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
411
+ estimate = {
412
+ estRows: Math.max(clampRows(rowCount * fraction), plan.residual ? 0 : 1),
413
+ estPages: levels + extraLeaves + matches,
414
+ basis,
415
+ };
416
+ break;
417
+ }
418
+ case 'IndexRangeScan': {
419
+ const low = plan.low && typeof plan.low.value === 'number' ? plan.low.value : null;
420
+ const high = plan.high && typeof plan.high.value === 'number' ? plan.high.value : null;
421
+ const prefixLength = plan.prefix?.length ?? 0;
422
+ const rangeColumn = columnsOfIndex(plan.column)[prefixLength] ?? plan.column;
423
+ const { fraction: rangeFraction, basis: rangeBasis } = boundedRangeSelectivity(rangeColumn, low, high, table);
424
+ // Within a composite index's equality prefix the range narrows only that run.
425
+ const within = prefixLength > 0 ? lookupSelectivity(table, plan.column, prefixLength) : null;
426
+ let fraction = within ? rangeFraction / within.distinct : rangeFraction;
427
+ let basis = within ? `${within.basis} × ${rangeBasis}` : rangeBasis;
428
+ if (plan.residual) {
429
+ const residual = selectivityOf(plan.residual, table);
430
+ fraction *= residual.fraction;
431
+ basis += ` + recheck ${residual.basis}`;
432
+ }
433
+ const estRows = clampRows(rowCount * fraction);
434
+ const levels = estimateIndexLevels(rowCount);
435
+ // Descend to the first leaf (`levels` node reads), walk however many
436
+ // more leaves the estimated rows span via sibling pointers, then one
437
+ // heap page per matching row — the same one-touch-per-row granularity
438
+ // IndexScan already prices, just repeated `estRows` times instead of
439
+ // once.
440
+ const leafSpan = Math.max(1, Math.ceil(estRows / FANOUT));
441
+ const extraLeaves = Math.max(0, leafSpan - 1);
442
+ estimate = {
443
+ estRows,
444
+ estPages: levels + extraLeaves + estRows,
445
+ basis: `${basis}; ${String(levels)} to descend, ${String(extraLeaves)} more leaf${extraLeaves === 1 ? '' : 's'} via sibling pointers, ${String(estRows)} heap page${estRows === 1 ? '' : 's'}`,
446
+ };
447
+ break;
448
+ }
449
+ case 'IndexOnlyScan': {
450
+ // The tree levels alone — no heap page, the honest difference from IndexScan — plus any further leaves a
451
+ // repeated value's run spans.
452
+ const { distinct, basis: lookupBasis } = lookupSelectivity(table, plan.column, lookupKeysOf(plan).length);
453
+ const matches = Math.max(clampRows(rowCount / distinct), 1);
454
+ const extraLeaves = Math.max(0, Math.ceil(matches / FANOUT) - 1);
455
+ estimate = {
456
+ estRows: matches,
457
+ estPages: estimateIndexLevels(rowCount) + extraLeaves,
458
+ basis: `${lookupBasis} — index-only, no heap page`,
459
+ };
460
+ break;
461
+ }
462
+ case 'Filter': {
463
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
464
+ const { fraction, basis } = selectivityOf(plan.predicate, table);
465
+ estimate = { estRows: clampRows(child.estRows * fraction), estPages: child.estPages, basis };
466
+ break;
467
+ }
468
+ case 'HashDistinct':
469
+ case 'SortDistinct': {
470
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
471
+ // How many distinct rows: bounded by the input, and — when the projection is plain columns — by the product
472
+ // of their distinct-value counts (the same independence assumption an AND makes). An expression, `*` or a
473
+ // column the table does not know has no statistic, so the input size stands.
474
+ const project = plan.child.op === 'Project' ? plan.child.columns : null;
475
+ let bound = child.estRows;
476
+ let basis = 'no distinct-value statistic for this list';
477
+ if (project?.kind === 'columns' && project.names.every((n) => table.columns.includes(n))) {
478
+ const distinct = project.names.reduce((product, n) => product * ndistinct(table, n), 1);
479
+ bound = Math.min(child.estRows, distinct);
480
+ basis = `min(${String(child.estRows)}, ${project.names.map((n) => `${String(ndistinct(table, n))} distinct ${n}`).join(' × ')})`;
481
+ }
482
+ // A sort-based DISTINCT spills exactly like `Sort`: every row written once and read back once.
483
+ const spill = plan.op === 'SortDistinct' ? 2 * Math.max(1, Math.ceil(child.estRows / rowsPerPage)) : 0;
484
+ estimate = {
485
+ estRows: clampRows(bound),
486
+ estPages: child.estPages + spill,
487
+ basis: plan.op === 'SortDistinct' ? `${basis}; sorted first (${String(spill)} spill pages)` : basis,
488
+ };
489
+ break;
490
+ }
491
+ case 'Having': {
492
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
493
+ // Aggregate values have no column statistics to consult, so every comparison against one falls to the
494
+ // range/equality *defaults* (`selectivityOf` finds no column) — honest, and labelled as such.
495
+ const { fraction, basis } = selectivityOf(plan.predicate, table);
496
+ estimate = { estRows: clampRows(child.estRows * fraction), estPages: child.estPages, basis: `HAVING: ${basis}` };
497
+ break;
498
+ }
499
+ case 'HashAggregate':
500
+ case 'SortAggregate': {
501
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
502
+ // A hash table adds no I/O of its own — everything happens in memory —
503
+ // but a sort aggregate spills exactly like `Sort` does, on top of
504
+ // whatever the child already costs to read; that extra spill, priced
505
+ // the same way `Sort`'s own case below prices it, is the entire point
506
+ // of comparison `/compare`'s third mode shows (plan.md §22.2).
507
+ const spillPages = plan.op === 'SortAggregate' ? 2 * Math.max(1, Math.ceil(child.estRows / rowsPerPage)) : 0;
508
+ if (plan.groupBy.length === 0) {
509
+ // No GROUP BY: the whole input is one group — even zero input rows
510
+ // still produce one, since COUNT(*) of nothing is 0, not absent.
511
+ estimate = {
512
+ estRows: 1,
513
+ estPages: child.estPages + spillPages,
514
+ basis: plan.op === 'SortAggregate'
515
+ ? `no GROUP BY — one group, but sorted first anyway (${String(spillPages)} spill pages)`
516
+ : 'no GROUP BY — the whole input is one group',
517
+ };
518
+ }
519
+ else {
520
+ // Textbook independence assumption, same as an AND's selectivity:
521
+ // the product of each column's distinct-value count, capped at the
522
+ // rows actually going in — there cannot be more groups than rows.
523
+ const groups = plan.groupBy.reduce((product, col) => product * ndistinct(table, col), 1);
524
+ const groupBasis = `${plan.groupBy.map((c) => `ndistinct(${c})`).join(' × ')}, capped at the input`;
525
+ estimate = {
526
+ estRows: Math.max(1, Math.min(child.estRows, clampRows(groups))),
527
+ estPages: child.estPages + spillPages,
528
+ basis: plan.op === 'SortAggregate'
529
+ ? `${groupBasis}; sorts ${String(child.estRows)} rows first (${String(spillPages)} spill pages)`
530
+ : groupBasis,
531
+ };
532
+ }
533
+ break;
534
+ }
535
+ case 'Sort': {
536
+ // Row order changes, not row count — but Sort's own I/O is the honest
537
+ // number to surface here: a spill writes and re-reads the data, so its
538
+ // page cost is real disk activity a Filter/Project never adds.
539
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
540
+ const spillPages = Math.max(1, Math.ceil(child.estRows / rowsPerPage));
541
+ estimate = {
542
+ estRows: child.estRows,
543
+ estPages: 2 * spillPages,
544
+ basis: `sorts ${String(child.estRows)} rows; at least ${String(2 * spillPages)} spill pages (run generation alone)`,
545
+ };
546
+ break;
547
+ }
548
+ case 'Project': {
549
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
550
+ estimate = { estRows: child.estRows, estPages: child.estPages, basis: child.basis };
551
+ break;
552
+ }
553
+ case 'Limit': {
554
+ const child = estimateNode(plan.child, table, rowsPerPage, into, otherTables);
555
+ const offset = plan.offset ?? 0;
556
+ estimate = {
557
+ estRows: Math.min(Math.max(0, child.estRows - offset), plan.count),
558
+ estPages: child.estPages,
559
+ basis: offset > 0
560
+ ? `min(${String(child.estRows)} − ${String(offset)} skipped, LIMIT ${String(plan.count)})`
561
+ : `min(${String(child.estRows)}, LIMIT ${String(plan.count)})`,
562
+ };
563
+ break;
564
+ }
565
+ }
566
+ into.set(plan, estimate);
567
+ return estimate;
568
+ }
569
+ /**
570
+ * Per-node estimates for a plan tree. `otherTables` is the join partner's
571
+ * `Table`, keyed by name — needed only when `plan` contains a `Join`; every
572
+ * other caller (no JOIN in the query) omits it.
573
+ */
574
+ export function estimatePlan(plan, table, rowsPerPage, otherTables) {
575
+ const cost = new Map();
576
+ estimateNode(plan, table, rowsPerPage, cost, otherTables);
577
+ return cost;
578
+ }
579
+ /** The estimate for the plan's root node — the query's estimated output. */
580
+ export function rootEstimate(plan, table, rowsPerPage, otherTables) {
581
+ return estimateNode(plan, table, rowsPerPage, new Map(), otherTables);
582
+ }