querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,445 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The logical plan. v0 keeps one representation rather than separate logical
|
|
3
|
+
* and physical trees: the scan node carries its own access method, which is
|
|
4
|
+
* what the optimizer actually chooses (plan.md §7.2–7.4).
|
|
5
|
+
*/
|
|
6
|
+
import { columnsOfIndex } from "../index/spec.js";
|
|
7
|
+
import { exprToSql } from "../parser/print.js";
|
|
8
|
+
import { walkExpr } from "../parser/index.js";
|
|
9
|
+
export function childOf(plan) {
|
|
10
|
+
return plan.op === 'Filter' ||
|
|
11
|
+
plan.op === 'Having' ||
|
|
12
|
+
plan.op === 'HashDistinct' ||
|
|
13
|
+
plan.op === 'SortDistinct' ||
|
|
14
|
+
plan.op === 'HashAggregate' ||
|
|
15
|
+
plan.op === 'SortAggregate' ||
|
|
16
|
+
plan.op === 'Sort' ||
|
|
17
|
+
plan.op === 'Project' ||
|
|
18
|
+
plan.op === 'Limit'
|
|
19
|
+
? plan.child
|
|
20
|
+
: undefined;
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* The `Sort` node, wherever it sits in the chain — `runQuery.ts` needs to
|
|
24
|
+
* know before the buffer pool exists, to reserve its spill's page ids. v0's
|
|
25
|
+
* plan is always a single chain (no joins yet), so walking `childOf` visits
|
|
26
|
+
* every node.
|
|
27
|
+
*/
|
|
28
|
+
export function findSort(plan) {
|
|
29
|
+
let node = plan;
|
|
30
|
+
while (node) {
|
|
31
|
+
if (node.op === 'Sort')
|
|
32
|
+
return node;
|
|
33
|
+
node = childOf(node);
|
|
34
|
+
}
|
|
35
|
+
return null;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Whether `plan` contains a node whose executor spills through the buffer
|
|
39
|
+
* pool via `externalMergeSort` — `Sort` itself, or a `SortAggregate` (which
|
|
40
|
+
* runs the same algorithm internally to order its groups). `runQuery.ts`
|
|
41
|
+
* needs this before the buffer pool exists, to reserve the spill's page ids;
|
|
42
|
+
* `Sort` and `SortAggregate` never both appear in one plan (`ORDER BY` and
|
|
43
|
+
* `GROUP BY` are mutually exclusive), but the check does not depend on that.
|
|
44
|
+
*/
|
|
45
|
+
/** Whether the plan has a `SortDistinct` — which reserves a spill block of its own, after every other sort's. */
|
|
46
|
+
export function findSortDistinct(plan) {
|
|
47
|
+
let node = plan;
|
|
48
|
+
while (node) {
|
|
49
|
+
if (node.op === 'SortDistinct')
|
|
50
|
+
return node;
|
|
51
|
+
node = childOf(node);
|
|
52
|
+
}
|
|
53
|
+
return null;
|
|
54
|
+
}
|
|
55
|
+
export function needsSortSpill(plan) {
|
|
56
|
+
let node = plan;
|
|
57
|
+
while (node) {
|
|
58
|
+
if (node.op === 'Sort' || node.op === 'SortAggregate')
|
|
59
|
+
return true;
|
|
60
|
+
node = childOf(node);
|
|
61
|
+
}
|
|
62
|
+
return false;
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Every `Join` node in `plan`, in source order — the first `JOIN` clause first, the last (outermost in the tree,
|
|
66
|
+
* plan.md §25.4 C3 slice b) last. `childOf` stops at a `Join` (it has `left`/`right`, not `child`), so finding more
|
|
67
|
+
* than the outermost one means recursing into `left` on purpose: a left-deep chain's `right` side is always a base
|
|
68
|
+
* table's own scan or probe, never itself a `Join`, so only `left` is ever worth walking further into. `ORDER
|
|
69
|
+
* BY`/`GROUP BY` still can't combine with `JOIN` at all, so this and `needsSortSpill` never both find something on
|
|
70
|
+
* the same plan.
|
|
71
|
+
*/
|
|
72
|
+
export function joinsIn(plan) {
|
|
73
|
+
let node = plan;
|
|
74
|
+
while (node && node.op !== 'Join')
|
|
75
|
+
node = childOf(node);
|
|
76
|
+
if (!node)
|
|
77
|
+
return [];
|
|
78
|
+
return [...joinsIn(node.left), node];
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Every `Join` in the chain that runs the `sort-merge` algorithm — `runQuery.ts` needs these before the buffer pool
|
|
82
|
+
* exists, to reserve scratch page ids for each one's own two internal sorts (each side its own `externalMergeSort`
|
|
83
|
+
* call, plan.md §22.2).
|
|
84
|
+
*/
|
|
85
|
+
export function findSortMergeJoins(plan) {
|
|
86
|
+
return joinsIn(plan).filter((j) => j.algorithm === 'sort-merge');
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Every `Join` in the chain that runs the `index-nested-loop` algorithm — `runQuery.ts` needs these before the
|
|
90
|
+
* buffer pool exists, to build each one's own inner table's B+Tree (plan.md §25.4 B4) and give its pages ids of
|
|
91
|
+
* their own.
|
|
92
|
+
*/
|
|
93
|
+
export function findIndexNestedLoopJoins(plan) {
|
|
94
|
+
return joinsIn(plan).filter((j) => j.algorithm === 'index-nested-loop');
|
|
95
|
+
}
|
|
96
|
+
/* -------------------------------------------------------------------------- */
|
|
97
|
+
/* Predicate helpers */
|
|
98
|
+
/* -------------------------------------------------------------------------- */
|
|
99
|
+
/** Flattens an AND tree into its individual comparisons. */
|
|
100
|
+
export function conjuncts(expr) {
|
|
101
|
+
return expr.kind === 'and'
|
|
102
|
+
? [...conjuncts(expr.left), ...conjuncts(expr.right)]
|
|
103
|
+
: [expr];
|
|
104
|
+
}
|
|
105
|
+
/** Rebuilds a left-associative AND chain. Returns undefined for an empty list. */
|
|
106
|
+
export function conjoin(parts) {
|
|
107
|
+
if (parts.length === 0)
|
|
108
|
+
return undefined;
|
|
109
|
+
return parts.reduce((left, right) => ({
|
|
110
|
+
kind: 'and',
|
|
111
|
+
left,
|
|
112
|
+
right,
|
|
113
|
+
span: { from: left.span.from, to: right.span.to },
|
|
114
|
+
}));
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* What every `IN (subquery)` in `where` resolved to (plan.md §25.4 C3 slice c) — `subquery.ts`'s
|
|
118
|
+
* `resolveWhereSubqueries` already ran each one exactly once, before this plan was even built, and filled its
|
|
119
|
+
* `items` with the real values found; this just says so, once, for the WHERE-clause narration (`emit.ts`,
|
|
120
|
+
* `emitDelete.ts`, `emitUpdate.ts` all share it). Empty when `where` has none — the ordinary case, and the only
|
|
121
|
+
* one before this slice existed at all.
|
|
122
|
+
*/
|
|
123
|
+
export function subqueryResolutionNotes(where) {
|
|
124
|
+
const notes = [];
|
|
125
|
+
walkExpr(where, (e) => {
|
|
126
|
+
if (e.kind === 'in' && e.subquery) {
|
|
127
|
+
notes.push(`its subquery \`SELECT ${e.subquery.column.name} FROM ${e.subquery.from.name}${e.subquery.where ? ' WHERE …' : ''}\` already ran once, before this plan was built, finding ${e.items.length} distinct value${e.items.length === 1 ? '' : 's'}`);
|
|
128
|
+
}
|
|
129
|
+
});
|
|
130
|
+
return notes;
|
|
131
|
+
}
|
|
132
|
+
/**
|
|
133
|
+
* `col = literal` against the given column, in either operand order.
|
|
134
|
+
*
|
|
135
|
+
* A NULL literal is never a key: `col = NULL` is UNKNOWN for every row, so it
|
|
136
|
+
* matches nothing whatever the index holds — and NULLs are not in the tree at
|
|
137
|
+
* all (`buildIndex`). Returning `undefined` leaves the comparison where it
|
|
138
|
+
* belongs, in the filter, which says UNKNOWN and rejects every row. (Found by
|
|
139
|
+
* the SQLite oracle, plan.md §25.4 A1; the same rule guards `rangeOn`.)
|
|
140
|
+
*/
|
|
141
|
+
export function equalityOn(expr, column) {
|
|
142
|
+
if (expr.kind !== 'compare' || expr.op !== '=')
|
|
143
|
+
return undefined;
|
|
144
|
+
if (expr.left.kind === 'column' && expr.left.name === column && expr.right.kind === 'literal') {
|
|
145
|
+
return expr.right.value ?? undefined;
|
|
146
|
+
}
|
|
147
|
+
if (expr.right.kind === 'column' && expr.right.name === column && expr.left.kind === 'literal') {
|
|
148
|
+
return expr.left.value ?? undefined;
|
|
149
|
+
}
|
|
150
|
+
return undefined;
|
|
151
|
+
}
|
|
152
|
+
const FLIPPED_COMPARE_OP = {
|
|
153
|
+
'<': '>',
|
|
154
|
+
'>': '<',
|
|
155
|
+
'<=': '>=',
|
|
156
|
+
'>=': '<=',
|
|
157
|
+
};
|
|
158
|
+
/**
|
|
159
|
+
* `col OP literal` against the given column, for `OP` in `<`, `>`, `<=`,
|
|
160
|
+
* `>=` — in either operand order, with the operator flipped when the column
|
|
161
|
+
* is on the right (`30 < age` reads as `age > 30`). `rangeIndexSelection`'s
|
|
162
|
+
* sibling to `equalityOn`, above.
|
|
163
|
+
*/
|
|
164
|
+
export function rangeOn(expr, column) {
|
|
165
|
+
if (expr.kind !== 'compare' || expr.op === '=' || expr.op === '<>')
|
|
166
|
+
return undefined;
|
|
167
|
+
if (expr.left.kind === 'column' && expr.left.name === column && expr.right.kind === 'literal') {
|
|
168
|
+
return expr.right.value === null ? undefined : { op: expr.op, value: expr.right.value };
|
|
169
|
+
}
|
|
170
|
+
if (expr.right.kind === 'column' && expr.right.name === column && expr.left.kind === 'literal') {
|
|
171
|
+
return expr.left.value === null ? undefined : { op: FLIPPED_COMPARE_OP[expr.op], value: expr.left.value };
|
|
172
|
+
}
|
|
173
|
+
return undefined;
|
|
174
|
+
}
|
|
175
|
+
/** `age ≥ 30`, `age < 50`, or `age between 20 and 40` — an `IndexRangeScan`'s bounds, in words. */
|
|
176
|
+
export function formatRangeBounds(column, low, high) {
|
|
177
|
+
if (low && high)
|
|
178
|
+
return `\`${column}\` between ${format(low.value)} and ${format(high.value)}`;
|
|
179
|
+
if (low)
|
|
180
|
+
return `\`${column}\` ${low.inclusive ? '≥' : '>'} ${format(low.value)}`;
|
|
181
|
+
if (high)
|
|
182
|
+
return `\`${column}\` ${high.inclusive ? '≤' : '<'} ${format(high.value)}`;
|
|
183
|
+
return `\`${column}\``;
|
|
184
|
+
}
|
|
185
|
+
/** The equality values an `IndexScan` / `IndexOnlyScan` looks up, in index-column order — one, unless the index is composite. */
|
|
186
|
+
export const lookupKeysOf = (plan) => plan.keys ?? [plan.key];
|
|
187
|
+
/** An index's name as it reads in a plan label: `id`, or `(lastName, firstName)` for a composite one. */
|
|
188
|
+
export const indexLabel = (name) => (columnsOfIndex(name).length > 1 ? `(${name})` : name);
|
|
189
|
+
/** `a = 1 AND b = 'x'` — the equalities a lookup on the index `name` makes, for as many leading columns as it has keys. */
|
|
190
|
+
export function lookupText(name, keys) {
|
|
191
|
+
const columns = columnsOfIndex(name);
|
|
192
|
+
return keys.map((value, i) => `${columns[i] ?? name} = ${format(value)}`).join(' AND ');
|
|
193
|
+
}
|
|
194
|
+
/** Re-exported: the printer lives beside the AST (`parser/print.ts`) so `selectItemKey` can use it. */
|
|
195
|
+
export { exprToSql } from "../parser/print.js";
|
|
196
|
+
/* -------------------------------------------------------------------------- */
|
|
197
|
+
/* Display conversion */
|
|
198
|
+
/* -------------------------------------------------------------------------- */
|
|
199
|
+
/**
|
|
200
|
+
* `changed` holds the Plan objects a rewrite produced. Identity comparison is
|
|
201
|
+
* exact — no id bookkeeping to drift — and lets the UI highlight precisely
|
|
202
|
+
* what the rule touched.
|
|
203
|
+
*
|
|
204
|
+
* `estimates` is optional per-node cost (plan.md §22.2). When supplied, each
|
|
205
|
+
* node carries `estRows` and the heuristic behind it; the pipeline always
|
|
206
|
+
* passes it, callers that only want the tree shape can leave it out.
|
|
207
|
+
*/
|
|
208
|
+
export function toPlanDisplay(plan, changed = new Set(), estimates) {
|
|
209
|
+
return convert(plan, 'p', changed, estimates);
|
|
210
|
+
}
|
|
211
|
+
/** A predicate over `column` for a range bound, as SQL — `age >= 30`. */
|
|
212
|
+
function boundSql(column, bound, side) {
|
|
213
|
+
const op = side === 'low' ? (bound.inclusive ? '>=' : '>') : bound.inclusive ? '<=' : '<';
|
|
214
|
+
return `${column} ${op} ${format(bound.value)}`;
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* The relational-algebra reading of a plan node (plan.md §25.4 B5). Physical
|
|
218
|
+
* choices vanish — a sequential scan with a filter and an index lookup both
|
|
219
|
+
* become σ over the relation — which is exactly the point: the algebra says
|
|
220
|
+
* *what* is computed, the plan tree says *how*.
|
|
221
|
+
*/
|
|
222
|
+
function raOf(plan) {
|
|
223
|
+
switch (plan.op) {
|
|
224
|
+
case 'SeqScan':
|
|
225
|
+
return plan.filter
|
|
226
|
+
? { symbol: 'σ', sub: exprToSql(plan.filter), rel: plan.table }
|
|
227
|
+
: { symbol: 'rel', rel: plan.table };
|
|
228
|
+
case 'IndexProbe':
|
|
229
|
+
// In algebra the inner side of a join is just the relation; *how* it is reached is the physical plan's business.
|
|
230
|
+
return { symbol: 'rel', rel: plan.table };
|
|
231
|
+
case 'IndexScan':
|
|
232
|
+
case 'IndexOnlyScan': {
|
|
233
|
+
const parts = [lookupText(plan.column, lookupKeysOf(plan))];
|
|
234
|
+
if (plan.op === 'IndexScan' && plan.residual)
|
|
235
|
+
parts.push(exprToSql(plan.residual));
|
|
236
|
+
return { symbol: 'σ', sub: parts.join(' AND '), rel: plan.table };
|
|
237
|
+
}
|
|
238
|
+
case 'IndexRangeScan': {
|
|
239
|
+
const rangeColumn = columnsOfIndex(plan.column)[plan.prefix?.length ?? 0] ?? plan.column;
|
|
240
|
+
const parts = [
|
|
241
|
+
...(plan.prefix && plan.prefix.length > 0 ? [lookupText(plan.column, plan.prefix)] : []),
|
|
242
|
+
...(plan.low ? [boundSql(rangeColumn, plan.low, 'low')] : []),
|
|
243
|
+
...(plan.high ? [boundSql(rangeColumn, plan.high, 'high')] : []),
|
|
244
|
+
...(plan.residual ? [exprToSql(plan.residual)] : []),
|
|
245
|
+
];
|
|
246
|
+
return { symbol: 'σ', sub: parts.join(' AND '), rel: plan.table };
|
|
247
|
+
}
|
|
248
|
+
case 'Filter':
|
|
249
|
+
case 'Having':
|
|
250
|
+
return { symbol: 'σ', sub: exprToSql(plan.predicate) };
|
|
251
|
+
case 'Join':
|
|
252
|
+
return { symbol: '⋈', sub: `${plan.leftTable}.${plan.leftColumn} = ${plan.rightTable}.${plan.rightColumn}` };
|
|
253
|
+
case 'HashAggregate':
|
|
254
|
+
case 'SortAggregate': {
|
|
255
|
+
const keys = plan.aggregates.map((a) => a.key).join(', ');
|
|
256
|
+
return { symbol: 'γ', sub: plan.groupBy.length > 0 ? `${plan.groupBy.join(', ')}; ${keys}` : keys };
|
|
257
|
+
}
|
|
258
|
+
case 'Sort':
|
|
259
|
+
return { symbol: 'τ', sub: `${plan.column} ${plan.direction === 'asc' ? '↑' : '↓'}` };
|
|
260
|
+
case 'Project':
|
|
261
|
+
return {
|
|
262
|
+
symbol: 'π',
|
|
263
|
+
sub: plan.columns.kind === 'star'
|
|
264
|
+
? '*'
|
|
265
|
+
: plan.columns.kind === 'columns'
|
|
266
|
+
? plan.columns.names.join(', ')
|
|
267
|
+
: plan.columns.items.map((i) => i.sql).join(', '),
|
|
268
|
+
};
|
|
269
|
+
case 'HashDistinct':
|
|
270
|
+
case 'SortDistinct':
|
|
271
|
+
return { symbol: 'δ' };
|
|
272
|
+
case 'Limit':
|
|
273
|
+
return { symbol: 'λ', sub: plan.offset ? `${String(plan.count)}, offset ${String(plan.offset)}` : String(plan.count) };
|
|
274
|
+
}
|
|
275
|
+
}
|
|
276
|
+
function convert(plan, id, changed, estimates) {
|
|
277
|
+
return { ...convertBody(plan, id, changed, estimates), ra: raOf(plan) };
|
|
278
|
+
}
|
|
279
|
+
function convertBody(plan, id, changed, estimates) {
|
|
280
|
+
const highlight = changed.has(plan);
|
|
281
|
+
const kid = (child) => [convert(child, `${id}.0`, changed, estimates)];
|
|
282
|
+
const estimate = estimates?.get(plan);
|
|
283
|
+
const cost = estimate
|
|
284
|
+
? { estRows: estimate.estRows, estBasis: estimate.basis }
|
|
285
|
+
: {};
|
|
286
|
+
switch (plan.op) {
|
|
287
|
+
case 'SeqScan':
|
|
288
|
+
return {
|
|
289
|
+
id,
|
|
290
|
+
op: 'SeqScan',
|
|
291
|
+
label: `SeqScan ${plan.table}`,
|
|
292
|
+
...(plan.filter ? { detail: `filter: ${exprToSql(plan.filter)}` } : {}),
|
|
293
|
+
table: plan.table,
|
|
294
|
+
...(highlight ? { highlight } : {}),
|
|
295
|
+
...cost,
|
|
296
|
+
};
|
|
297
|
+
case 'Join':
|
|
298
|
+
return {
|
|
299
|
+
id,
|
|
300
|
+
op: 'Join',
|
|
301
|
+
label: `Join (${plan.algorithm}) ${plan.leftTable} × ${plan.rightTable}`,
|
|
302
|
+
detail: `on: ${plan.leftTable}.${plan.leftColumn} = ${plan.rightTable}.${plan.rightColumn}`,
|
|
303
|
+
// `rightTable` is the table this specific `Join` step introduces — unique across a chain, since the parser
|
|
304
|
+
// never allows a query to name the same table twice (plan.md §25.4 C3 slice b) — so it is what disambiguates
|
|
305
|
+
// this node from another `Join` in the same chain for EXPLAIN ANALYZE's actual-row lookup (see `table`'s
|
|
306
|
+
// own doc comment on `PlanNode`), the same way `plan.table` does for a scan below.
|
|
307
|
+
table: plan.rightTable,
|
|
308
|
+
...(highlight ? { highlight } : {}),
|
|
309
|
+
...cost,
|
|
310
|
+
children: [
|
|
311
|
+
convert(plan.left, `${id}.0`, changed, estimates),
|
|
312
|
+
convert(plan.right, `${id}.1`, changed, estimates),
|
|
313
|
+
],
|
|
314
|
+
};
|
|
315
|
+
case 'IndexProbe':
|
|
316
|
+
return {
|
|
317
|
+
id,
|
|
318
|
+
op: 'IndexProbe',
|
|
319
|
+
label: `IndexProbe ${plan.table}.${indexLabel(plan.column)}`,
|
|
320
|
+
detail: `key: ${plan.column} = ${plan.outerTable}.${plan.outerColumn} — once per ${plan.outerTable} row`,
|
|
321
|
+
...(highlight ? { highlight } : {}),
|
|
322
|
+
...cost,
|
|
323
|
+
};
|
|
324
|
+
case 'IndexScan':
|
|
325
|
+
return {
|
|
326
|
+
id,
|
|
327
|
+
op: 'IndexScan',
|
|
328
|
+
label: `IndexScan ${plan.table}.${indexLabel(plan.column)}`,
|
|
329
|
+
detail: plan.residual
|
|
330
|
+
? `key: ${lookupText(plan.column, lookupKeysOf(plan))} · recheck: ${exprToSql(plan.residual)}`
|
|
331
|
+
: `key: ${lookupText(plan.column, lookupKeysOf(plan))}`,
|
|
332
|
+
table: plan.table,
|
|
333
|
+
...(highlight ? { highlight } : {}),
|
|
334
|
+
...cost,
|
|
335
|
+
};
|
|
336
|
+
case 'IndexOnlyScan':
|
|
337
|
+
return {
|
|
338
|
+
id,
|
|
339
|
+
op: 'IndexOnlyScan',
|
|
340
|
+
label: `IndexOnlyScan ${plan.table}.${indexLabel(plan.column)}`,
|
|
341
|
+
detail: `key: ${lookupText(plan.column, lookupKeysOf(plan))} · never touches the heap`,
|
|
342
|
+
table: plan.table,
|
|
343
|
+
...(highlight ? { highlight } : {}),
|
|
344
|
+
...cost,
|
|
345
|
+
};
|
|
346
|
+
case 'IndexRangeScan':
|
|
347
|
+
return {
|
|
348
|
+
id,
|
|
349
|
+
op: 'IndexRangeScan',
|
|
350
|
+
label: `IndexRangeScan ${plan.table}.${indexLabel(plan.column)}`,
|
|
351
|
+
detail: (() => {
|
|
352
|
+
const rangeColumn = columnsOfIndex(plan.column)[plan.prefix?.length ?? 0] ?? plan.column;
|
|
353
|
+
const within = plan.prefix && plan.prefix.length > 0 ? `${lookupText(plan.column, plan.prefix)}, ` : '';
|
|
354
|
+
const range = `range: ${within}${formatRangeBounds(rangeColumn, plan.low, plan.high)}`;
|
|
355
|
+
return plan.residual ? `${range} · recheck: ${exprToSql(plan.residual)}` : range;
|
|
356
|
+
})(),
|
|
357
|
+
table: plan.table,
|
|
358
|
+
...(highlight ? { highlight } : {}),
|
|
359
|
+
...cost,
|
|
360
|
+
};
|
|
361
|
+
case 'Filter':
|
|
362
|
+
return {
|
|
363
|
+
id,
|
|
364
|
+
op: 'Filter',
|
|
365
|
+
label: 'Filter',
|
|
366
|
+
detail: exprToSql(plan.predicate),
|
|
367
|
+
...(highlight ? { highlight } : {}),
|
|
368
|
+
...cost,
|
|
369
|
+
children: kid(plan.child),
|
|
370
|
+
};
|
|
371
|
+
case 'Having':
|
|
372
|
+
return {
|
|
373
|
+
id,
|
|
374
|
+
op: 'Having',
|
|
375
|
+
label: 'Having',
|
|
376
|
+
detail: exprToSql(plan.predicate),
|
|
377
|
+
...(highlight ? { highlight } : {}),
|
|
378
|
+
...cost,
|
|
379
|
+
children: kid(plan.child),
|
|
380
|
+
};
|
|
381
|
+
case 'HashAggregate':
|
|
382
|
+
case 'SortAggregate':
|
|
383
|
+
return {
|
|
384
|
+
id,
|
|
385
|
+
op: plan.op,
|
|
386
|
+
label: plan.op,
|
|
387
|
+
detail: plan.groupBy.length > 0
|
|
388
|
+
? `GROUP BY ${plan.groupBy.join(', ')}${plan.aggregates.length > 0 ? `; ${plan.aggregates.map((a) => a.key).join(', ')}` : ''}`
|
|
389
|
+
: plan.aggregates.map((a) => a.key).join(', '),
|
|
390
|
+
...(highlight ? { highlight } : {}),
|
|
391
|
+
...cost,
|
|
392
|
+
children: kid(plan.child),
|
|
393
|
+
};
|
|
394
|
+
case 'Sort':
|
|
395
|
+
return {
|
|
396
|
+
id,
|
|
397
|
+
op: 'Sort',
|
|
398
|
+
label: 'Sort',
|
|
399
|
+
detail: `${plan.column} ${plan.direction === 'asc' ? 'ASC' : 'DESC'}`,
|
|
400
|
+
...(highlight ? { highlight } : {}),
|
|
401
|
+
...cost,
|
|
402
|
+
children: kid(plan.child),
|
|
403
|
+
};
|
|
404
|
+
case 'Project':
|
|
405
|
+
return {
|
|
406
|
+
id,
|
|
407
|
+
op: 'Project',
|
|
408
|
+
label: 'Project',
|
|
409
|
+
detail: plan.columns.kind === 'star'
|
|
410
|
+
? '*'
|
|
411
|
+
: plan.columns.kind === 'columns'
|
|
412
|
+
? plan.columns.names.join(', ')
|
|
413
|
+
: plan.columns.items.map((i) => i.sql).join(', '),
|
|
414
|
+
...(highlight ? { highlight } : {}),
|
|
415
|
+
...cost,
|
|
416
|
+
children: kid(plan.child),
|
|
417
|
+
};
|
|
418
|
+
case 'HashDistinct':
|
|
419
|
+
case 'SortDistinct':
|
|
420
|
+
return {
|
|
421
|
+
id,
|
|
422
|
+
op: plan.op,
|
|
423
|
+
label: plan.op,
|
|
424
|
+
detail: plan.op === 'SortDistinct' && plan.sortKey ? `DISTINCT, sorted by ${plan.sortKey.column} ${plan.sortKey.direction.toUpperCase()}` : 'DISTINCT',
|
|
425
|
+
...(highlight ? { highlight } : {}),
|
|
426
|
+
...cost,
|
|
427
|
+
children: kid(plan.child),
|
|
428
|
+
};
|
|
429
|
+
case 'Limit':
|
|
430
|
+
return {
|
|
431
|
+
id,
|
|
432
|
+
op: 'Limit',
|
|
433
|
+
label: `Limit ${String(plan.count)}`,
|
|
434
|
+
...(plan.offset ? { detail: `offset ${String(plan.offset)}` } : {}),
|
|
435
|
+
...(highlight ? { highlight } : {}),
|
|
436
|
+
...cost,
|
|
437
|
+
children: kid(plan.child),
|
|
438
|
+
};
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
function format(value) {
|
|
442
|
+
if (value === null)
|
|
443
|
+
return 'NULL';
|
|
444
|
+
return typeof value === 'string' ? `'${value}'` : String(value);
|
|
445
|
+
}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Predict-then-reveal prompts.
|
|
3
|
+
*
|
|
4
|
+
* RESEARCH.md §1: across 24 experiments, passive viewing did not reliably beat
|
|
5
|
+
* reading text "no matter how high the level of their epistemic fidelity" —
|
|
6
|
+
* the gains came from prediction and question-answering. Everything else in
|
|
7
|
+
* this engine is the fidelity half. This is the other half.
|
|
8
|
+
*
|
|
9
|
+
* Every prompt is generated from real engine state, so the correct answer is
|
|
10
|
+
* whatever the engine is actually about to do. None of it is authored.
|
|
11
|
+
*/
|
|
12
|
+
function format(value) {
|
|
13
|
+
if (value === null)
|
|
14
|
+
return 'NULL';
|
|
15
|
+
return typeof value === 'string' ? `'${value}'` : String(value);
|
|
16
|
+
}
|
|
17
|
+
const POLICY_LABEL = {
|
|
18
|
+
lru: 'LRU',
|
|
19
|
+
clock: 'clock sweep',
|
|
20
|
+
fifo: 'FIFO',
|
|
21
|
+
'second-chance': 'second-chance',
|
|
22
|
+
optimal: 'the optimal policy',
|
|
23
|
+
midpoint: 'midpoint LRU',
|
|
24
|
+
'lru-k': 'LRU-2',
|
|
25
|
+
'two-q': '2Q',
|
|
26
|
+
};
|
|
27
|
+
const POLICY_WHY = {
|
|
28
|
+
lru: 'LRU evicts whichever page was accessed longest ago. On a sequential scan that is simply the first one still resident.',
|
|
29
|
+
clock: 'The clock hand walks forward, decrementing usage counts as it goes, and takes the first frame it finds at zero with nothing pinning it.',
|
|
30
|
+
fifo: 'FIFO evicts whichever page has been resident longest, regardless of how often it has been used since — which is exactly what clock sweep sets out to fix.',
|
|
31
|
+
'second-chance': 'Second-chance is FIFO with a reprieve: a page used since it loaded is passed over once, its reference bit cleared, and sent to the back of the queue. The victim is the oldest page that has not been touched.',
|
|
32
|
+
optimal: 'The optimal policy evicts the page whose next use is farthest in the future. It cannot be built for real — nothing knows the future — but it is the fewest-misses bound every real policy is judged against.',
|
|
33
|
+
midpoint: 'Midpoint LRU keeps an "old" sublist of pages read but not yet re-used. New pages land there, and eviction takes the least-recently-used one — so a one-pass scan flushes only itself, never the hot pages in the young sublist.',
|
|
34
|
+
'lru-k': 'LRU-2 evicts by the time of each page\'s second-most-recent reference. A page touched only once has no second reference at all, so it goes first, and a page touched twice survives a one-pass scan.',
|
|
35
|
+
'two-q': '2Q keeps the pages read only once in a small FIFO and promotes a page only when it is read again after leaving that FIFO. A one-pass scan fills the FIFO and evicts only its own pages, so the hot set stays.',
|
|
36
|
+
};
|
|
37
|
+
/** "Which frame gets evicted?" — the canonical prompt from the research. */
|
|
38
|
+
export function evictionPrompt(frames, victimFrameId, incomingPage, policy) {
|
|
39
|
+
const occupied = frames.filter((f) => f.pageId !== null);
|
|
40
|
+
if (occupied.length < 2)
|
|
41
|
+
return null;
|
|
42
|
+
return {
|
|
43
|
+
question: `The pool is full and page ${String(incomingPage)} has to be read in. Under ${POLICY_LABEL[policy]}, which frame gets evicted?`,
|
|
44
|
+
options: occupied.map((f) => ({
|
|
45
|
+
id: `frame-${String(f.frameId)}`,
|
|
46
|
+
label: `frame ${String(f.frameId)} · page ${String(f.pageId)}`,
|
|
47
|
+
})),
|
|
48
|
+
correctOptionId: `frame-${String(victimFrameId)}`,
|
|
49
|
+
explanation: POLICY_WHY[policy],
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
/** "Which child pointer does the traversal follow?" */
|
|
53
|
+
export function traversalPrompt(nodeId, separators, key, correctChildIndex) {
|
|
54
|
+
if (separators.length === 0)
|
|
55
|
+
return null;
|
|
56
|
+
const options = Array.from({ length: separators.length + 1 }, (_, i) => {
|
|
57
|
+
const low = i === 0 ? null : separators[i - 1];
|
|
58
|
+
const high = i === separators.length ? null : separators[i];
|
|
59
|
+
const range = low === null
|
|
60
|
+
? `< ${format(high)}`
|
|
61
|
+
: high === null
|
|
62
|
+
? `≥ ${format(low)}`
|
|
63
|
+
: `${format(low)} … < ${format(high)}`;
|
|
64
|
+
return { id: `child-${String(i)}`, label: `child ${String(i)} (${range})` };
|
|
65
|
+
});
|
|
66
|
+
return {
|
|
67
|
+
question: `Node ${nodeId} separates on ${separators.map(format).join(' ')}. Looking for ${format(key)} — which child pointer does the search follow?`,
|
|
68
|
+
options,
|
|
69
|
+
correctOptionId: `child-${String(correctChildIndex)}`,
|
|
70
|
+
explanation: 'A separator key is the lower bound of the subtree to its right. The search follows the first child whose range contains the key — which is why the tree stays balanced and the lookup costs one read per level.',
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
/** "Where does the filter end up?" — asked before the optimizer rewrites. */
|
|
74
|
+
export function indexSelectionPrompt(column, key, predicate = `${column} = ${format(key)}`) {
|
|
75
|
+
return {
|
|
76
|
+
question: `There is a B+Tree index on ${column.startsWith('(') ? column : `\`${column}\``}, and the predicate is \`${predicate}\`. What should the optimizer do?`,
|
|
77
|
+
options: [
|
|
78
|
+
{ id: 'keep', label: 'Leave it as a full sequential scan' },
|
|
79
|
+
{ id: 'push', label: 'Turn the scan into an index lookup on that key' },
|
|
80
|
+
{ id: 'drop', label: 'Drop the filter — the index makes it redundant' },
|
|
81
|
+
],
|
|
82
|
+
correctOptionId: 'push',
|
|
83
|
+
explanation: 'The index finds the matching row directly, so the other pages are never read. Dropping the filter would be wrong in general: an index narrows the search, it does not verify the whole predicate — anything the index cannot answer is rechecked on each row it returns.',
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* "Can the index answer this?" — asked when a composite index is entered through only some of its leading columns
|
|
88
|
+
* (plan.md §25.4 B3): the leftmost-prefix rule. The wrong answers are the two natural mistakes: that the index needs
|
|
89
|
+
* *every* column to be constrained, and that it can serve a predicate on any one of its columns.
|
|
90
|
+
*/
|
|
91
|
+
export function leftmostPrefixPrompt(index, columns, named) {
|
|
92
|
+
const missing = columns.filter((c) => !named.includes(c));
|
|
93
|
+
return {
|
|
94
|
+
question: `There is one B+Tree index on (${columns.join(', ')}), and the predicate is an equality on ${named.map((c) => `\`${c}\``).join(' and ')} only. Can the optimizer use the index?`,
|
|
95
|
+
options: [
|
|
96
|
+
{ id: 'yes', label: `Yes — ${named.length === 1 ? 'that is the index\'s leftmost column' : 'those are its leftmost columns'}, so it can be entered there and read along the matching run` },
|
|
97
|
+
{ id: 'no-all', label: `No — a composite index only helps when every one of its columns is constrained` },
|
|
98
|
+
{ id: 'no-scan', label: `No — an index on several columns is only used for a full scan of the table anyway` },
|
|
99
|
+
],
|
|
100
|
+
correctOptionId: 'yes',
|
|
101
|
+
explanation: `The entries are ordered by \`${columns[0]}\` first, then by the next column within each value of the one before. An equality on the leading column${named.length > 1 ? 's' : ''} selects one contiguous run of entries, so \`${index}\` is usable — the columns it does not name (${missing.map((c) => `\`${c}\``).join(', ')}) only order the entries inside that run. The reverse is not true: an equality on \`${missing[0]}\` alone matches entries scattered through the whole index, so it could not be used.`,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
/** "How many rows come back?" — asked before the result is revealed. */
|
|
105
|
+
export function resultPrompt(actual, tableRows) {
|
|
106
|
+
const candidates = [...new Set([0, 1, actual, tableRows])].sort((a, b) => a - b);
|
|
107
|
+
if (candidates.length < 2)
|
|
108
|
+
return null;
|
|
109
|
+
return {
|
|
110
|
+
question: 'Before the rows are revealed: how many does this query return?',
|
|
111
|
+
options: candidates.map((n) => ({
|
|
112
|
+
id: `rows-${String(n)}`,
|
|
113
|
+
label: n === tableRows ? `${String(n)} (every row)` : String(n),
|
|
114
|
+
})),
|
|
115
|
+
correctOptionId: `rows-${String(actual)}`,
|
|
116
|
+
explanation: actual === 0
|
|
117
|
+
? 'Nothing matched the predicate. A query reading many pages can still return nothing — work done is not the same as rows produced.'
|
|
118
|
+
: `${String(actual)} of ${String(tableRows)} rows satisfied the predicate. Every other row was read and discarded, which is exactly what the buffer pool numbers were counting.`,
|
|
119
|
+
};
|
|
120
|
+
}
|