querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,445 @@
1
+ /**
2
+ * The logical plan. v0 keeps one representation rather than separate logical
3
+ * and physical trees: the scan node carries its own access method, which is
4
+ * what the optimizer actually chooses (plan.md §7.2–7.4).
5
+ */
6
+ import { columnsOfIndex } from "../index/spec.js";
7
+ import { exprToSql } from "../parser/print.js";
8
+ import { walkExpr } from "../parser/index.js";
9
+ export function childOf(plan) {
10
+ return plan.op === 'Filter' ||
11
+ plan.op === 'Having' ||
12
+ plan.op === 'HashDistinct' ||
13
+ plan.op === 'SortDistinct' ||
14
+ plan.op === 'HashAggregate' ||
15
+ plan.op === 'SortAggregate' ||
16
+ plan.op === 'Sort' ||
17
+ plan.op === 'Project' ||
18
+ plan.op === 'Limit'
19
+ ? plan.child
20
+ : undefined;
21
+ }
22
+ /**
23
+ * The `Sort` node, wherever it sits in the chain — `runQuery.ts` needs to
24
+ * know before the buffer pool exists, to reserve its spill's page ids. v0's
25
+ * plan is always a single chain (no joins yet), so walking `childOf` visits
26
+ * every node.
27
+ */
28
+ export function findSort(plan) {
29
+ let node = plan;
30
+ while (node) {
31
+ if (node.op === 'Sort')
32
+ return node;
33
+ node = childOf(node);
34
+ }
35
+ return null;
36
+ }
37
+ /**
38
+ * Whether `plan` contains a node whose executor spills through the buffer
39
+ * pool via `externalMergeSort` — `Sort` itself, or a `SortAggregate` (which
40
+ * runs the same algorithm internally to order its groups). `runQuery.ts`
41
+ * needs this before the buffer pool exists, to reserve the spill's page ids;
42
+ * `Sort` and `SortAggregate` never both appear in one plan (`ORDER BY` and
43
+ * `GROUP BY` are mutually exclusive), but the check does not depend on that.
44
+ */
45
+ /** Whether the plan has a `SortDistinct` — which reserves a spill block of its own, after every other sort's. */
46
+ export function findSortDistinct(plan) {
47
+ let node = plan;
48
+ while (node) {
49
+ if (node.op === 'SortDistinct')
50
+ return node;
51
+ node = childOf(node);
52
+ }
53
+ return null;
54
+ }
55
+ export function needsSortSpill(plan) {
56
+ let node = plan;
57
+ while (node) {
58
+ if (node.op === 'Sort' || node.op === 'SortAggregate')
59
+ return true;
60
+ node = childOf(node);
61
+ }
62
+ return false;
63
+ }
64
+ /**
65
+ * Every `Join` node in `plan`, in source order — the first `JOIN` clause first, the last (outermost in the tree,
66
+ * plan.md §25.4 C3 slice b) last. `childOf` stops at a `Join` (it has `left`/`right`, not `child`), so finding more
67
+ * than the outermost one means recursing into `left` on purpose: a left-deep chain's `right` side is always a base
68
+ * table's own scan or probe, never itself a `Join`, so only `left` is ever worth walking further into. `ORDER
69
+ * BY`/`GROUP BY` still can't combine with `JOIN` at all, so this and `needsSortSpill` never both find something on
70
+ * the same plan.
71
+ */
72
+ export function joinsIn(plan) {
73
+ let node = plan;
74
+ while (node && node.op !== 'Join')
75
+ node = childOf(node);
76
+ if (!node)
77
+ return [];
78
+ return [...joinsIn(node.left), node];
79
+ }
80
+ /**
81
+ * Every `Join` in the chain that runs the `sort-merge` algorithm — `runQuery.ts` needs these before the buffer pool
82
+ * exists, to reserve scratch page ids for each one's own two internal sorts (each side its own `externalMergeSort`
83
+ * call, plan.md §22.2).
84
+ */
85
+ export function findSortMergeJoins(plan) {
86
+ return joinsIn(plan).filter((j) => j.algorithm === 'sort-merge');
87
+ }
88
+ /**
89
+ * Every `Join` in the chain that runs the `index-nested-loop` algorithm — `runQuery.ts` needs these before the
90
+ * buffer pool exists, to build each one's own inner table's B+Tree (plan.md §25.4 B4) and give its pages ids of
91
+ * their own.
92
+ */
93
+ export function findIndexNestedLoopJoins(plan) {
94
+ return joinsIn(plan).filter((j) => j.algorithm === 'index-nested-loop');
95
+ }
96
+ /* -------------------------------------------------------------------------- */
97
+ /* Predicate helpers */
98
+ /* -------------------------------------------------------------------------- */
99
+ /** Flattens an AND tree into its individual comparisons. */
100
+ export function conjuncts(expr) {
101
+ return expr.kind === 'and'
102
+ ? [...conjuncts(expr.left), ...conjuncts(expr.right)]
103
+ : [expr];
104
+ }
105
+ /** Rebuilds a left-associative AND chain. Returns undefined for an empty list. */
106
+ export function conjoin(parts) {
107
+ if (parts.length === 0)
108
+ return undefined;
109
+ return parts.reduce((left, right) => ({
110
+ kind: 'and',
111
+ left,
112
+ right,
113
+ span: { from: left.span.from, to: right.span.to },
114
+ }));
115
+ }
116
+ /**
117
+ * What every `IN (subquery)` in `where` resolved to (plan.md §25.4 C3 slice c) — `subquery.ts`'s
118
+ * `resolveWhereSubqueries` already ran each one exactly once, before this plan was even built, and filled its
119
+ * `items` with the real values found; this just says so, once, for the WHERE-clause narration (`emit.ts`,
120
+ * `emitDelete.ts`, `emitUpdate.ts` all share it). Empty when `where` has none — the ordinary case, and the only
121
+ * one before this slice existed at all.
122
+ */
123
+ export function subqueryResolutionNotes(where) {
124
+ const notes = [];
125
+ walkExpr(where, (e) => {
126
+ if (e.kind === 'in' && e.subquery) {
127
+ notes.push(`its subquery \`SELECT ${e.subquery.column.name} FROM ${e.subquery.from.name}${e.subquery.where ? ' WHERE …' : ''}\` already ran once, before this plan was built, finding ${e.items.length} distinct value${e.items.length === 1 ? '' : 's'}`);
128
+ }
129
+ });
130
+ return notes;
131
+ }
132
+ /**
133
+ * `col = literal` against the given column, in either operand order.
134
+ *
135
+ * A NULL literal is never a key: `col = NULL` is UNKNOWN for every row, so it
136
+ * matches nothing whatever the index holds — and NULLs are not in the tree at
137
+ * all (`buildIndex`). Returning `undefined` leaves the comparison where it
138
+ * belongs, in the filter, which says UNKNOWN and rejects every row. (Found by
139
+ * the SQLite oracle, plan.md §25.4 A1; the same rule guards `rangeOn`.)
140
+ */
141
+ export function equalityOn(expr, column) {
142
+ if (expr.kind !== 'compare' || expr.op !== '=')
143
+ return undefined;
144
+ if (expr.left.kind === 'column' && expr.left.name === column && expr.right.kind === 'literal') {
145
+ return expr.right.value ?? undefined;
146
+ }
147
+ if (expr.right.kind === 'column' && expr.right.name === column && expr.left.kind === 'literal') {
148
+ return expr.left.value ?? undefined;
149
+ }
150
+ return undefined;
151
+ }
152
+ const FLIPPED_COMPARE_OP = {
153
+ '<': '>',
154
+ '>': '<',
155
+ '<=': '>=',
156
+ '>=': '<=',
157
+ };
158
+ /**
159
+ * `col OP literal` against the given column, for `OP` in `<`, `>`, `<=`,
160
+ * `>=` — in either operand order, with the operator flipped when the column
161
+ * is on the right (`30 < age` reads as `age > 30`). `rangeIndexSelection`'s
162
+ * sibling to `equalityOn`, above.
163
+ */
164
+ export function rangeOn(expr, column) {
165
+ if (expr.kind !== 'compare' || expr.op === '=' || expr.op === '<>')
166
+ return undefined;
167
+ if (expr.left.kind === 'column' && expr.left.name === column && expr.right.kind === 'literal') {
168
+ return expr.right.value === null ? undefined : { op: expr.op, value: expr.right.value };
169
+ }
170
+ if (expr.right.kind === 'column' && expr.right.name === column && expr.left.kind === 'literal') {
171
+ return expr.left.value === null ? undefined : { op: FLIPPED_COMPARE_OP[expr.op], value: expr.left.value };
172
+ }
173
+ return undefined;
174
+ }
175
+ /** `age ≥ 30`, `age < 50`, or `age between 20 and 40` — an `IndexRangeScan`'s bounds, in words. */
176
+ export function formatRangeBounds(column, low, high) {
177
+ if (low && high)
178
+ return `\`${column}\` between ${format(low.value)} and ${format(high.value)}`;
179
+ if (low)
180
+ return `\`${column}\` ${low.inclusive ? '≥' : '>'} ${format(low.value)}`;
181
+ if (high)
182
+ return `\`${column}\` ${high.inclusive ? '≤' : '<'} ${format(high.value)}`;
183
+ return `\`${column}\``;
184
+ }
185
+ /** The equality values an `IndexScan` / `IndexOnlyScan` looks up, in index-column order — one, unless the index is composite. */
186
+ export const lookupKeysOf = (plan) => plan.keys ?? [plan.key];
187
+ /** An index's name as it reads in a plan label: `id`, or `(lastName, firstName)` for a composite one. */
188
+ export const indexLabel = (name) => (columnsOfIndex(name).length > 1 ? `(${name})` : name);
189
+ /** `a = 1 AND b = 'x'` — the equalities a lookup on the index `name` makes, for as many leading columns as it has keys. */
190
+ export function lookupText(name, keys) {
191
+ const columns = columnsOfIndex(name);
192
+ return keys.map((value, i) => `${columns[i] ?? name} = ${format(value)}`).join(' AND ');
193
+ }
194
+ /** Re-exported: the printer lives beside the AST (`parser/print.ts`) so `selectItemKey` can use it. */
195
+ export { exprToSql } from "../parser/print.js";
196
+ /* -------------------------------------------------------------------------- */
197
+ /* Display conversion */
198
+ /* -------------------------------------------------------------------------- */
199
+ /**
200
+ * `changed` holds the Plan objects a rewrite produced. Identity comparison is
201
+ * exact — no id bookkeeping to drift — and lets the UI highlight precisely
202
+ * what the rule touched.
203
+ *
204
+ * `estimates` is optional per-node cost (plan.md §22.2). When supplied, each
205
+ * node carries `estRows` and the heuristic behind it; the pipeline always
206
+ * passes it, callers that only want the tree shape can leave it out.
207
+ */
208
+ export function toPlanDisplay(plan, changed = new Set(), estimates) {
209
+ return convert(plan, 'p', changed, estimates);
210
+ }
211
+ /** A predicate over `column` for a range bound, as SQL — `age >= 30`. */
212
+ function boundSql(column, bound, side) {
213
+ const op = side === 'low' ? (bound.inclusive ? '>=' : '>') : bound.inclusive ? '<=' : '<';
214
+ return `${column} ${op} ${format(bound.value)}`;
215
+ }
216
+ /**
217
+ * The relational-algebra reading of a plan node (plan.md §25.4 B5). Physical
218
+ * choices vanish — a sequential scan with a filter and an index lookup both
219
+ * become σ over the relation — which is exactly the point: the algebra says
220
+ * *what* is computed, the plan tree says *how*.
221
+ */
222
+ function raOf(plan) {
223
+ switch (plan.op) {
224
+ case 'SeqScan':
225
+ return plan.filter
226
+ ? { symbol: 'σ', sub: exprToSql(plan.filter), rel: plan.table }
227
+ : { symbol: 'rel', rel: plan.table };
228
+ case 'IndexProbe':
229
+ // In algebra the inner side of a join is just the relation; *how* it is reached is the physical plan's business.
230
+ return { symbol: 'rel', rel: plan.table };
231
+ case 'IndexScan':
232
+ case 'IndexOnlyScan': {
233
+ const parts = [lookupText(plan.column, lookupKeysOf(plan))];
234
+ if (plan.op === 'IndexScan' && plan.residual)
235
+ parts.push(exprToSql(plan.residual));
236
+ return { symbol: 'σ', sub: parts.join(' AND '), rel: plan.table };
237
+ }
238
+ case 'IndexRangeScan': {
239
+ const rangeColumn = columnsOfIndex(plan.column)[plan.prefix?.length ?? 0] ?? plan.column;
240
+ const parts = [
241
+ ...(plan.prefix && plan.prefix.length > 0 ? [lookupText(plan.column, plan.prefix)] : []),
242
+ ...(plan.low ? [boundSql(rangeColumn, plan.low, 'low')] : []),
243
+ ...(plan.high ? [boundSql(rangeColumn, plan.high, 'high')] : []),
244
+ ...(plan.residual ? [exprToSql(plan.residual)] : []),
245
+ ];
246
+ return { symbol: 'σ', sub: parts.join(' AND '), rel: plan.table };
247
+ }
248
+ case 'Filter':
249
+ case 'Having':
250
+ return { symbol: 'σ', sub: exprToSql(plan.predicate) };
251
+ case 'Join':
252
+ return { symbol: '⋈', sub: `${plan.leftTable}.${plan.leftColumn} = ${plan.rightTable}.${plan.rightColumn}` };
253
+ case 'HashAggregate':
254
+ case 'SortAggregate': {
255
+ const keys = plan.aggregates.map((a) => a.key).join(', ');
256
+ return { symbol: 'γ', sub: plan.groupBy.length > 0 ? `${plan.groupBy.join(', ')}; ${keys}` : keys };
257
+ }
258
+ case 'Sort':
259
+ return { symbol: 'τ', sub: `${plan.column} ${plan.direction === 'asc' ? '↑' : '↓'}` };
260
+ case 'Project':
261
+ return {
262
+ symbol: 'π',
263
+ sub: plan.columns.kind === 'star'
264
+ ? '*'
265
+ : plan.columns.kind === 'columns'
266
+ ? plan.columns.names.join(', ')
267
+ : plan.columns.items.map((i) => i.sql).join(', '),
268
+ };
269
+ case 'HashDistinct':
270
+ case 'SortDistinct':
271
+ return { symbol: 'δ' };
272
+ case 'Limit':
273
+ return { symbol: 'λ', sub: plan.offset ? `${String(plan.count)}, offset ${String(plan.offset)}` : String(plan.count) };
274
+ }
275
+ }
276
+ function convert(plan, id, changed, estimates) {
277
+ return { ...convertBody(plan, id, changed, estimates), ra: raOf(plan) };
278
+ }
279
+ function convertBody(plan, id, changed, estimates) {
280
+ const highlight = changed.has(plan);
281
+ const kid = (child) => [convert(child, `${id}.0`, changed, estimates)];
282
+ const estimate = estimates?.get(plan);
283
+ const cost = estimate
284
+ ? { estRows: estimate.estRows, estBasis: estimate.basis }
285
+ : {};
286
+ switch (plan.op) {
287
+ case 'SeqScan':
288
+ return {
289
+ id,
290
+ op: 'SeqScan',
291
+ label: `SeqScan ${plan.table}`,
292
+ ...(plan.filter ? { detail: `filter: ${exprToSql(plan.filter)}` } : {}),
293
+ table: plan.table,
294
+ ...(highlight ? { highlight } : {}),
295
+ ...cost,
296
+ };
297
+ case 'Join':
298
+ return {
299
+ id,
300
+ op: 'Join',
301
+ label: `Join (${plan.algorithm}) ${plan.leftTable} × ${plan.rightTable}`,
302
+ detail: `on: ${plan.leftTable}.${plan.leftColumn} = ${plan.rightTable}.${plan.rightColumn}`,
303
+ // `rightTable` is the table this specific `Join` step introduces — unique across a chain, since the parser
304
+ // never allows a query to name the same table twice (plan.md §25.4 C3 slice b) — so it is what disambiguates
305
+ // this node from another `Join` in the same chain for EXPLAIN ANALYZE's actual-row lookup (see `table`'s
306
+ // own doc comment on `PlanNode`), the same way `plan.table` does for a scan below.
307
+ table: plan.rightTable,
308
+ ...(highlight ? { highlight } : {}),
309
+ ...cost,
310
+ children: [
311
+ convert(plan.left, `${id}.0`, changed, estimates),
312
+ convert(plan.right, `${id}.1`, changed, estimates),
313
+ ],
314
+ };
315
+ case 'IndexProbe':
316
+ return {
317
+ id,
318
+ op: 'IndexProbe',
319
+ label: `IndexProbe ${plan.table}.${indexLabel(plan.column)}`,
320
+ detail: `key: ${plan.column} = ${plan.outerTable}.${plan.outerColumn} — once per ${plan.outerTable} row`,
321
+ ...(highlight ? { highlight } : {}),
322
+ ...cost,
323
+ };
324
+ case 'IndexScan':
325
+ return {
326
+ id,
327
+ op: 'IndexScan',
328
+ label: `IndexScan ${plan.table}.${indexLabel(plan.column)}`,
329
+ detail: plan.residual
330
+ ? `key: ${lookupText(plan.column, lookupKeysOf(plan))} · recheck: ${exprToSql(plan.residual)}`
331
+ : `key: ${lookupText(plan.column, lookupKeysOf(plan))}`,
332
+ table: plan.table,
333
+ ...(highlight ? { highlight } : {}),
334
+ ...cost,
335
+ };
336
+ case 'IndexOnlyScan':
337
+ return {
338
+ id,
339
+ op: 'IndexOnlyScan',
340
+ label: `IndexOnlyScan ${plan.table}.${indexLabel(plan.column)}`,
341
+ detail: `key: ${lookupText(plan.column, lookupKeysOf(plan))} · never touches the heap`,
342
+ table: plan.table,
343
+ ...(highlight ? { highlight } : {}),
344
+ ...cost,
345
+ };
346
+ case 'IndexRangeScan':
347
+ return {
348
+ id,
349
+ op: 'IndexRangeScan',
350
+ label: `IndexRangeScan ${plan.table}.${indexLabel(plan.column)}`,
351
+ detail: (() => {
352
+ const rangeColumn = columnsOfIndex(plan.column)[plan.prefix?.length ?? 0] ?? plan.column;
353
+ const within = plan.prefix && plan.prefix.length > 0 ? `${lookupText(plan.column, plan.prefix)}, ` : '';
354
+ const range = `range: ${within}${formatRangeBounds(rangeColumn, plan.low, plan.high)}`;
355
+ return plan.residual ? `${range} · recheck: ${exprToSql(plan.residual)}` : range;
356
+ })(),
357
+ table: plan.table,
358
+ ...(highlight ? { highlight } : {}),
359
+ ...cost,
360
+ };
361
+ case 'Filter':
362
+ return {
363
+ id,
364
+ op: 'Filter',
365
+ label: 'Filter',
366
+ detail: exprToSql(plan.predicate),
367
+ ...(highlight ? { highlight } : {}),
368
+ ...cost,
369
+ children: kid(plan.child),
370
+ };
371
+ case 'Having':
372
+ return {
373
+ id,
374
+ op: 'Having',
375
+ label: 'Having',
376
+ detail: exprToSql(plan.predicate),
377
+ ...(highlight ? { highlight } : {}),
378
+ ...cost,
379
+ children: kid(plan.child),
380
+ };
381
+ case 'HashAggregate':
382
+ case 'SortAggregate':
383
+ return {
384
+ id,
385
+ op: plan.op,
386
+ label: plan.op,
387
+ detail: plan.groupBy.length > 0
388
+ ? `GROUP BY ${plan.groupBy.join(', ')}${plan.aggregates.length > 0 ? `; ${plan.aggregates.map((a) => a.key).join(', ')}` : ''}`
389
+ : plan.aggregates.map((a) => a.key).join(', '),
390
+ ...(highlight ? { highlight } : {}),
391
+ ...cost,
392
+ children: kid(plan.child),
393
+ };
394
+ case 'Sort':
395
+ return {
396
+ id,
397
+ op: 'Sort',
398
+ label: 'Sort',
399
+ detail: `${plan.column} ${plan.direction === 'asc' ? 'ASC' : 'DESC'}`,
400
+ ...(highlight ? { highlight } : {}),
401
+ ...cost,
402
+ children: kid(plan.child),
403
+ };
404
+ case 'Project':
405
+ return {
406
+ id,
407
+ op: 'Project',
408
+ label: 'Project',
409
+ detail: plan.columns.kind === 'star'
410
+ ? '*'
411
+ : plan.columns.kind === 'columns'
412
+ ? plan.columns.names.join(', ')
413
+ : plan.columns.items.map((i) => i.sql).join(', '),
414
+ ...(highlight ? { highlight } : {}),
415
+ ...cost,
416
+ children: kid(plan.child),
417
+ };
418
+ case 'HashDistinct':
419
+ case 'SortDistinct':
420
+ return {
421
+ id,
422
+ op: plan.op,
423
+ label: plan.op,
424
+ detail: plan.op === 'SortDistinct' && plan.sortKey ? `DISTINCT, sorted by ${plan.sortKey.column} ${plan.sortKey.direction.toUpperCase()}` : 'DISTINCT',
425
+ ...(highlight ? { highlight } : {}),
426
+ ...cost,
427
+ children: kid(plan.child),
428
+ };
429
+ case 'Limit':
430
+ return {
431
+ id,
432
+ op: 'Limit',
433
+ label: `Limit ${String(plan.count)}`,
434
+ ...(plan.offset ? { detail: `offset ${String(plan.offset)}` } : {}),
435
+ ...(highlight ? { highlight } : {}),
436
+ ...cost,
437
+ children: kid(plan.child),
438
+ };
439
+ }
440
+ }
441
+ function format(value) {
442
+ if (value === null)
443
+ return 'NULL';
444
+ return typeof value === 'string' ? `'${value}'` : String(value);
445
+ }
@@ -0,0 +1,120 @@
1
+ /**
2
+ * Predict-then-reveal prompts.
3
+ *
4
+ * RESEARCH.md §1: across 24 experiments, passive viewing did not reliably beat
5
+ * reading text "no matter how high the level of their epistemic fidelity" —
6
+ * the gains came from prediction and question-answering. Everything else in
7
+ * this engine is the fidelity half. This is the other half.
8
+ *
9
+ * Every prompt is generated from real engine state, so the correct answer is
10
+ * whatever the engine is actually about to do. None of it is authored.
11
+ */
12
+ function format(value) {
13
+ if (value === null)
14
+ return 'NULL';
15
+ return typeof value === 'string' ? `'${value}'` : String(value);
16
+ }
17
+ const POLICY_LABEL = {
18
+ lru: 'LRU',
19
+ clock: 'clock sweep',
20
+ fifo: 'FIFO',
21
+ 'second-chance': 'second-chance',
22
+ optimal: 'the optimal policy',
23
+ midpoint: 'midpoint LRU',
24
+ 'lru-k': 'LRU-2',
25
+ 'two-q': '2Q',
26
+ };
27
+ const POLICY_WHY = {
28
+ lru: 'LRU evicts whichever page was accessed longest ago. On a sequential scan that is simply the first one still resident.',
29
+ clock: 'The clock hand walks forward, decrementing usage counts as it goes, and takes the first frame it finds at zero with nothing pinning it.',
30
+ fifo: 'FIFO evicts whichever page has been resident longest, regardless of how often it has been used since — which is exactly what clock sweep sets out to fix.',
31
+ 'second-chance': 'Second-chance is FIFO with a reprieve: a page used since it loaded is passed over once, its reference bit cleared, and sent to the back of the queue. The victim is the oldest page that has not been touched.',
32
+ optimal: 'The optimal policy evicts the page whose next use is farthest in the future. It cannot be built for real — nothing knows the future — but it is the fewest-misses bound every real policy is judged against.',
33
+ midpoint: 'Midpoint LRU keeps an "old" sublist of pages read but not yet re-used. New pages land there, and eviction takes the least-recently-used one — so a one-pass scan flushes only itself, never the hot pages in the young sublist.',
34
+ 'lru-k': 'LRU-2 evicts by the time of each page\'s second-most-recent reference. A page touched only once has no second reference at all, so it goes first, and a page touched twice survives a one-pass scan.',
35
+ 'two-q': '2Q keeps the pages read only once in a small FIFO and promotes a page only when it is read again after leaving that FIFO. A one-pass scan fills the FIFO and evicts only its own pages, so the hot set stays.',
36
+ };
37
+ /** "Which frame gets evicted?" — the canonical prompt from the research. */
38
+ export function evictionPrompt(frames, victimFrameId, incomingPage, policy) {
39
+ const occupied = frames.filter((f) => f.pageId !== null);
40
+ if (occupied.length < 2)
41
+ return null;
42
+ return {
43
+ question: `The pool is full and page ${String(incomingPage)} has to be read in. Under ${POLICY_LABEL[policy]}, which frame gets evicted?`,
44
+ options: occupied.map((f) => ({
45
+ id: `frame-${String(f.frameId)}`,
46
+ label: `frame ${String(f.frameId)} · page ${String(f.pageId)}`,
47
+ })),
48
+ correctOptionId: `frame-${String(victimFrameId)}`,
49
+ explanation: POLICY_WHY[policy],
50
+ };
51
+ }
52
+ /** "Which child pointer does the traversal follow?" */
53
+ export function traversalPrompt(nodeId, separators, key, correctChildIndex) {
54
+ if (separators.length === 0)
55
+ return null;
56
+ const options = Array.from({ length: separators.length + 1 }, (_, i) => {
57
+ const low = i === 0 ? null : separators[i - 1];
58
+ const high = i === separators.length ? null : separators[i];
59
+ const range = low === null
60
+ ? `< ${format(high)}`
61
+ : high === null
62
+ ? `≥ ${format(low)}`
63
+ : `${format(low)} … < ${format(high)}`;
64
+ return { id: `child-${String(i)}`, label: `child ${String(i)} (${range})` };
65
+ });
66
+ return {
67
+ question: `Node ${nodeId} separates on ${separators.map(format).join(' ')}. Looking for ${format(key)} — which child pointer does the search follow?`,
68
+ options,
69
+ correctOptionId: `child-${String(correctChildIndex)}`,
70
+ explanation: 'A separator key is the lower bound of the subtree to its right. The search follows the first child whose range contains the key — which is why the tree stays balanced and the lookup costs one read per level.',
71
+ };
72
+ }
73
+ /** "Where does the filter end up?" — asked before the optimizer rewrites. */
74
+ export function indexSelectionPrompt(column, key, predicate = `${column} = ${format(key)}`) {
75
+ return {
76
+ question: `There is a B+Tree index on ${column.startsWith('(') ? column : `\`${column}\``}, and the predicate is \`${predicate}\`. What should the optimizer do?`,
77
+ options: [
78
+ { id: 'keep', label: 'Leave it as a full sequential scan' },
79
+ { id: 'push', label: 'Turn the scan into an index lookup on that key' },
80
+ { id: 'drop', label: 'Drop the filter — the index makes it redundant' },
81
+ ],
82
+ correctOptionId: 'push',
83
+ explanation: 'The index finds the matching row directly, so the other pages are never read. Dropping the filter would be wrong in general: an index narrows the search, it does not verify the whole predicate — anything the index cannot answer is rechecked on each row it returns.',
84
+ };
85
+ }
86
+ /**
87
+ * "Can the index answer this?" — asked when a composite index is entered through only some of its leading columns
88
+ * (plan.md §25.4 B3): the leftmost-prefix rule. The wrong answers are the two natural mistakes: that the index needs
89
+ * *every* column to be constrained, and that it can serve a predicate on any one of its columns.
90
+ */
91
+ export function leftmostPrefixPrompt(index, columns, named) {
92
+ const missing = columns.filter((c) => !named.includes(c));
93
+ return {
94
+ question: `There is one B+Tree index on (${columns.join(', ')}), and the predicate is an equality on ${named.map((c) => `\`${c}\``).join(' and ')} only. Can the optimizer use the index?`,
95
+ options: [
96
+ { id: 'yes', label: `Yes — ${named.length === 1 ? 'that is the index\'s leftmost column' : 'those are its leftmost columns'}, so it can be entered there and read along the matching run` },
97
+ { id: 'no-all', label: `No — a composite index only helps when every one of its columns is constrained` },
98
+ { id: 'no-scan', label: `No — an index on several columns is only used for a full scan of the table anyway` },
99
+ ],
100
+ correctOptionId: 'yes',
101
+ explanation: `The entries are ordered by \`${columns[0]}\` first, then by the next column within each value of the one before. An equality on the leading column${named.length > 1 ? 's' : ''} selects one contiguous run of entries, so \`${index}\` is usable — the columns it does not name (${missing.map((c) => `\`${c}\``).join(', ')}) only order the entries inside that run. The reverse is not true: an equality on \`${missing[0]}\` alone matches entries scattered through the whole index, so it could not be used.`,
102
+ };
103
+ }
104
+ /** "How many rows come back?" — asked before the result is revealed. */
105
+ export function resultPrompt(actual, tableRows) {
106
+ const candidates = [...new Set([0, 1, actual, tableRows])].sort((a, b) => a - b);
107
+ if (candidates.length < 2)
108
+ return null;
109
+ return {
110
+ question: 'Before the rows are revealed: how many does this query return?',
111
+ options: candidates.map((n) => ({
112
+ id: `rows-${String(n)}`,
113
+ label: n === tableRows ? `${String(n)} (every row)` : String(n),
114
+ })),
115
+ correctOptionId: `rows-${String(actual)}`,
116
+ explanation: actual === 0
117
+ ? 'Nothing matched the predicate. A query reading many pages can still return nothing — work done is not the same as rows produced.'
118
+ : `${String(actual)} of ${String(tableRows)} rows satisfied the predicate. Every other row was read and discarded, which is exactly what the buffer pool numbers were counting.`,
119
+ };
120
+ }