querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,1290 @@
1
+ /**
2
+ * Volcano-style execution: every operator exposes `open` / `next` / `close`,
3
+ * and rows are *pulled* from the root downward. One `next()` is one natural
4
+ * visualization tick (plan.md §7.4).
5
+ *
6
+ * Operators emit through a trace buffer — the `emit` callback threaded down in
7
+ * the context — rather than returning events, so the shape of the operator
8
+ * tree is unaffected by the fact that it is being watched.
9
+ */
10
+ import { lookupKeysOf } from "../planner/index.js";
11
+ import { columnsOfIndex, emitIndexLookup, emitRangeStart, nextInRange, pageCountFor, resolveClustered } from "../index/index.js";
12
+ import { compare, equalityKeyOf, evaluateValue, passes } from "./evaluate.js";
13
+ import { externalMergeSort } from "./sort.js";
14
+ /**
15
+ * Counts rows an operator has produced, so `execute` events stay cumulative. `table` disambiguates this specific
16
+ * node from another one with the same `name` elsewhere in the plan — a multi-join chain's several `SeqScan`s or
17
+ * `Join`s — so EXPLAIN ANALYZE's actual-row lookup (`explain.ts`) does not mix up two nodes that share an `op`;
18
+ * see `StepEventBody`'s own doc comment. Omitted for operators that can only appear once per query.
19
+ */
20
+ function counter(ctx, name, table) {
21
+ let produced = 0;
22
+ return (label) => {
23
+ produced += 1;
24
+ ctx.emit(label, { stage: 'execute', op: name, rowsProduced: produced, ...(table === undefined ? {} : { table }) });
25
+ };
26
+ }
27
+ /* -------------------------------------------------------------------------- */
28
+ function seqScan(plan, ctx) {
29
+ let pageIndex = 0;
30
+ let rowIndex = 0;
31
+ let pinnedFrame = null;
32
+ let lockedPage = null;
33
+ let skipped = 0;
34
+ const produced = counter(ctx, 'SeqScan', plan.table);
35
+ function releasePage() {
36
+ if (pinnedFrame !== null) {
37
+ ctx.pool.unpin(pinnedFrame);
38
+ pinnedFrame = null;
39
+ }
40
+ if (lockedPage !== null) {
41
+ ctx.locks.release(lockedPage, 'shared');
42
+ lockedPage = null;
43
+ }
44
+ }
45
+ return {
46
+ name: 'SeqScan',
47
+ open() {
48
+ pageIndex = 0;
49
+ rowIndex = 0;
50
+ },
51
+ next() {
52
+ for (;;) {
53
+ const page = ctx.heap.pages[pageIndex];
54
+ if (!page) {
55
+ releasePage();
56
+ return null;
57
+ }
58
+ if (rowIndex === 0) {
59
+ releasePage();
60
+ ctx.locks.acquire(page.pageId, 'shared', `Take a shared lock on page ${String(page.pageId)} before reading its rows.`);
61
+ lockedPage = page.pageId;
62
+ const fetched = ctx.pool.fetch(page.pageId);
63
+ if (!fetched)
64
+ return null;
65
+ // Held while its rows are read; released when we move on. Only an
66
+ // unpinned frame can be evicted.
67
+ pinnedFrame = fetched.frameId;
68
+ }
69
+ const row = page.rows[rowIndex];
70
+ if (!row) {
71
+ pageIndex += 1;
72
+ rowIndex = 0;
73
+ continue;
74
+ }
75
+ rowIndex += 1;
76
+ if (!passes(plan.filter, row)) {
77
+ skipped += 1;
78
+ continue;
79
+ }
80
+ produced(`SeqScan returns a row from page ${page.pageId}${skipped > 0 ? ` (${skipped} row${skipped === 1 ? '' : 's'} rejected so far)` : ''}.`);
81
+ return row;
82
+ }
83
+ },
84
+ close: releasePage,
85
+ };
86
+ }
87
+ /**
88
+ * Follows one index entry to its row's heap page: a shared lock, a pool fetch, a release — the real price of an
89
+ * index scan, paid once per match. Exported for the write paths, which read every page they may modify.
90
+ */
91
+ export function fetchHeapPage(ctx, pointer) {
92
+ ctx.locks.acquire(pointer.pageId, 'shared', `Lock heap page ${String(pointer.pageId)}, where the matching row lives.`);
93
+ const fetched = ctx.pool.fetch(pointer.pageId);
94
+ if (fetched)
95
+ ctx.pool.unpin(fetched.frameId);
96
+ ctx.locks.release(pointer.pageId, 'shared');
97
+ }
98
+ /**
99
+ * Does `row` satisfy the range `plan` scans? The walk already stays inside it, so this is a second, independent
100
+ * check on the row itself — with `compare`'s own rules, not the index's. It matters inside a composite index, where a
101
+ * NULL in the range column sorts among the entries but never satisfies a range, and it means a mixed-type
102
+ * comparison the index orders one way and SQL another can only ever lose a row, never invent one.
103
+ */
104
+ export function rangeHolds(row, plan) {
105
+ const column = columnsOfIndex(plan.column)[plan.prefix?.length ?? 0];
106
+ if (column === undefined)
107
+ return true;
108
+ const value = row[column] ?? null;
109
+ if (plan.low && compare(plan.low.inclusive ? '>=' : '>', value, plan.low.value) !== true)
110
+ return false;
111
+ if (plan.high && compare(plan.high.inclusive ? '<=' : '<', value, plan.high.value) !== true)
112
+ return false;
113
+ return true;
114
+ }
115
+ /** The row an index entry points at, read out of `heap` — this query's own, unless a join's other side is meant. */
116
+ export function rowAt(ctx, pointer, heap = ctx.heap) {
117
+ const page = heap.pages.find((p) => p.pageId === pointer.pageId);
118
+ return page?.rows[pointer.slot] ?? null;
119
+ }
120
+ /**
121
+ * An equality lookup. The index may hold several entries for the value (a
122
+ * non-unique column, plan.md §25.4 B3), so this pulls one row per entry: the
123
+ * lookup itself fetched the first match's heap page, and each later one is
124
+ * fetched here as its row is asked for.
125
+ */
126
+ function indexScan(plan, ctx) {
127
+ let pointers = [];
128
+ let clusterKeys;
129
+ let clusteredTree;
130
+ let clusteringColumn = '';
131
+ let cursor = 0;
132
+ const produced = counter(ctx, 'IndexScan', plan.table);
133
+ return {
134
+ name: 'IndexScan',
135
+ open() {
136
+ const tree = ctx.trees[plan.column];
137
+ if (!tree)
138
+ return;
139
+ const outcome = emitIndexLookup(ctx.emit, tree, lookupKeysOf(plan), ctx.pool, plan.column, ctx.locks);
140
+ pointers = outcome.pointers;
141
+ clusterKeys = outcome.clusterKeys;
142
+ if (tree.clusteredVia !== undefined) {
143
+ clusteringColumn = tree.clusteredVia;
144
+ clusteredTree = ctx.trees[tree.clusteredVia];
145
+ }
146
+ },
147
+ next() {
148
+ for (;;) {
149
+ const at = cursor;
150
+ cursor += 1;
151
+ const pointer = pointers[at];
152
+ if (!pointer)
153
+ return null;
154
+ // A clustered table's secondary entry has no address (plan.md §25.4 B3e): every match, the first included,
155
+ // costs a real second descent through the clustered index. Otherwise the lookup already fetched the first
156
+ // match's own heap page, and each later one is fetched here, lazily, as it is pulled.
157
+ let resolved = pointer;
158
+ const clusterKey = clusterKeys?.[at];
159
+ if (clusterKey && clusteredTree) {
160
+ resolved = resolveClustered(ctx.emit, clusteredTree, clusteringColumn, clusterKey, ctx.pool, ctx.locks);
161
+ }
162
+ else if (at > 0) {
163
+ fetchHeapPage(ctx, pointer);
164
+ }
165
+ if (!resolved)
166
+ continue;
167
+ const row = rowAt(ctx, resolved);
168
+ if (!row)
169
+ continue;
170
+ if (!passes(plan.residual, row)) {
171
+ ctx.emit('The row the index found fails the rest of the predicate, so it is discarded. The index narrowed the search; it did not verify the whole condition.', { stage: 'execute', op: 'IndexScan', rowsProduced: 0, table: plan.table });
172
+ continue;
173
+ }
174
+ produced(pointers.length === 1
175
+ ? 'IndexScan returns the single row the lookup pointed at.'
176
+ : `IndexScan returns entry ${String(at + 1)} of the ${String(pointers.length)} the lookup found.`);
177
+ return row;
178
+ }
179
+ },
180
+ close() {
181
+ pointers = [];
182
+ clusterKeys = undefined;
183
+ clusteredTree = undefined;
184
+ cursor = 0;
185
+ },
186
+ };
187
+ }
188
+ /**
189
+ * Same equality lookup as `IndexScan`, but the query needs nothing the leaf
190
+ * doesn't already hold — `column`'s own value, which is the key it was
191
+ * found by. `emitIndexLookup`'s `touchHeap: false` skips the heap fetch and
192
+ * lock entirely, so the row this returns is built from the key alone, never
193
+ * from a page read (plan.md §22.2). One row per entry the index holds for the value.
194
+ */
195
+ function indexOnlyScan(plan, ctx) {
196
+ let entries = [];
197
+ let cursor = 0;
198
+ const produced = counter(ctx, 'IndexOnlyScan', plan.table);
199
+ const columns = columnsOfIndex(plan.column);
200
+ return {
201
+ name: 'IndexOnlyScan',
202
+ open() {
203
+ const tree = ctx.trees[plan.column];
204
+ if (!tree)
205
+ return;
206
+ entries = emitIndexLookup(ctx.emit, tree, lookupKeysOf(plan), ctx.pool, plan.column, ctx.locks, false).entries;
207
+ },
208
+ next() {
209
+ const entry = entries[cursor];
210
+ if (!entry)
211
+ return null;
212
+ cursor += 1;
213
+ produced(columns.length > 1
214
+ ? `IndexOnlyScan returns ${columns.map((c) => `\`${c}\``).join(' and ')} straight from the entry — the index covers the query, so the heap is never touched.`
215
+ : 'IndexOnlyScan returns the key itself — the heap is never touched.');
216
+ // The leaf's own values, one per indexed column — a row with exactly the columns the index holds.
217
+ const row = {};
218
+ columns.forEach((column, i) => { row[column] = entry[i] ?? null; });
219
+ return row;
220
+ },
221
+ close() {
222
+ entries = [];
223
+ cursor = 0;
224
+ },
225
+ };
226
+ }
227
+ /**
228
+ * `<`, `>`, `<=`, `>=` (or a `BETWEEN`) on an indexed column (plan.md
229
+ * §23.1): descend once to the leaf the lower bound starts at, then pull
230
+ * matching rows one at a time by walking the leaf chain — `nextInRange`
231
+ * crosses a `next` sibling pointer exactly when the current leaf runs out,
232
+ * so this stays as lazy as `SeqScan` rather than materialising the whole
233
+ * range in `open()`. Always touches the heap, one page per matching row,
234
+ * the same granularity `IndexScan`'s single row already costs — no
235
+ * index-only variant yet.
236
+ */
237
+ function indexRangeScan(plan, ctx) {
238
+ let leafId = null;
239
+ let cursorIndex = 0;
240
+ let started = false;
241
+ const produced = counter(ctx, 'IndexRangeScan', plan.table);
242
+ return {
243
+ name: 'IndexRangeScan',
244
+ open() {
245
+ const tree = ctx.trees[plan.column];
246
+ if (!tree)
247
+ return;
248
+ const start = emitRangeStart(ctx.emit, tree, plan.column, plan.low, ctx.pool, ctx.locks, plan.prefix);
249
+ leafId = start.leafId;
250
+ cursorIndex = start.startIndex;
251
+ started = true;
252
+ },
253
+ next() {
254
+ if (!started)
255
+ return null;
256
+ const tree = ctx.trees[plan.column];
257
+ if (!tree)
258
+ return null;
259
+ for (;;) {
260
+ if (leafId === null)
261
+ return null;
262
+ const found = nextInRange(ctx.emit, tree, leafId, cursorIndex, plan.high, { pool: ctx.pool, locks: ctx.locks }, plan.prefix);
263
+ if (!found) {
264
+ leafId = null;
265
+ return null;
266
+ }
267
+ leafId = found.leafId;
268
+ cursorIndex = found.nextIndex;
269
+ fetchHeapPage(ctx, found.pointer);
270
+ const row = rowAt(ctx, found.pointer);
271
+ if (!row)
272
+ continue;
273
+ if (!rangeHolds(row, plan) || !passes(plan.residual, row))
274
+ continue;
275
+ produced(`IndexRangeScan returns a row via leaf ${found.leafId}.`);
276
+ return row;
277
+ }
278
+ },
279
+ close() {
280
+ leafId = null;
281
+ started = false;
282
+ },
283
+ };
284
+ }
285
+ /**
286
+ * `Filter` and — with `name: 'Having'` — HAVING (plan.md §25.4 B2): the same
287
+ * per-row predicate, but what flows through a `Having` is a *group* the
288
+ * aggregate below produced, and its aggregate operands read the values that
289
+ * aggregate already computed.
290
+ */
291
+ function filter(predicate, child, ctx, name = 'Filter') {
292
+ const produced = counter(ctx, name);
293
+ let rejected = 0;
294
+ return {
295
+ name,
296
+ open: child.open,
297
+ next() {
298
+ for (;;) {
299
+ const row = child.next();
300
+ if (!row)
301
+ return null;
302
+ if (!passes(predicate, row)) {
303
+ rejected += 1;
304
+ continue;
305
+ }
306
+ produced(name === 'Having'
307
+ ? `Having passes a group through${rejected > 0 ? `, having rejected ${rejected}` : ''} — its aggregates satisfy the predicate.`
308
+ : `Filter passes a row through${rejected > 0 ? `, having rejected ${rejected}` : ''}.`);
309
+ return row;
310
+ }
311
+ },
312
+ close: child.close,
313
+ };
314
+ }
315
+ function project(columns, child, ctx) {
316
+ const produced = counter(ctx, 'Project');
317
+ return {
318
+ name: 'Project',
319
+ open: child.open,
320
+ next() {
321
+ const row = child.next();
322
+ if (!row)
323
+ return null;
324
+ if (columns.kind === 'star') {
325
+ produced('Project passes the row through unchanged — `*` keeps every column.');
326
+ return row;
327
+ }
328
+ if (columns.kind === 'expressions') {
329
+ // Computed columns (plan.md §25.4 B1b): each item is evaluated against the row.
330
+ const computed = {};
331
+ for (const { key, expr } of columns.items)
332
+ computed[key] = evaluateValue(expr, row);
333
+ produced(`Project computes ${columns.items.map((i) => i.key).join(', ')} for this row.`);
334
+ return computed;
335
+ }
336
+ const narrowed = {};
337
+ for (const name of columns.names)
338
+ narrowed[name] = row[name] ?? null;
339
+ produced(`Project keeps ${columns.names.join(', ')} and drops the rest.`);
340
+ return narrowed;
341
+ },
342
+ close: child.close,
343
+ };
344
+ }
345
+ /** A NULL-safe composite key for a group — collision-free via JSON, unlike joining with a separator. */
346
+ function groupKeyOf(row, groupBy) {
347
+ return JSON.stringify(groupBy.map((col) => row[col] ?? null));
348
+ }
349
+ function computeAggregate(spec, rows) {
350
+ if (spec.fn === 'COUNT') {
351
+ return spec.column === null
352
+ ? rows.length
353
+ : rows.filter((r) => (r[spec.column] ?? null) !== null).length;
354
+ }
355
+ const values = rows.map((r) => r[spec.column] ?? null).filter((v) => v !== null);
356
+ if (values.length === 0)
357
+ return null;
358
+ switch (spec.fn) {
359
+ case 'SUM': {
360
+ const nums = values.filter((v) => typeof v === 'number');
361
+ return nums.length === 0 ? null : nums.reduce((a, b) => a + b, 0);
362
+ }
363
+ case 'AVG': {
364
+ const nums = values.filter((v) => typeof v === 'number');
365
+ return nums.length === 0 ? null : nums.reduce((a, b) => a + b, 0) / nums.length;
366
+ }
367
+ case 'MIN':
368
+ return values.reduce((min, v) => (compare('<', v, min) === true ? v : min));
369
+ case 'MAX':
370
+ return values.reduce((max, v) => (compare('>', v, max) === true ? v : max));
371
+ }
372
+ }
373
+ /** Groups `rows` by `groupBy` (or one group over everything, if empty) and computes each aggregate per group. */
374
+ function runHashAggregate(rows, groupBy, aggregates) {
375
+ if (groupBy.length === 0) {
376
+ const groupRow = {};
377
+ for (const spec of aggregates)
378
+ groupRow[spec.key] = computeAggregate(spec, rows);
379
+ return [groupRow];
380
+ }
381
+ const groups = new Map();
382
+ for (const row of rows) {
383
+ const groupKey = groupKeyOf(row, groupBy);
384
+ let bucket = groups.get(groupKey);
385
+ if (!bucket) {
386
+ bucket = { key: Object.fromEntries(groupBy.map((col) => [col, row[col] ?? null])), rows: [] };
387
+ groups.set(groupKey, bucket);
388
+ }
389
+ bucket.rows.push(row);
390
+ }
391
+ return [...groups.values()].map(({ key, rows: groupRows }) => {
392
+ const groupRow = { ...key };
393
+ for (const spec of aggregates)
394
+ groupRow[spec.key] = computeAggregate(spec, groupRows);
395
+ return groupRow;
396
+ });
397
+ }
398
+ /**
399
+ * `GROUP BY` / aggregate functions, computed with a hash table — the other
400
+ * operator here that cannot stream, for the same reason `Sort` cannot: a
401
+ * group is not known to be complete until the whole input has been seen, so
402
+ * the first `next()` call materialises the child, builds one hash-table
403
+ * bucket per distinct `groupBy` combination (a single bucket, if there is no
404
+ * `GROUP BY`), and every later call just streams the finished groups
405
+ * (plan.md §22.2 — `sortAggregate` below is the second strategy this is
406
+ * compared against).
407
+ */
408
+ function hashAggregate(plan, child, ctx) {
409
+ let rows = null;
410
+ let index = 0;
411
+ const produced = counter(ctx, 'HashAggregate');
412
+ return {
413
+ name: 'HashAggregate',
414
+ open() {
415
+ child.open();
416
+ rows = null;
417
+ index = 0;
418
+ },
419
+ next() {
420
+ if (rows === null) {
421
+ const materialised = [];
422
+ for (;;) {
423
+ const row = child.next();
424
+ if (!row)
425
+ break;
426
+ materialised.push(row);
427
+ }
428
+ child.close();
429
+ rows = runHashAggregate(materialised, plan.groupBy, plan.aggregates);
430
+ }
431
+ const row = rows[index];
432
+ if (!row)
433
+ return null;
434
+ index += 1;
435
+ produced(plan.groupBy.length > 0
436
+ ? `HashAggregate returns group ${index} of ${rows.length} (${plan.groupBy
437
+ .map((col) => `${col} = ${JSON.stringify(row[col] ?? null)}`)
438
+ .join(', ')}).`
439
+ : 'HashAggregate returns the one row the whole table collapses into.');
440
+ return row;
441
+ },
442
+ close() {
443
+ if (rows === null)
444
+ child.close();
445
+ },
446
+ };
447
+ }
448
+ /** One streaming pass over rows already sorted by `groupBy`: each run of equal keys collapses into one group. */
449
+ function collapseSortedGroups(sorted, groupBy, aggregates) {
450
+ if (groupBy.length === 0) {
451
+ const groupRow = {};
452
+ for (const spec of aggregates)
453
+ groupRow[spec.key] = computeAggregate(spec, sorted);
454
+ return [groupRow];
455
+ }
456
+ const finishBucket = (bucket) => {
457
+ const groupRow = {};
458
+ for (const col of groupBy)
459
+ groupRow[col] = bucket[0][col] ?? null;
460
+ for (const spec of aggregates)
461
+ groupRow[spec.key] = computeAggregate(spec, bucket);
462
+ return groupRow;
463
+ };
464
+ const result = [];
465
+ let bucket = [];
466
+ let bucketKey = null;
467
+ for (const row of sorted) {
468
+ const key = groupKeyOf(row, groupBy);
469
+ if (bucketKey !== null && key !== bucketKey) {
470
+ result.push(finishBucket(bucket));
471
+ bucket = [];
472
+ }
473
+ bucketKey = key;
474
+ bucket.push(row);
475
+ }
476
+ if (bucket.length > 0)
477
+ result.push(finishBucket(bucket));
478
+ return result;
479
+ }
480
+ /**
481
+ * The second aggregate strategy (plan.md §22.2): rather than a hash table,
482
+ * sort the input on `groupBy` first — the exact same `externalMergeSort`
483
+ * `Sort` uses below, so its spill genuinely competes for the same buffer
484
+ * frames a heap scan does — then one streaming pass collapses each run of
485
+ * equal keys into a group with `collapseSortedGroups`, computing every
486
+ * aggregate with the same `computeAggregate` `hashAggregate` uses. Same
487
+ * output, same row count as a `HashAggregate` over the same query; the only
488
+ * difference is *how* it gets there, which is the whole point of naming two
489
+ * strategies and putting them side by side in `/compare`.
490
+ */
491
+ function sortAggregate(plan, child, ctx) {
492
+ let rows = null;
493
+ let index = 0;
494
+ const produced = counter(ctx, 'SortAggregate');
495
+ return {
496
+ name: 'SortAggregate',
497
+ open() {
498
+ child.open();
499
+ rows = null;
500
+ index = 0;
501
+ },
502
+ next() {
503
+ if (rows === null) {
504
+ const materialised = [];
505
+ for (;;) {
506
+ const row = child.next();
507
+ if (!row)
508
+ break;
509
+ materialised.push(row);
510
+ }
511
+ child.close();
512
+ const sorted = externalMergeSort(materialised, (r) => plan.groupBy.map((col) => r[col] ?? null), 'asc', {
513
+ pool: ctx.pool,
514
+ emit: ctx.emit,
515
+ base: pageCountFor(ctx.heap, ctx.trees),
516
+ memPages: ctx.options.bufferFrames,
517
+ rowsPerPage: ctx.options.rowsPerPage,
518
+ }, 'SortAggregate', 'SortAggregate');
519
+ rows = collapseSortedGroups(sorted, plan.groupBy, plan.aggregates);
520
+ }
521
+ const row = rows[index];
522
+ if (!row)
523
+ return null;
524
+ index += 1;
525
+ produced(plan.groupBy.length > 0
526
+ ? `SortAggregate returns group ${index} of ${rows.length} (${plan.groupBy
527
+ .map((col) => `${col} = ${JSON.stringify(row[col] ?? null)}`)
528
+ .join(', ')}).`
529
+ : 'SortAggregate returns the one row the whole table collapses into.');
530
+ return row;
531
+ },
532
+ close() {
533
+ if (rows === null)
534
+ child.close();
535
+ },
536
+ };
537
+ }
538
+ /**
539
+ * `ORDER BY` — the one operator here that cannot stream. Volcano pulls a row
540
+ * at a time, but a sort cannot hand back its first row until it has seen
541
+ * every one, so the first `next()` call materialises the whole child, runs
542
+ * `externalMergeSort` (its own driver, plan.md §22.2 — real page I/O through
543
+ * `ctx.pool`, not `Array.prototype.sort`), and every later call just streams
544
+ * the already-sorted result. A `LIMIT` above an `ORDER BY` therefore caps the
545
+ * *output*, not the scan: the sort has already read everything by the time
546
+ * `Limit` sees a row.
547
+ */
548
+ function sort(plan, child, ctx) {
549
+ let rows = null;
550
+ let index = 0;
551
+ const produced = counter(ctx, 'Sort');
552
+ return {
553
+ name: 'Sort',
554
+ open() {
555
+ child.open();
556
+ rows = null;
557
+ index = 0;
558
+ },
559
+ next() {
560
+ if (rows === null) {
561
+ const materialised = [];
562
+ for (;;) {
563
+ const row = child.next();
564
+ if (!row)
565
+ break;
566
+ materialised.push(row);
567
+ }
568
+ child.close();
569
+ rows = externalMergeSort(materialised, (r) => [r[plan.column] ?? null], plan.direction, {
570
+ pool: ctx.pool,
571
+ emit: ctx.emit,
572
+ base: pageCountFor(ctx.heap, ctx.trees),
573
+ memPages: ctx.options.bufferFrames,
574
+ rowsPerPage: ctx.options.rowsPerPage,
575
+ });
576
+ }
577
+ const row = rows[index];
578
+ if (!row)
579
+ return null;
580
+ index += 1;
581
+ produced(`Sort returns row ${index} of ${rows.length}, in ${plan.direction === 'asc' ? 'ascending' : 'descending'} ${plan.column} order.`);
582
+ return row;
583
+ },
584
+ close() {
585
+ if (rows === null)
586
+ child.close();
587
+ },
588
+ };
589
+ }
590
+ /**
591
+ * Stops pulling from the child once `count` rows have been produced — the
592
+ * visible Volcano short-circuit (plan.md §22.2). Everything below a `Limit`
593
+ * that has not been pulled yet is simply never asked: a `SeqScan` beneath one
594
+ * can leave most of the table unread, which `close()` on the child makes
595
+ * final rather than merely paused.
596
+ *
597
+ * `offset` (plan.md §25.4 B1c) is the part that is *not* free: the first
598
+ * `offset` rows have to be produced by everything beneath before they can be
599
+ * thrown away, so `LIMIT 10 OFFSET 1000` reads a thousand rows to return
600
+ * ten — the reason deep OFFSET pagination gets slower the further you page.
601
+ */
602
+ function limit(count, offset, child, ctx) {
603
+ const produced = counter(ctx, 'Limit');
604
+ let taken = 0;
605
+ let skipped = 0;
606
+ let stopped = false;
607
+ return {
608
+ name: 'Limit',
609
+ open: child.open,
610
+ next() {
611
+ if (stopped || taken >= count)
612
+ return null;
613
+ if (skipped < offset) {
614
+ while (skipped < offset) {
615
+ const discarded = child.next();
616
+ if (!discarded) {
617
+ ctx.emit(`Limit discards ${String(skipped)} row${skipped === 1 ? '' : 's'} and then the input runs out — fewer than the OFFSET of ${String(offset)}, so nothing is left to return.`, { stage: 'execute', op: 'Limit', rowsProduced: 0 });
618
+ return null;
619
+ }
620
+ skipped += 1;
621
+ }
622
+ ctx.emit(`Limit discards the first ${String(offset)} row${offset === 1 ? '' : 's'}: everything beneath still produced each one, and Limit threw it away. That is the cost of OFFSET.`, { stage: 'execute', op: 'Limit', rowsProduced: 0 });
623
+ }
624
+ const row = child.next();
625
+ if (!row)
626
+ return null;
627
+ taken += 1;
628
+ const capped = taken >= count;
629
+ if (capped)
630
+ stopped = true;
631
+ produced(`Limit passes row ${taken} of ${count} through${capped ? ' — the cap is reached, so the rows beneath it are never pulled again' : ''}.`);
632
+ if (capped)
633
+ child.close();
634
+ return row;
635
+ },
636
+ close() {
637
+ if (!stopped)
638
+ child.close();
639
+ },
640
+ };
641
+ }
642
+ /**
643
+ * A stable text key for a whole row: two rows collide exactly when every column is
644
+ * *equal* — NULL equal to NULL (as DISTINCT and GROUP BY define it), and any other
645
+ * pair equal under `=`'s own coercion (`equalityKeyOf`, the same one the hash and
646
+ * sort-merge joins use). The coercion matters for consistency: the sort orders `1`,
647
+ * `'1'` and `true` as equal, so a hash key that told them apart would let the two
648
+ * DISTINCT strategies disagree on a column that mixes types.
649
+ */
650
+ function rowKey(row) {
651
+ return JSON.stringify(Object.keys(row).sort().map((k) => {
652
+ const v = row[k] ?? null;
653
+ return [k, v === null ? null : equalityKeyOf(v)];
654
+ }));
655
+ }
656
+ /**
657
+ * `SELECT DISTINCT` by hashing (plan.md §25.4 B2). A row is passed on the first
658
+ * time it is seen and dropped every time after — one hash-set lookup per row,
659
+ * no I/O of its own, and *streaming*: the first new row leaves immediately, so
660
+ * a `Limit` above it can stop the whole scan early. The price is memory
661
+ * proportional to the number of distinct rows.
662
+ */
663
+ function hashDistinct(child, ctx) {
664
+ const produced = counter(ctx, 'HashDistinct');
665
+ let seen = new Set();
666
+ let dropped = 0;
667
+ return {
668
+ name: 'HashDistinct',
669
+ open() {
670
+ child.open();
671
+ seen = new Set();
672
+ dropped = 0;
673
+ },
674
+ next() {
675
+ for (;;) {
676
+ const row = child.next();
677
+ if (!row)
678
+ return null;
679
+ const key = rowKey(row);
680
+ if (seen.has(key)) {
681
+ dropped += 1;
682
+ continue;
683
+ }
684
+ seen.add(key);
685
+ produced(`HashDistinct passes a row it has not seen before (${String(seen.size)} distinct so far${dropped > 0 ? `, ${String(dropped)} repeat${dropped === 1 ? '' : 's'} dropped` : ''}).`);
686
+ return row;
687
+ }
688
+ },
689
+ close: child.close,
690
+ };
691
+ }
692
+ /**
693
+ * `SELECT DISTINCT` by sorting (plan.md §25.4 B2) — the other half of the hash-vs-
694
+ * sort contrast `GROUP BY` already has. It cannot stream: it pulls *every* row,
695
+ * runs them through `externalMergeSort` (real spill I/O through the buffer pool,
696
+ * in its own reserved block of pages), and only then keeps one row of each run
697
+ * of equals. The sort key is the `ORDER BY` column first (so the caller's order
698
+ * survives) and then every column, which is what makes equal rows adjacent.
699
+ */
700
+ function sortDistinct(plan, child, ctx) {
701
+ const produced = counter(ctx, 'SortDistinct');
702
+ let rows = null;
703
+ let index = 0;
704
+ return {
705
+ name: 'SortDistinct',
706
+ open() {
707
+ child.open();
708
+ rows = null;
709
+ index = 0;
710
+ },
711
+ next() {
712
+ if (rows === null) {
713
+ const materialised = [];
714
+ for (;;) {
715
+ const row = child.next();
716
+ if (!row)
717
+ break;
718
+ materialised.push(row);
719
+ }
720
+ child.close();
721
+ const keyOf = (r) => {
722
+ const columns = Object.keys(r).sort();
723
+ const first = plan.sortKey ? [plan.sortKey.column] : [];
724
+ return [...first, ...columns].map((c) => r[c] ?? null);
725
+ };
726
+ const sorted = externalMergeSort(materialised, keyOf, plan.sortKey?.direction ?? 'asc', {
727
+ pool: ctx.pool,
728
+ emit: ctx.emit,
729
+ base: ctx.distinctBase ?? pageCountFor(ctx.heap, ctx.trees),
730
+ memPages: ctx.options.bufferFrames,
731
+ rowsPerPage: ctx.options.rowsPerPage,
732
+ }, 'SortDistinct', 'SortDistinct');
733
+ const kept = [];
734
+ let last = null;
735
+ for (const row of sorted) {
736
+ const key = rowKey(row);
737
+ if (key !== last)
738
+ kept.push(row);
739
+ last = key;
740
+ }
741
+ rows = kept;
742
+ }
743
+ const row = rows[index];
744
+ if (!row)
745
+ return null;
746
+ index += 1;
747
+ produced(`SortDistinct returns distinct row ${String(index)} of ${String(rows.length)}.`);
748
+ return row;
749
+ },
750
+ close() {
751
+ if (rows === null)
752
+ child.close();
753
+ },
754
+ };
755
+ }
756
+ /**
757
+ * Merges a left and right row into one, qualifying every key `table.column` so the two can never collide. `leftRow`
758
+ * came straight off a base-table scan the *first* time a query joins anything, so its own keys are still bare — but
759
+ * a later `Join` in a chain (plan.md §25.4 C3 slice b) receives, as its `left`, the *previous* Join's already-merged
760
+ * row, whose keys are already `table.column`-qualified for every table in the chain so far. Re-qualifying an
761
+ * already-qualified key would nest a table name inside another (`b.a.id`), so a key that already has a `.` in it is
762
+ * kept exactly as it is — column names themselves never contain one (the tokenizer's `.` is only ever the qualifier
763
+ * separator), so this can never mistake a real column for an already-merged one.
764
+ */
765
+ function mergeJoinedRow(leftTable, leftRow, rightTable, rightRow) {
766
+ const merged = {};
767
+ for (const [key, value] of Object.entries(leftRow))
768
+ merged[key.includes('.') ? key : `${leftTable}.${key}`] = value;
769
+ for (const [key, value] of Object.entries(rightRow))
770
+ merged[`${rightTable}.${key}`] = value;
771
+ return merged;
772
+ }
773
+ /** A `LEFT JOIN`'s own row for a `left` row that matched nothing on `right` (plan.md §25.4 C3): every one of
774
+ * `right`'s declared columns comes back NULL, qualified the same way `mergeJoinedRow` qualifies a real match —
775
+ * the row is still there, it just has nothing real to report on that side. */
776
+ function nullPaddedRow(leftTable, leftRow, rightTable, rightColumns) {
777
+ const merged = {};
778
+ for (const [key, value] of Object.entries(leftRow))
779
+ merged[key.includes('.') ? key : `${leftTable}.${key}`] = value;
780
+ for (const column of rightColumns)
781
+ merged[`${rightTable}.${column}`] = null;
782
+ return merged;
783
+ }
784
+ /**
785
+ * Reads a `Join`'s own left-hand join-key value off `row` — `column` bare if `row` came straight off a base-table
786
+ * scan (a query's first `JOIN`), or `table.column` if an earlier `Join` in the chain already merged and qualified it
787
+ * (plan.md §25.4 C3 slice b). Only ever needed for the *left* side: a left-deep chain's `right` side is always a
788
+ * fresh, unqualified base-table scan or probe, so every plain `row[plan.rightColumn]` elsewhere is still correct
789
+ * exactly as written.
790
+ */
791
+ function leftJoinValue(row, table, column) {
792
+ return (column in row ? row[column] : row[`${table}.${column}`]) ?? null;
793
+ }
794
+ /** `leftJoinValue`, coerced for equality the same way `joinKeyOf` coerces a right-side value — so sort-merge's two
795
+ * sides, sorted independently, still agree on which keys are equal. */
796
+ function leftJoinKeyOf(row, table, column) {
797
+ const value = leftJoinValue(row, table, column);
798
+ return value === null ? null : equalityKeyOf(value);
799
+ }
800
+ /**
801
+ * Nested-loop join — the simplest algorithm, and the first one v1 had
802
+ * (plan.md §22.2; `hashJoin` below is the second). For every row the left
803
+ * (outer) side produces, the right (inner) side is rescanned from scratch
804
+ * looking for a match: `right`'s operator is only ever built once and
805
+ * re-opened per outer row, exactly the reset `open()` already means for
806
+ * every other operator here. `ctx.joinPartners[plan.rightTable]` carries the right table's own heap (built by
807
+ * `runQuery.ts`, its pages numbered right after whatever came before it), so `right`'s `SeqScan` reads through a
808
+ * genuinely different table than
809
+ * `left`'s — the one thing an `ExecContext` never had to represent before a
810
+ * query could name two tables at once. `left` may itself be another `Join`
811
+ * (plan.md §25.4 C3 slice b's chain) — `buildOperator`'s own recursion
812
+ * handles that with no special case here at all.
813
+ */
814
+ function join(plan, ctx) {
815
+ const joinRight = ctx.joinPartners?.[plan.rightTable];
816
+ if (!joinRight)
817
+ throw new Error("unreachable: a Join plan implies ExecContext.joinPartners has this Join's rightTable");
818
+ const isLeft = plan.joinType === 'left';
819
+ const left = buildOperator(plan.left, ctx);
820
+ const right = buildOperator(plan.right, {
821
+ ...ctx,
822
+ table: joinRight.table,
823
+ heap: joinRight.heap,
824
+ trees: joinRight.trees,
825
+ });
826
+ const produced = counter(ctx, 'Join', plan.rightTable);
827
+ let leftRow = null;
828
+ let rightOpen = false;
829
+ let checked = 0;
830
+ // LEFT JOIN only (plan.md §25.4 C3): whether the current left row has matched anything yet.
831
+ let matched = false;
832
+ return {
833
+ name: 'Join',
834
+ open() {
835
+ left.open();
836
+ leftRow = null;
837
+ rightOpen = false;
838
+ checked = 0;
839
+ matched = false;
840
+ },
841
+ next() {
842
+ for (;;) {
843
+ if (!rightOpen) {
844
+ leftRow = left.next();
845
+ if (!leftRow)
846
+ return null;
847
+ matched = false;
848
+ right.open();
849
+ rightOpen = true;
850
+ }
851
+ const rightRow = right.next();
852
+ if (!rightRow) {
853
+ rightOpen = false;
854
+ // The whole inner table was rescanned and nothing matched this left row — a LEFT JOIN keeps it anyway.
855
+ if (isLeft && !matched) {
856
+ produced(`Join finds no match in ${plan.rightTable} for a row from ${plan.leftTable} — LEFT JOIN keeps it, with ${plan.rightTable}'s columns NULL.`);
857
+ return nullPaddedRow(plan.leftTable, leftRow, plan.rightTable, joinRight.table.columns);
858
+ }
859
+ continue;
860
+ }
861
+ checked += 1;
862
+ // NULL never matches NULL — the same three-valued rule an ordinary
863
+ // `=` predicate follows, so a NULL join key simply finds no partner.
864
+ if (compare('=', leftJoinValue(leftRow, plan.leftTable, plan.leftColumn), rightRow[plan.rightColumn] ?? null) !== true) {
865
+ continue;
866
+ }
867
+ matched = true;
868
+ produced(`Join matches a row from ${plan.leftTable} with one from ${plan.rightTable} (${checked} candidate${checked === 1 ? '' : 's'} checked so far).`);
869
+ return mergeJoinedRow(plan.leftTable, leftRow, plan.rightTable, rightRow);
870
+ }
871
+ },
872
+ close() {
873
+ left.close();
874
+ right.close();
875
+ },
876
+ };
877
+ }
878
+ /**
879
+ * Hash join — the second algorithm (plan.md §22.2). The build phase drains
880
+ * the right side once into an in-memory hash table keyed by
881
+ * `equalityKeyOf(rightColumn)` — a NULL key is never inserted, mirroring the
882
+ * "NULL never matches NULL" rule nested-loop's `compare('=', ...)` already
883
+ * enforces — reading every one of its pages exactly once through the real
884
+ * buffer pool; the probe phase then streams the left side one row at a
885
+ * time, looking each one up instead of rescanning. `equalityKeyOf` is used
886
+ * on both sides specifically so a hash join agrees with nested-loop even
887
+ * when the two columns hold mixed types (a number and a numeric string, say)
888
+ * — the two algorithms must never disagree on which rows match. No
889
+ * partitioning or spill: the table itself lives in a plain JS `Map`, the
890
+ * same simplification `HashAggregate` already makes for `GROUP BY`; a build
891
+ * side too large for memory needs partition spill (Grace hash join),
892
+ * `/tools/hash-join`'s own deferred future work.
893
+ */
894
+ function hashJoin(plan, ctx) {
895
+ const joinRight = ctx.joinPartners?.[plan.rightTable];
896
+ if (!joinRight)
897
+ throw new Error("unreachable: a Join plan implies ExecContext.joinPartners has this Join's rightTable");
898
+ const isLeft = plan.joinType === 'left';
899
+ const left = buildOperator(plan.left, ctx);
900
+ const right = buildOperator(plan.right, {
901
+ ...ctx,
902
+ table: joinRight.table,
903
+ heap: joinRight.heap,
904
+ trees: joinRight.trees,
905
+ });
906
+ const produced = counter(ctx, 'Join', plan.rightTable);
907
+ let built = null;
908
+ let leftRow = null;
909
+ let bucket = [];
910
+ let bucketIndex = 0;
911
+ let matched = 0;
912
+ return {
913
+ name: 'Join',
914
+ open() {
915
+ left.open();
916
+ right.open();
917
+ built = null;
918
+ leftRow = null;
919
+ bucket = [];
920
+ bucketIndex = 0;
921
+ matched = 0;
922
+ },
923
+ next() {
924
+ if (!built) {
925
+ const table = new Map();
926
+ for (;;) {
927
+ const row = right.next();
928
+ if (!row)
929
+ break;
930
+ const value = row[plan.rightColumn] ?? null;
931
+ if (value === null)
932
+ continue;
933
+ const key = equalityKeyOf(value);
934
+ const existing = table.get(key);
935
+ if (existing)
936
+ existing.push(row);
937
+ else
938
+ table.set(key, [row]);
939
+ }
940
+ right.close();
941
+ built = table;
942
+ ctx.emit(`Join builds an in-memory hash table over ${plan.rightTable}, keyed on ${plan.rightColumn} (${String(table.size)} distinct key${table.size === 1 ? '' : 's'}).`, { stage: 'execute', op: 'Join', rowsProduced: 0, table: plan.rightTable });
943
+ }
944
+ for (;;) {
945
+ if (bucketIndex >= bucket.length) {
946
+ leftRow = left.next();
947
+ if (!leftRow)
948
+ return null;
949
+ const value = leftJoinValue(leftRow, plan.leftTable, plan.leftColumn);
950
+ bucket = value === null ? [] : (built.get(equalityKeyOf(value)) ?? []);
951
+ bucketIndex = 0;
952
+ // Empty bucket, or a NULL key that was never even looked up — nothing matched this left row at all.
953
+ if (isLeft && bucket.length === 0) {
954
+ produced(`Join probes the hash table with a row from ${plan.leftTable} and finds nothing in ${plan.rightTable} — LEFT JOIN keeps it, with ${plan.rightTable}'s columns NULL.`);
955
+ return nullPaddedRow(plan.leftTable, leftRow, plan.rightTable, joinRight.table.columns);
956
+ }
957
+ continue;
958
+ }
959
+ const rightRow = bucket[bucketIndex];
960
+ bucketIndex += 1;
961
+ matched += 1;
962
+ produced(`Join probes the hash table with a row from ${plan.leftTable} and finds a match in ${plan.rightTable} (${matched} match${matched === 1 ? '' : 'es'} so far).`);
963
+ return mergeJoinedRow(plan.leftTable, leftRow, plan.rightTable, rightRow);
964
+ }
965
+ },
966
+ close() {
967
+ left.close();
968
+ if (!built)
969
+ right.close();
970
+ },
971
+ };
972
+ }
973
+ /**
974
+ * Index nested-loop join — the fourth algorithm (plan.md §25.4 B4). The
975
+ * nested-loop join's shape (an outer loop over `left`, an inner lookup per
976
+ * row), but the inner lookup is a descent of `right`'s own B+Tree on the join
977
+ * column instead of a rescan of its whole heap: each outer row costs the
978
+ * tree's few levels plus a heap page per matching row, whatever the inner
979
+ * table's size. That is why "a small outer over a big indexed inner" is where
980
+ * this algorithm wins — and why it is the *worst* of the four when the outer is
981
+ * large and the inner is small enough that one scan of it would have done.
982
+ *
983
+ * The inner index may repeat a value (a foreign key: several posts per
984
+ * author), so a probe returns *every* entry for the key — the first entry's
985
+ * heap page is fetched by the lookup itself and each later one as its row is
986
+ * pulled, so a `LIMIT` above the join stops the probing early. A NULL join key
987
+ * is never probed: `= NULL` is never true and an index holds no NULLs.
988
+ * Every candidate is rechecked against the ON equality, so this can never
989
+ * return a row the other three would not.
990
+ */
991
+ function indexNestedLoopJoin(plan, ctx) {
992
+ const joinRight = ctx.joinPartners?.[plan.rightTable];
993
+ if (!joinRight)
994
+ throw new Error("unreachable: a Join plan implies ExecContext.joinPartners has this Join's rightTable");
995
+ const inner = plan.right;
996
+ if (inner.op !== 'IndexProbe')
997
+ throw new Error('unreachable: an index nested-loop Join has an IndexProbe as its inner side');
998
+ const tree = joinRight.trees[inner.column];
999
+ if (!tree)
1000
+ throw new Error('unreachable: an index nested-loop Join implies its inner index was built');
1001
+ const isLeft = plan.joinType === 'left';
1002
+ const left = buildOperator(plan.left, ctx);
1003
+ const produced = counter(ctx, 'Join', plan.rightTable);
1004
+ let leftRow = null;
1005
+ let pointers = [];
1006
+ let cursor = 0;
1007
+ let probes = 0;
1008
+ let matched = 0;
1009
+ // LEFT JOIN only (plan.md §25.4 C3): whether the current left row has matched anything yet.
1010
+ let matchedThisRow = false;
1011
+ return {
1012
+ name: 'Join',
1013
+ open() {
1014
+ left.open();
1015
+ leftRow = null;
1016
+ pointers = [];
1017
+ cursor = 0;
1018
+ probes = 0;
1019
+ matched = 0;
1020
+ matchedThisRow = false;
1021
+ },
1022
+ next() {
1023
+ for (;;) {
1024
+ const pointer = pointers[cursor];
1025
+ if (pointer === undefined) {
1026
+ // Every candidate the probe found (a NULL key never even probed counts too) has been checked and rejected
1027
+ // — LEFT JOIN keeps this left row anyway, before moving on to the next one.
1028
+ if (leftRow && isLeft && !matchedThisRow) {
1029
+ const unmatched = leftRow;
1030
+ leftRow = null;
1031
+ produced(`Join's probe of ${plan.rightTable} finds nothing for a row from ${plan.leftTable} — LEFT JOIN keeps it, with ${plan.rightTable}'s columns NULL.`);
1032
+ return nullPaddedRow(plan.leftTable, unmatched, plan.rightTable, joinRight.table.columns);
1033
+ }
1034
+ leftRow = left.next();
1035
+ if (!leftRow)
1036
+ return null;
1037
+ pointers = [];
1038
+ cursor = 0;
1039
+ matchedThisRow = false;
1040
+ const key = leftJoinValue(leftRow, plan.leftTable, plan.leftColumn);
1041
+ if (key === null)
1042
+ continue; // NULL joins nothing, so there is nothing to look up
1043
+ probes += 1;
1044
+ ctx.emit(`Join takes a row from ${plan.leftTable} whose ${plan.leftColumn} is ${typeof key === 'string' ? `'${key}'` : String(key)} and probes ${plan.rightTable}'s B+Tree on ${plan.rightColumn} for it (probe ${String(probes)}).`, { stage: 'execute', op: 'Join', rowsProduced: matched, table: plan.rightTable });
1045
+ pointers = emitIndexLookup(ctx.emit, tree, [key], ctx.pool, inner.column, ctx.locks).pointers;
1046
+ continue;
1047
+ }
1048
+ const at = cursor;
1049
+ cursor += 1;
1050
+ // The lookup itself fetched the first match's heap page; each later one is fetched as its row is pulled.
1051
+ if (at > 0)
1052
+ fetchHeapPage(ctx, pointer);
1053
+ const rightRow = rowAt(ctx, pointer, joinRight.heap);
1054
+ if (!rightRow)
1055
+ continue;
1056
+ if (compare('=', leftJoinValue(leftRow, plan.leftTable, plan.leftColumn), rightRow[plan.rightColumn] ?? null) !== true)
1057
+ continue;
1058
+ matchedThisRow = true;
1059
+ matched += 1;
1060
+ produced(`Join pairs a row from ${plan.leftTable} with entry ${String(at + 1)} of ${String(pointers.length)} its probe found in ${plan.rightTable} (${String(matched)} match${matched === 1 ? '' : 'es'} so far).`);
1061
+ return mergeJoinedRow(plan.leftTable, leftRow, plan.rightTable, rightRow);
1062
+ }
1063
+ },
1064
+ close() {
1065
+ left.close();
1066
+ pointers = [];
1067
+ },
1068
+ };
1069
+ }
1070
+ /**
1071
+ * A join key, coerced the exact way `equalityKeyOf` already does for hash
1072
+ * join — so a value that hash join and nested-loop's `compare('=', ...)`
1073
+ * would call equal (a number and a numeric string, say) sorts to the *same*
1074
+ * key here too. Without that, the sort could scatter two "equal" rows apart
1075
+ * instead of adjacent, and the merge below would silently miss the match —
1076
+ * sort-merge join must never disagree with the other two on which rows
1077
+ * match, the same property hash join's own mixed-type test defends.
1078
+ */
1079
+ function joinKeyOf(row, column) {
1080
+ const value = row[column] ?? null;
1081
+ return value === null ? null : equalityKeyOf(value);
1082
+ }
1083
+ /**
1084
+ * The merge phase, as a generator: `left`/`right` are already fully sorted
1085
+ * on their own join key (ascending, NULLs last — `externalMergeSort`'s own
1086
+ * rule). NULL never matches, so once either pointer reaches the NULL tail it
1087
+ * only ever advances, never matches. A key can repeat on either side (the
1088
+ * join column need not be unique), so a matching key advances through the
1089
+ * *entire* run on both sides — every left row in the run paired with every
1090
+ * right row in it — before either pointer moves past the run, the textbook
1091
+ * refinement a plain two-pointer merge misses.
1092
+ *
1093
+ * `isLeft` (plan.md §25.4 C3): a plain merge only ever needs to look as far as
1094
+ * whichever side runs out first — once one is exhausted, nothing further can
1095
+ * match. A `LEFT JOIN` cannot stop there: every remaining `left` row, and any
1096
+ * `left` row the walk passes with no run of equal keys on `right` at all, must
1097
+ * still be yielded — paired with `null` rather than dropped — so the loop
1098
+ * keeps going until `left` itself is exhausted, not `right`.
1099
+ *
1100
+ * `leftTable` (plan.md §25.4 C3 slice b): `left`'s own rows may already carry qualified keys from an earlier `Join`
1101
+ * in a chain, so every left-side key read goes through `leftJoinKeyOf` rather than `joinKeyOf` directly — `right` is
1102
+ * always a fresh base-table scan, so its own reads are unaffected.
1103
+ */
1104
+ function* sortMergeMatches(left, right, leftTable, leftColumn, rightColumn, isLeft) {
1105
+ let i = 0;
1106
+ let j = 0;
1107
+ while (i < left.length) {
1108
+ const leftKey = leftJoinKeyOf(left[i], leftTable, leftColumn);
1109
+ if (leftKey === null) {
1110
+ if (isLeft)
1111
+ yield [left[i], null];
1112
+ i += 1;
1113
+ continue;
1114
+ }
1115
+ // Advance past whatever this key can never match: a NULL key, or one sorted strictly behind it.
1116
+ while (j < right.length) {
1117
+ const aheadKey = joinKeyOf(right[j], rightColumn);
1118
+ if (aheadKey !== null && aheadKey >= leftKey)
1119
+ break;
1120
+ j += 1;
1121
+ }
1122
+ const rightKey = j < right.length ? joinKeyOf(right[j], rightColumn) : null;
1123
+ if (rightKey !== leftKey) {
1124
+ // Either the right side is exhausted, or its next key is already past this one — nothing here matches.
1125
+ if (isLeft)
1126
+ yield [left[i], null];
1127
+ i += 1;
1128
+ continue;
1129
+ }
1130
+ const runStart = j;
1131
+ while (i < left.length && leftJoinKeyOf(left[i], leftTable, leftColumn) === leftKey) {
1132
+ for (let k = runStart; k < right.length && joinKeyOf(right[k], rightColumn) === leftKey; k++) {
1133
+ yield [left[i], right[k]];
1134
+ }
1135
+ i += 1;
1136
+ }
1137
+ j = runStart;
1138
+ while (j < right.length && joinKeyOf(right[j], rightColumn) === leftKey)
1139
+ j += 1;
1140
+ }
1141
+ }
1142
+ /**
1143
+ * Sort-merge join — the third algorithm (plan.md §22.2). Sorts both sides on
1144
+ * the join key, each its own `externalMergeSort` call (the same driver
1145
+ * `Sort` uses, real spill I/O through `ctx.pool`), then merges the two
1146
+ * sorted lists in a single pass — the classic trade against the other two:
1147
+ * no per-row rescan (nested-loop's cost) and no in-memory table sized to one
1148
+ * side (hash join's), at the price of sorting both sides first even when
1149
+ * neither would otherwise need to be. Like `Sort`, this cannot produce a row
1150
+ * until *both* sides are fully sorted, so a `LIMIT` above it caps only the
1151
+ * merge's output — the sorting has already happened by the time any row
1152
+ * comes out, the same honest gap `Sort`'s own doc comment names.
1153
+ *
1154
+ * `left`'s sort scratch starts right after every page id already spoken for; `right`'s starts right after `left`'s
1155
+ * own reservation. For a chain of several `JOIN`s (plan.md §25.4 C3 slice b), each sort-merge step's own pair of
1156
+ * bases depends on how much scratch space every *earlier* step already claimed — genuinely sequential bookkeeping,
1157
+ * not something this operator alone can work out from its own two children — so `runQuery.ts` computes every
1158
+ * sort-merge `Join`'s bases up front, in chain order, and hands them back keyed by that `Join`'s own plan-node
1159
+ * identity (`ctx.sortMergeBases`); this operator just looks its own up rather than deriving it.
1160
+ */
1161
+ function sortMergeJoin(plan, ctx) {
1162
+ const joinRight = ctx.joinPartners?.[plan.rightTable];
1163
+ if (!joinRight)
1164
+ throw new Error("unreachable: a Join plan implies ExecContext.joinPartners has this Join's rightTable");
1165
+ const bases = ctx.sortMergeBases?.get(plan);
1166
+ if (!bases)
1167
+ throw new Error('unreachable: a sort-merge Join plan implies runQuery.ts reserved its scratch bases');
1168
+ const isLeft = plan.joinType === 'left';
1169
+ const left = buildOperator(plan.left, ctx);
1170
+ const right = buildOperator(plan.right, {
1171
+ ...ctx,
1172
+ table: joinRight.table,
1173
+ heap: joinRight.heap,
1174
+ trees: joinRight.trees,
1175
+ });
1176
+ const produced = counter(ctx, 'Join', plan.rightTable);
1177
+ let matches = null;
1178
+ let matched = 0;
1179
+ return {
1180
+ name: 'Join',
1181
+ open() {
1182
+ left.open();
1183
+ right.open();
1184
+ matches = null;
1185
+ matched = 0;
1186
+ },
1187
+ next() {
1188
+ if (!matches) {
1189
+ const leftRows = [];
1190
+ for (;;) {
1191
+ const row = left.next();
1192
+ if (!row)
1193
+ break;
1194
+ leftRows.push(row);
1195
+ }
1196
+ left.close();
1197
+ const rightRows = [];
1198
+ for (;;) {
1199
+ const row = right.next();
1200
+ if (!row)
1201
+ break;
1202
+ rightRows.push(row);
1203
+ }
1204
+ right.close();
1205
+ const leftSorted = externalMergeSort(leftRows, (r) => [leftJoinKeyOf(r, plan.leftTable, plan.leftColumn)], 'asc', {
1206
+ pool: ctx.pool,
1207
+ emit: ctx.emit,
1208
+ base: bases.leftBase,
1209
+ memPages: ctx.options.bufferFrames,
1210
+ rowsPerPage: ctx.options.rowsPerPage,
1211
+ }, plan.leftTable, 'Join', plan.rightTable);
1212
+ const rightSorted = externalMergeSort(rightRows, (r) => [joinKeyOf(r, plan.rightColumn)], 'asc', {
1213
+ pool: ctx.pool,
1214
+ emit: ctx.emit,
1215
+ base: bases.rightBase,
1216
+ memPages: ctx.options.bufferFrames,
1217
+ rowsPerPage: ctx.options.rowsPerPage,
1218
+ }, plan.rightTable, 'Join', plan.rightTable);
1219
+ ctx.emit(`Join sorts ${plan.leftTable} and ${plan.rightTable} on the join key, then merges the two sorted lists in one pass.`, { stage: 'execute', op: 'Join', rowsProduced: 0, table: plan.rightTable });
1220
+ matches = sortMergeMatches(leftSorted, rightSorted, plan.leftTable, plan.leftColumn, plan.rightColumn, isLeft);
1221
+ }
1222
+ const step = matches.next();
1223
+ if (step.done)
1224
+ return null;
1225
+ const [leftRow, rightRow] = step.value;
1226
+ if (rightRow === null) {
1227
+ produced(`Join's merge passes a row from ${plan.leftTable} with no run of matching keys on ${plan.rightTable} — LEFT JOIN keeps it, with ${plan.rightTable}'s columns NULL.`);
1228
+ return nullPaddedRow(plan.leftTable, leftRow, plan.rightTable, joinRight.table.columns);
1229
+ }
1230
+ matched += 1;
1231
+ produced(`Join merges a row from ${plan.leftTable} with one from ${plan.rightTable} — both sides already sorted on the join key (${matched} match${matched === 1 ? '' : 'es'} so far).`);
1232
+ return mergeJoinedRow(plan.leftTable, leftRow, plan.rightTable, rightRow);
1233
+ },
1234
+ close() {
1235
+ if (!matches) {
1236
+ left.close();
1237
+ right.close();
1238
+ }
1239
+ },
1240
+ };
1241
+ }
1242
+ /** Dispatches a `Join` node to its own algorithm's operator. */
1243
+ function buildJoinOperator(plan, ctx) {
1244
+ switch (plan.algorithm) {
1245
+ case 'hash':
1246
+ return hashJoin(plan, ctx);
1247
+ case 'sort-merge':
1248
+ return sortMergeJoin(plan, ctx);
1249
+ case 'nested-loop':
1250
+ return join(plan, ctx);
1251
+ case 'index-nested-loop':
1252
+ return indexNestedLoopJoin(plan, ctx);
1253
+ }
1254
+ }
1255
+ /** Turns a plan tree into an operator tree of the same shape. */
1256
+ export function buildOperator(plan, ctx) {
1257
+ switch (plan.op) {
1258
+ case 'SeqScan':
1259
+ return seqScan(plan, ctx);
1260
+ case 'IndexScan':
1261
+ return indexScan(plan, ctx);
1262
+ case 'IndexOnlyScan':
1263
+ return indexOnlyScan(plan, ctx);
1264
+ case 'IndexRangeScan':
1265
+ return indexRangeScan(plan, ctx);
1266
+ case 'IndexProbe':
1267
+ // Only ever the inner side of an index nested-loop `Join`, which probes it itself, once per outer row.
1268
+ throw new Error('unreachable: an IndexProbe is driven by its Join, never built on its own');
1269
+ case 'Join':
1270
+ return buildJoinOperator(plan, ctx);
1271
+ case 'Filter':
1272
+ return filter(plan.predicate, buildOperator(plan.child, ctx), ctx);
1273
+ case 'Having':
1274
+ return filter(plan.predicate, buildOperator(plan.child, ctx), ctx, 'Having');
1275
+ case 'HashAggregate':
1276
+ return hashAggregate(plan, buildOperator(plan.child, ctx), ctx);
1277
+ case 'SortAggregate':
1278
+ return sortAggregate(plan, buildOperator(plan.child, ctx), ctx);
1279
+ case 'Sort':
1280
+ return sort(plan, buildOperator(plan.child, ctx), ctx);
1281
+ case 'Project':
1282
+ return project(plan.columns, buildOperator(plan.child, ctx), ctx);
1283
+ case 'HashDistinct':
1284
+ return hashDistinct(buildOperator(plan.child, ctx), ctx);
1285
+ case 'SortDistinct':
1286
+ return sortDistinct(plan, buildOperator(plan.child, ctx), ctx);
1287
+ case 'Limit':
1288
+ return limit(plan.count, plan.offset ?? 0, buildOperator(plan.child, ctx), ctx);
1289
+ }
1290
+ }