querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,481 @@
1
+ /**
2
+ * A B+Tree. All data lives in the leaves; internal nodes hold only separator
3
+ * keys. Leaves are chained left-to-right so a range scan can walk them.
4
+ *
5
+ * THE SPLIT RULE (plan.md §7.6) — this is where implementations go wrong:
6
+ *
7
+ * leaf split → COPY up. The separator goes to the parent *and stays in
8
+ * the leaf*, because leaves must hold every key.
9
+ * internal split → PUSH up. The middle key moves to the parent and is
10
+ * *removed* from the node.
11
+ *
12
+ * Applying one where the other belongs yields a tree that looks plausible and
13
+ * silently returns wrong rows. The property tests exist for exactly that.
14
+ *
15
+ * Fan-out is deliberately tiny (4 keys). Real fan-out is in the hundreds and
16
+ * would never split on a demo dataset — a case where the simplified model
17
+ * teaches better than the faithful one.
18
+ */
19
+ import { compareValues } from "../value.js";
20
+ export const isProbe = (key) => typeof key === 'object' && key !== null && 'bound' in key;
21
+ export const isEntry = (key) => typeof key === 'object' && key !== null && 'page' in key;
22
+ /** The entry an index over `values` holds for the row at `pointer` — or, on a clustered table's secondary index, for `clusterKey` instead (plan.md §25.4 B3e). */
23
+ export function entryKey(values, pointer, clusterKey) {
24
+ return { values, page: pointer.pageId, slot: pointer.slot, ...(clusterKey ? { clusterKey } : {}) };
25
+ }
26
+ /** The bound just before every entry with these leading values (`'low'`) or just after them all (`'high'`). */
27
+ export function probe(values, bound) {
28
+ return { values, bound };
29
+ }
30
+ /** The column values of a key: a scalar is its own one value, an entry or probe carries its list. */
31
+ export function valuesOf(key) {
32
+ return typeof key === 'object' && key !== null ? key.values : [key];
33
+ }
34
+ export const DEFAULT_MAX_KEYS = 4;
35
+ // NULL sorts before everything and equals only NULL. A NULL never reaches a plain index's leading column (it is
36
+ // not indexed), but a composite index's *later* columns can hold one — and `(0, NULL)` must not read as `(0, '')`.
37
+ // The shared order every index-shaped structure in this engine uses (`value.ts`'s own doc comment names the rest).
38
+ const compareScalar = compareValues;
39
+ /**
40
+ * Orders numbers numerically, strings lexicographically, booleans as 0/1 — and,
41
+ * for a table's index, entries by their values (column by column) and then by
42
+ * the row's address. A probe sorts before (`'low'`) or after (`'high'`) every
43
+ * entry that shares its leading values; a shorter list of values is a prefix.
44
+ */
45
+ export function compareKeys(a, b) {
46
+ const aObj = typeof a === 'object' && a !== null;
47
+ const bObj = typeof b === 'object' && b !== null;
48
+ if (!aObj && !bObj)
49
+ return compareScalar(a, b);
50
+ const av = valuesOf(a);
51
+ const bv = valuesOf(b);
52
+ const shared = Math.min(av.length, bv.length);
53
+ for (let i = 0; i < shared; i++) {
54
+ const c = compareScalar(av[i], bv[i]);
55
+ if (c !== 0)
56
+ return c;
57
+ }
58
+ // Equal on every shared value: a probe decides by its bound, otherwise the shorter list is smaller.
59
+ const rank = (key) => (isProbe(key) ? (key.bound === 'low' ? -1 : 1) : 0);
60
+ if (isProbe(a) || isProbe(b)) {
61
+ if (isProbe(a) && isProbe(b))
62
+ return rank(a) - rank(b);
63
+ return isProbe(a) ? rank(a) : -rank(b);
64
+ }
65
+ if (av.length !== bv.length)
66
+ return av.length - bv.length;
67
+ const ea = a;
68
+ const eb = b;
69
+ // A clustered table's secondary entry has no address to break the tie with (plan.md §25.4 B3e) — its clustering
70
+ // key does the job instead, exactly the way the row address does everywhere else: it is unique per row.
71
+ if (ea.clusterKey || eb.clusterKey) {
72
+ const ak = ea.clusterKey ?? [];
73
+ const bk = eb.clusterKey ?? [];
74
+ const shared2 = Math.min(ak.length, bk.length);
75
+ for (let i = 0; i < shared2; i++) {
76
+ const c = compareScalar(ak[i], bk[i]);
77
+ if (c !== 0)
78
+ return c;
79
+ }
80
+ return ak.length - bk.length;
81
+ }
82
+ return ea.page - eb.page || ea.slot - eb.slot;
83
+ }
84
+ /** A key as text for a step's detail: the value for a scalar, `value @page.slot` for an entry — or, on a clustered
85
+ * table's secondary index, `value → clustering key` (plan.md §25.4 B3e), since it holds no address at all. */
86
+ function describeKey(key) {
87
+ if (isEntry(key)) {
88
+ return key.clusterKey
89
+ ? `${key.values.map(String).join(', ')} → (${key.clusterKey.map(String).join(', ')})`
90
+ : `${key.values.map(String).join(', ')} @${String(key.page)}.${String(key.slot)}`;
91
+ }
92
+ if (isProbe(key))
93
+ return key.values.map(String).join(', ');
94
+ return String(key);
95
+ }
96
+ export function node(tree, id) {
97
+ const n = tree.nodes[id];
98
+ if (!n)
99
+ throw new Error(`B+Tree node ${id} is missing`);
100
+ return n;
101
+ }
102
+ export function createTree(maxKeys = DEFAULT_MAX_KEYS, firstPageId = 0, tiebreak = false) {
103
+ const root = {
104
+ kind: 'leaf',
105
+ id: 'n0',
106
+ pageId: firstPageId,
107
+ keys: [],
108
+ pointers: [],
109
+ next: null,
110
+ };
111
+ return {
112
+ ...(tiebreak ? { tiebreak } : {}),
113
+ rootId: root.id,
114
+ nodes: { [root.id]: root },
115
+ maxKeys,
116
+ nextNodeId: 1,
117
+ nextPageId: firstPageId + 1,
118
+ };
119
+ }
120
+ function allocate(tree, make) {
121
+ const created = make(`n${tree.nextNodeId}`, tree.nextPageId);
122
+ tree.nextNodeId += 1;
123
+ tree.nextPageId += 1;
124
+ tree.nodes[created.id] = created;
125
+ return created;
126
+ }
127
+ /** Index of the first key >= `key`. Exported for `index/rangeLookup.ts` — a range scan's own starting position within its first leaf. */
128
+ export function lowerBound(keys, key) {
129
+ let lo = 0;
130
+ let hi = keys.length;
131
+ while (lo < hi) {
132
+ const mid = (lo + hi) >> 1;
133
+ if (compareKeys(keys[mid], key) < 0)
134
+ lo = mid + 1;
135
+ else
136
+ hi = mid;
137
+ }
138
+ return lo;
139
+ }
140
+ /** Which child of an internal node covers `key`. */
141
+ export function childIndexFor(internal, key) {
142
+ let i = 0;
143
+ while (i < internal.keys.length && compareKeys(key, internal.keys[i]) >= 0)
144
+ i += 1;
145
+ return i;
146
+ }
147
+ /** Root-to-leaf node ids for `key`. The last entry is always the leaf. */
148
+ export function descend(tree, key) {
149
+ const path = [tree.rootId];
150
+ let current = node(tree, tree.rootId);
151
+ while (current.kind === 'internal') {
152
+ const childId = current.children[childIndexFor(current, key)];
153
+ path.push(childId);
154
+ current = node(tree, childId);
155
+ }
156
+ return path;
157
+ }
158
+ /** How many values an entry of this tree holds: one for a plain index, more for a composite one. */
159
+ export function entryWidth(tree) {
160
+ const first = node(tree, leftmostPath(tree).at(-1));
161
+ const key = first.keys[0];
162
+ return key === undefined ? 1 : valuesOf(key).length;
163
+ }
164
+ /**
165
+ * The lowest key of the leaf to the right of `leaf` — what its parent's separator says, and what a real engine keeps on the
166
+ * page itself as its "high key" (PostgreSQL does) so a scan can tell the run of equal values has ended *without reading the
167
+ * next page*. `null` for the rightmost leaf, and for an empty one. Pure bookkeeping: it touches no buffer-pool page.
168
+ */
169
+ export function highKey(tree, leaf) {
170
+ const first = leaf.keys[0];
171
+ if (first === undefined)
172
+ return null;
173
+ let bound = null;
174
+ let current = node(tree, tree.rootId);
175
+ while (current.kind === 'internal') {
176
+ const at = childIndexFor(current, first);
177
+ if (at < current.keys.length)
178
+ bound = current.keys[at];
179
+ current = node(tree, current.children[at]);
180
+ }
181
+ return bound;
182
+ }
183
+ /**
184
+ * Root-to-leaf node ids along the tree's leftmost edge — always child 0 at
185
+ * every internal node. The starting point for a range scan with no lower
186
+ * bound (`age < 30`, with nothing to descend *toward*), mirroring `descend`'s
187
+ * shape exactly so `index/rangeLookup.ts` can narrate either the same way.
188
+ */
189
+ export function leftmostPath(tree) {
190
+ const path = [tree.rootId];
191
+ let current = node(tree, tree.rootId);
192
+ while (current.kind === 'internal') {
193
+ const childId = current.children[0];
194
+ path.push(childId);
195
+ current = node(tree, childId);
196
+ }
197
+ return path;
198
+ }
199
+ /**
200
+ * Finds `key`. Given a plain value on a table's index — which holds entries, not bare values — it finds the *first*
201
+ * entry for that value (the one lowest in row-address order); `emitIndexLookup` is the narrated version that also
202
+ * returns the rest of the run.
203
+ */
204
+ export function search(tree, key) {
205
+ const target = tree.tiebreak && !isEntry(key) && !isProbe(key) ? probe([key], 'low') : key;
206
+ const path = descend(tree, target);
207
+ let leaf = node(tree, path[path.length - 1]);
208
+ let at = lowerBound(leaf.keys, target);
209
+ // The first entry for a value can start the *next* leaf (the probe routed left of a separator that equals it).
210
+ if (at >= leaf.keys.length && leaf.next !== null && tree.tiebreak) {
211
+ leaf = node(tree, leaf.next);
212
+ at = 0;
213
+ }
214
+ const matches = (candidate) => target === key ? compareKeys(candidate, key) === 0 : compareKeys(valuesOf(candidate)[0], key) === 0;
215
+ const found = at < leaf.keys.length && matches(leaf.keys[at]);
216
+ return {
217
+ path,
218
+ leafId: leaf.id,
219
+ slotInLeaf: found ? at : -1,
220
+ pointer: found ? leaf.pointers[at] : null,
221
+ };
222
+ }
223
+ /* -------------------------------------------------------------------------- */
224
+ /* Insert */
225
+ /* -------------------------------------------------------------------------- */
226
+ export function insert(tree, key, pointer) {
227
+ const path = descend(tree, key);
228
+ const leaf = node(tree, path[path.length - 1]);
229
+ const at = lowerBound(leaf.keys, key);
230
+ if (at < leaf.keys.length && compareKeys(leaf.keys[at], key) === 0) {
231
+ // The same key again — for a table's index that is the same *row* (the address is part of the key), so
232
+ // there is nothing new to hold; for a scalar sandbox tree, last writer wins.
233
+ leaf.pointers[at] = pointer;
234
+ return;
235
+ }
236
+ leaf.keys.splice(at, 0, key);
237
+ leaf.pointers.splice(at, 0, pointer);
238
+ if (leaf.keys.length <= tree.maxKeys)
239
+ return;
240
+ splitLeaf(tree, path, leaf);
241
+ }
242
+ function splitLeaf(tree, path, leaf) {
243
+ const mid = Math.ceil(leaf.keys.length / 2);
244
+ const right = allocate(tree, (id, pageId) => ({
245
+ kind: 'leaf',
246
+ id,
247
+ pageId,
248
+ keys: leaf.keys.slice(mid),
249
+ pointers: leaf.pointers.slice(mid),
250
+ next: leaf.next,
251
+ }));
252
+ leaf.keys = leaf.keys.slice(0, mid);
253
+ leaf.pointers = leaf.pointers.slice(0, mid);
254
+ leaf.next = right.id;
255
+ // COPY up: the separator also remains as the right leaf's first key.
256
+ const separator = right.keys[0];
257
+ insertIntoParent(tree, path.slice(0, -1), leaf.id, separator, right.id);
258
+ }
259
+ function insertIntoParent(tree, ancestors, leftId, separator, rightId) {
260
+ if (ancestors.length === 0) {
261
+ const root = allocate(tree, (id, pageId) => ({
262
+ kind: 'internal',
263
+ id,
264
+ pageId,
265
+ keys: [separator],
266
+ children: [leftId, rightId],
267
+ }));
268
+ tree.rootId = root.id;
269
+ return;
270
+ }
271
+ const parent = node(tree, ancestors[ancestors.length - 1]);
272
+ const at = parent.children.indexOf(leftId);
273
+ parent.keys.splice(at, 0, separator);
274
+ parent.children.splice(at + 1, 0, rightId);
275
+ if (parent.keys.length <= tree.maxKeys)
276
+ return;
277
+ const mid = Math.floor(parent.keys.length / 2);
278
+ // PUSH up: the middle key leaves this node entirely.
279
+ const pushed = parent.keys[mid];
280
+ const right = allocate(tree, (id, pageId) => ({
281
+ kind: 'internal',
282
+ id,
283
+ pageId,
284
+ keys: parent.keys.slice(mid + 1),
285
+ children: parent.children.slice(mid + 1),
286
+ }));
287
+ parent.keys = parent.keys.slice(0, mid);
288
+ parent.children = parent.children.slice(0, mid + 1);
289
+ insertIntoParent(tree, ancestors.slice(0, -1), parent.id, pushed, right.id);
290
+ }
291
+ /* -------------------------------------------------------------------------- */
292
+ /* Delete */
293
+ /* -------------------------------------------------------------------------- */
294
+ /**
295
+ * THE DELETE RULE (plan.md §7.6) — the mirror of the split rule, and just as
296
+ * easy to get subtly wrong:
297
+ *
298
+ * leaf underflow → borrow a key from a sibling, or MERGE with one. The
299
+ * separator between the two in the parent is *rewritten*
300
+ * to the new boundary (borrow) or *removed* (merge).
301
+ * internal underflow → borrow, or merge — and the parent's separator is
302
+ * *pulled down* into the merged node, because internal
303
+ * nodes hold no data and must not duplicate a key.
304
+ * deleted key was a separator → the copy in every ancestor that routes into
305
+ * this leaf's subtree is replaced with the leaf's new
306
+ * minimum, so `validate`'s "every separator exists as a
307
+ * leaf key" invariant still holds after any delete.
308
+ *
309
+ * "Under half full" is `ceil(maxKeys / 2)` keys for a leaf and
310
+ * `ceil((maxKeys + 1) / 2)` children for an internal node — the same floors
311
+ * `validate` enforces. The root is exempt: it may hold as little as one key
312
+ * (leaf) or two children (internal), and collapses into its only child when it
313
+ * drops below that.
314
+ */
315
+ const minLeafKeys = (tree) => Math.ceil(tree.maxKeys / 2);
316
+ const minInternalChildren = (tree) => Math.ceil((tree.maxKeys + 1) / 2);
317
+ export function deleteKey(tree, key) {
318
+ const steps = [];
319
+ const path = descend(tree, key);
320
+ const leaf = node(tree, path[path.length - 1]);
321
+ const at = lowerBound(leaf.keys, key);
322
+ if (at >= leaf.keys.length || compareKeys(leaf.keys[at], key) !== 0) {
323
+ return { found: false, steps };
324
+ }
325
+ leaf.keys.splice(at, 1);
326
+ leaf.pointers.splice(at, 1);
327
+ steps.push({ kind: 'remove', node: leaf.id, detail: describeKey(key) });
328
+ // The deleted key can only be a separator if it was this leaf's minimum.
329
+ // Rewrite it wherever an ancestor routes right into this leaf's subtree.
330
+ if (at === 0 && leaf.keys.length > 0) {
331
+ updateSeparators(tree, path, key, leaf.keys[0], steps);
332
+ }
333
+ // The root leaf is allowed to sit below the floor, empty included.
334
+ if (path.length === 1 || leaf.keys.length >= minLeafKeys(tree)) {
335
+ return { found: true, steps };
336
+ }
337
+ rebalanceLeaf(tree, path, steps);
338
+ return { found: true, steps };
339
+ }
340
+ function updateSeparators(tree, path, oldKey, newKey, steps) {
341
+ for (let i = 0; i < path.length - 1; i++) {
342
+ const parent = node(tree, path[i]);
343
+ const childIndex = parent.children.indexOf(path[i + 1]);
344
+ if (childIndex > 0 && compareKeys(parent.keys[childIndex - 1], oldKey) === 0) {
345
+ parent.keys[childIndex - 1] = newKey;
346
+ steps.push({
347
+ kind: 'update-separator',
348
+ node: parent.id,
349
+ detail: `${describeKey(oldKey)} → ${describeKey(newKey)}`,
350
+ });
351
+ }
352
+ }
353
+ }
354
+ function rebalanceLeaf(tree, path, steps) {
355
+ const leaf = node(tree, path[path.length - 1]);
356
+ const parent = node(tree, path[path.length - 2]);
357
+ const idx = parent.children.indexOf(leaf.id);
358
+ const floor = minLeafKeys(tree);
359
+ const left = idx > 0 ? node(tree, parent.children[idx - 1]) : null;
360
+ const right = idx < parent.children.length - 1
361
+ ? node(tree, parent.children[idx + 1])
362
+ : null;
363
+ if (left && left.keys.length > floor) {
364
+ leaf.keys.unshift(left.keys.pop());
365
+ leaf.pointers.unshift(left.pointers.pop());
366
+ parent.keys[idx - 1] = leaf.keys[0];
367
+ steps.push({ kind: 'borrow-left', node: leaf.id, detail: `from ${left.id}` });
368
+ return;
369
+ }
370
+ if (right && right.keys.length > floor) {
371
+ leaf.keys.push(right.keys.shift());
372
+ leaf.pointers.push(right.pointers.shift());
373
+ parent.keys[idx] = right.keys[0];
374
+ steps.push({ kind: 'borrow-right', node: leaf.id, detail: `from ${right.id}` });
375
+ return;
376
+ }
377
+ // No sibling has a key to spare — merge, and drop the separator between them.
378
+ if (left) {
379
+ left.keys.push(...leaf.keys);
380
+ left.pointers.push(...leaf.pointers);
381
+ left.next = leaf.next;
382
+ delete tree.nodes[leaf.id];
383
+ parent.keys.splice(idx - 1, 1);
384
+ parent.children.splice(idx, 1);
385
+ steps.push({ kind: 'merge', node: left.id, detail: `absorbed ${leaf.id}` });
386
+ }
387
+ else if (right) {
388
+ leaf.keys.push(...right.keys);
389
+ leaf.pointers.push(...right.pointers);
390
+ leaf.next = right.next;
391
+ delete tree.nodes[right.id];
392
+ parent.keys.splice(idx, 1);
393
+ parent.children.splice(idx + 1, 1);
394
+ steps.push({ kind: 'merge', node: leaf.id, detail: `absorbed ${right.id}` });
395
+ }
396
+ rebalanceInternal(tree, path.slice(0, -1), steps);
397
+ }
398
+ function rebalanceInternal(tree, path, steps) {
399
+ const target = node(tree, path[path.length - 1]);
400
+ if (path.length === 1) {
401
+ // The root may thin all the way to a single child; then it *is* that child.
402
+ if (target.children.length === 1) {
403
+ tree.rootId = target.children[0];
404
+ delete tree.nodes[target.id];
405
+ steps.push({ kind: 'root-collapse', node: tree.rootId, detail: `dropped ${target.id}` });
406
+ }
407
+ return;
408
+ }
409
+ if (target.children.length >= minInternalChildren(tree))
410
+ return;
411
+ const parent = node(tree, path[path.length - 2]);
412
+ const idx = parent.children.indexOf(target.id);
413
+ const left = idx > 0 ? node(tree, parent.children[idx - 1]) : null;
414
+ const right = idx < parent.children.length - 1
415
+ ? node(tree, parent.children[idx + 1])
416
+ : null;
417
+ const floor = minInternalChildren(tree);
418
+ // Borrow: rotate a child across, threading it through the parent separator.
419
+ if (left && left.children.length > floor) {
420
+ target.children.unshift(left.children.pop());
421
+ target.keys.unshift(parent.keys[idx - 1]);
422
+ parent.keys[idx - 1] = left.keys.pop();
423
+ steps.push({ kind: 'borrow-left', node: target.id, detail: `from ${left.id}` });
424
+ return;
425
+ }
426
+ if (right && right.children.length > floor) {
427
+ target.children.push(right.children.shift());
428
+ target.keys.push(parent.keys[idx]);
429
+ parent.keys[idx] = right.keys.shift();
430
+ steps.push({ kind: 'borrow-right', node: target.id, detail: `from ${right.id}` });
431
+ return;
432
+ }
433
+ // Merge: the parent separator comes down between the two nodes' children.
434
+ if (left) {
435
+ left.keys.push(parent.keys[idx - 1], ...target.keys);
436
+ left.children.push(...target.children);
437
+ delete tree.nodes[target.id];
438
+ parent.keys.splice(idx - 1, 1);
439
+ parent.children.splice(idx, 1);
440
+ steps.push({ kind: 'merge', node: left.id, detail: `absorbed ${target.id}` });
441
+ }
442
+ else if (right) {
443
+ target.keys.push(parent.keys[idx], ...right.keys);
444
+ target.children.push(...right.children);
445
+ delete tree.nodes[right.id];
446
+ parent.keys.splice(idx, 1);
447
+ parent.children.splice(idx + 1, 1);
448
+ steps.push({ kind: 'merge', node: target.id, detail: `absorbed ${right.id}` });
449
+ }
450
+ rebalanceInternal(tree, path.slice(0, -1), steps);
451
+ }
452
+ /* -------------------------------------------------------------------------- */
453
+ /* Reading */
454
+ /* -------------------------------------------------------------------------- */
455
+ /** Leaf ids left to right, following the sibling chain from the leftmost leaf. */
456
+ export function leafChain(tree) {
457
+ let current = node(tree, tree.rootId);
458
+ while (current.kind === 'internal')
459
+ current = node(tree, current.children[0]);
460
+ const ids = [];
461
+ let cursor = current;
462
+ const seen = new Set();
463
+ while (cursor && !seen.has(cursor.id)) {
464
+ seen.add(cursor.id);
465
+ ids.push(cursor.id);
466
+ cursor = cursor.next ? node(tree, cursor.next) : null;
467
+ }
468
+ return ids;
469
+ }
470
+ export function allKeys(tree) {
471
+ return leafChain(tree).flatMap((id) => node(tree, id).keys);
472
+ }
473
+ export function height(tree) {
474
+ let h = 1;
475
+ let current = node(tree, tree.rootId);
476
+ while (current.kind === 'internal') {
477
+ current = node(tree, current.children[0]);
478
+ h += 1;
479
+ }
480
+ return h;
481
+ }
@@ -0,0 +1,99 @@
1
+ import { createTree, entryKey, insert } from "./btree.js";
2
+ import { bulkLoad, DEFAULT_FILL_FACTOR } from "./bulk.js";
3
+ import { DEFAULT_MAX_KEYS } from "./btree.js";
4
+ import { indexNameOf, isClusteringIndex } from "./spec.js";
5
+ /** A sentinel — never dereferenced — for a clustered table's secondary entries (plan.md §25.4 B3e): they carry a
6
+ * `clusterKey` instead of an address, so `page`/`slot` are meaningless and must never be fetched as a heap page. */
7
+ const NO_ADDRESS = { pageId: -1, slot: -1 };
8
+ /**
9
+ * The entry `columns`'s index holds for `row`, and the pointer to store alongside it. Ordinarily that pointer *is*
10
+ * the row's real heap address — `pointer`, unchanged. On a CLUSTERED table (plan.md §25.4 B3e), the clustering key's
11
+ * own index still gets that real address (its whole point: no second structure is needed to find a row by it), but
12
+ * every OTHER index gets the row's clustering key instead, behind the sentinel `NO_ADDRESS` — a clustered leaf can
13
+ * split and move rows between pages, and the clustering key never goes stale the way a physical pointer would.
14
+ */
15
+ export function entryFor(columns, row, pointer, clusteredKey) {
16
+ if (!clusteredKey || isClusteringIndex(clusteredKey, columns)) {
17
+ return { key: entryKey(keyValuesOf(row, columns), pointer), pointer };
18
+ }
19
+ const clusterKey = keyValuesOf(row, clusteredKey);
20
+ return { key: entryKey(keyValuesOf(row, columns), NO_ADDRESS, clusterKey), pointer: NO_ADDRESS };
21
+ }
22
+ /**
23
+ * Builds one index over the heap. Index nodes are numbered starting at
24
+ * `firstPageId` (the last heap page by default), so it competes for the same
25
+ * buffer frames as the heap — which is the honest arrangement, and the
26
+ * reason an index lookup shows up in the pool at all. `firstPageId` is a
27
+ * parameter, not always `heap.pages.length`, so `buildIndexes` below can
28
+ * chain several trees over the same heap without any two ever claiming the
29
+ * same page id.
30
+ *
31
+ * Building is setup, not query execution, so it emits nothing: the trace
32
+ * describes running your query, not preparing the database.
33
+ */
34
+ export function buildIndex(heap,
35
+ /** A column name (the plain, single-column index) or a full spec (a composite one). */
36
+ index, maxKeys = DEFAULT_MAX_KEYS, firstPageId = heap.pages.length,
37
+ /** How to build it — one entry at a time (the default), or bulk-loaded and packed to a fill factor (plan.md §25.4 B3d). */
38
+ build = { method: 'incremental' },
39
+ /** The table's clustering key (plan.md §25.4 B3e), if it has one. */
40
+ clusteredKey) {
41
+ const columns = typeof index === 'string' ? [index] : index.columns;
42
+ const entries = [];
43
+ heap.pages.forEach((page) => {
44
+ page.rows.forEach((row, slot) => {
45
+ const values = keyValuesOf(row, columns);
46
+ // A NULL *leading* column is not indexed — an equality on it never matches, so a lookup has nothing to
47
+ // find and NULL-keyed rows fall back to a scan. A NULL further along (`(1, NULL)`) is an ordinary entry:
48
+ // the row must still be found by an equality on the columns before it.
49
+ if (values[0] === null)
50
+ return;
51
+ entries.push(entryFor(columns, row, { pageId: page.pageId, slot }, clusteredKey));
52
+ });
53
+ });
54
+ // Entry keys (values + row address, or a clustered secondary's clustering key), so the key may repeat: two rows
55
+ // with the same values are two distinct entries, adjacent in the leaves, and neither overwrites the other.
56
+ let tree;
57
+ if (build.method === 'bulk') {
58
+ tree = bulkLoad(entries, maxKeys, firstPageId, build.fillFactor ?? DEFAULT_FILL_FACTOR, true);
59
+ }
60
+ else {
61
+ tree = createTree(maxKeys, firstPageId, true);
62
+ for (const { key, pointer } of entries)
63
+ insert(tree, key, pointer);
64
+ }
65
+ if (typeof index !== 'string' && index.unique)
66
+ tree.unique = true;
67
+ if (clusteredKey && !isClusteringIndex(clusteredKey, columns))
68
+ tree.clusteredVia = indexNameOf(clusteredKey);
69
+ return tree;
70
+ }
71
+ /** The values of `columns` in `row`, in key order — what an index entry for that row holds. */
72
+ export function keyValuesOf(row, columns) {
73
+ return columns.map((column) => row[column] ?? null);
74
+ }
75
+ /**
76
+ * One independent B+Tree per index — a plain single-column one for each bare name, or a composite one for a
77
+ * spec — plan.md §22.2's "candidate indexes." Trees are keyed by the index's name (`indexNameOf`). Each is built in
78
+ * turn, starting where the previous one's own page ids left off (`heap.pages.length` for the first), so however
79
+ * many indexes a table declares, none of their page ids ever collide with the heap's or each other's.
80
+ */
81
+ export function buildIndexes(heap, indexes, maxKeys = DEFAULT_MAX_KEYS,
82
+ /** Where the first tree's pages begin — the heap's end by default; a join's *inner* table starts past the outer's. */
83
+ firstPageId = heap.pages.length, build,
84
+ /** The table's clustering key (plan.md §25.4 B3e), if it has one. */
85
+ clusteredKey) {
86
+ const trees = {};
87
+ for (const index of indexes) {
88
+ const name = indexNameOf(typeof index === 'string' ? [index] : index.columns);
89
+ const tree = buildIndex(heap, index, maxKeys, firstPageId, build, clusteredKey);
90
+ trees[name] = tree;
91
+ firstPageId = tree.nextPageId;
92
+ }
93
+ return trees;
94
+ }
95
+ /** Total pages in the database: heap pages, then every index's own pages. */
96
+ export function pageCountFor(heap, trees) {
97
+ const nextPageIds = Object.values(trees).map((t) => t.nextPageId);
98
+ return nextPageIds.length > 0 ? Math.max(...nextPageIds) : heap.pages.length;
99
+ }
@@ -0,0 +1,107 @@
1
+ /**
2
+ * Bulk-loading a B+Tree (plan.md §25.4 B3d) — the other way to build an index.
3
+ *
4
+ * `insert` builds one entry at a time, splitting a node whenever it overflows. Fed keys in ascending order that
5
+ * leaves every left leaf a fixed fraction full — three keys of four at the default fan-out: a full leaf splits, and
6
+ * nothing is ever inserted into its left half again.
7
+ * A **bulk load** knows all the entries up front. It sorts them, packs them into leaves left to right up to a
8
+ * **fill factor** — 100% leaves no room to grow, 70% leaves 30% free so the next inserts do not split at once — and
9
+ * builds each level above from the one below, bottom-up, with no splits at all. It is how a database builds an index
10
+ * over data that already exists (`CREATE INDEX`), and why that is much faster and yields a smaller, shallower tree
11
+ * than inserting the rows one by one.
12
+ *
13
+ * The result is an ordinary tree — the same node shapes, the same invariants (`validate`), the same `insert` and
14
+ * `deleteKey` afterwards. Only its *packing* differs, and packing is what decides the height and how many pages a
15
+ * lookup reads.
16
+ */
17
+ import { compareKeys, createTree, DEFAULT_MAX_KEYS } from "./btree.js";
18
+ /** How full each node is packed: a fraction of its capacity. Below one half a node would break the tree's minimum occupancy. */
19
+ export const MIN_FILL_FACTOR = 0.5;
20
+ export const DEFAULT_FILL_FACTOR = 1;
21
+ /** Splits `count` items into groups of about `size`, each at least `min` and at most `max` — the last group is the only one that can fall short, and is topped up from (or merged into) the one before. */
22
+ function groupSizes(count, size, min, max) {
23
+ const sizes = [];
24
+ let left = count;
25
+ while (left > 0) {
26
+ const take = Math.min(size, left);
27
+ sizes.push(take);
28
+ left -= take;
29
+ }
30
+ const last = sizes.length - 1;
31
+ if (last > 0 && sizes[last] < min) {
32
+ const together = sizes[last - 1] + sizes[last];
33
+ if (together <= max) {
34
+ sizes.splice(last - 1, 2, together);
35
+ }
36
+ else {
37
+ // Too many for one node, so share them: each gets at least `min` because together > max ≥ 2·min − 1.
38
+ sizes[last - 1] = Math.ceil(together / 2);
39
+ sizes[last] = together - sizes[last - 1];
40
+ }
41
+ }
42
+ return sizes;
43
+ }
44
+ /**
45
+ * Builds a tree over `entries` — sorted here, so callers need not — packing every node to `fillFactor`.
46
+ * `tiebreak` says the keys are entry keys (a table's index) rather than plain values.
47
+ */
48
+ export function bulkLoad(entries, maxKeys = DEFAULT_MAX_KEYS, firstPageId = 0, fillFactor = DEFAULT_FILL_FACTOR, tiebreak = true) {
49
+ const tree = createTree(maxKeys, firstPageId, tiebreak);
50
+ const sorted = [...entries].sort((a, b) => compareKeys(a.key, b.key));
51
+ const fill = Math.min(1, Math.max(MIN_FILL_FACTOR, fillFactor));
52
+ // Everything fits in one leaf: the root, as `createTree` already made it.
53
+ if (sorted.length <= maxKeys) {
54
+ const root = tree.nodes[tree.rootId];
55
+ root.keys = sorted.map((e) => e.key);
56
+ root.pointers = sorted.map((e) => e.pointer);
57
+ return tree;
58
+ }
59
+ const nodes = {};
60
+ let nextId = 0;
61
+ let nextPage = firstPageId;
62
+ const idFor = () => ({ id: `n${String(nextId++)}`, pageId: nextPage++ });
63
+ // Leaves: as full as the fill factor allows, but never below the half-full floor.
64
+ const minLeaf = Math.ceil(maxKeys / 2);
65
+ const leafSize = Math.max(minLeaf, Math.floor(maxKeys * fill));
66
+ let level = [];
67
+ let cursor = 0;
68
+ let previous = null;
69
+ for (const size of groupSizes(sorted.length, leafSize, minLeaf, maxKeys)) {
70
+ const chunk = sorted.slice(cursor, cursor + size);
71
+ cursor += size;
72
+ const leaf = {
73
+ kind: 'leaf',
74
+ ...idFor(),
75
+ keys: chunk.map((e) => e.key),
76
+ pointers: chunk.map((e) => e.pointer),
77
+ next: null,
78
+ };
79
+ if (previous)
80
+ previous.next = leaf.id;
81
+ previous = leaf;
82
+ nodes[leaf.id] = leaf;
83
+ level.push({ id: leaf.id, min: leaf.keys[0] });
84
+ }
85
+ // Each level above groups the one below into nodes of up to `maxKeys + 1` children, packed to the same fill factor.
86
+ const minChildren = Math.ceil((maxKeys + 1) / 2);
87
+ const childSize = Math.max(minChildren, Math.floor((maxKeys + 1) * fill));
88
+ while (level.length > 1) {
89
+ const above = [];
90
+ let at = 0;
91
+ for (const size of groupSizes(level.length, childSize, minChildren, maxKeys + 1)) {
92
+ const group = level.slice(at, at + size);
93
+ at += size;
94
+ const internal = {
95
+ kind: 'internal',
96
+ ...idFor(),
97
+ // A separator is the lowest key of the subtree to its right.
98
+ keys: group.slice(1).map((child) => child.min),
99
+ children: group.map((child) => child.id),
100
+ };
101
+ nodes[internal.id] = internal;
102
+ above.push({ id: internal.id, min: group[0].min });
103
+ }
104
+ level = above;
105
+ }
106
+ return { ...tree, rootId: level[0].id, nodes, nextNodeId: nextId, nextPageId: nextPage };
107
+ }