querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,38 @@
1
+ import { node, valuesOf } from "./btree.js";
2
+ function formatValue(value) {
3
+ if (value === null)
4
+ return 'NULL';
5
+ return typeof value === 'string' ? `'${value}'` : String(value);
6
+ }
7
+ /** A key as the tree draws it: a table index's entry shows its value(s) — the row address it also carries is not drawn. */
8
+ function formatKey(key) {
9
+ const values = valuesOf(key);
10
+ return values.length === 1 ? formatValue(values[0]) : `(${values.map(formatValue).join(', ')})`;
11
+ }
12
+ /** An index key as the write narration says it: `2`, `'ada'`, or `(2, 'ada')` for a composite index's. */
13
+ export function keyText(values) {
14
+ return values.length === 1 ? JSON.stringify(values[0]) : `(${values.map((v) => JSON.stringify(v)).join(', ')})`;
15
+ }
16
+ export function keyList(keys) {
17
+ return keys.length === 0 ? '·' : keys.map(formatKey).join(' ');
18
+ }
19
+ /** The whole tree, as the UI draws it. */
20
+ export function toIndexDisplay(tree, id = tree.rootId) {
21
+ const n = node(tree, id);
22
+ if (n.kind === 'leaf') {
23
+ return {
24
+ id: n.id,
25
+ kind: 'leaf',
26
+ label: keyList(n.keys),
27
+ pageId: n.pageId,
28
+ ...(n.next === null ? {} : { next: n.next }),
29
+ };
30
+ }
31
+ return {
32
+ id: n.id,
33
+ kind: 'internal',
34
+ label: keyList(n.keys),
35
+ pageId: n.pageId,
36
+ children: n.children.map((childId) => toIndexDisplay(tree, childId)),
37
+ };
38
+ }
@@ -0,0 +1,9 @@
1
+ export { allKeys, childIndexFor, compareKeys, createTree, DEFAULT_MAX_KEYS, deleteKey, descend, entryKey, entryWidth, height, highKey, insert, leafChain, leftmostPath, lowerBound, node, probe, search, valuesOf, } from "./btree.js";
2
+ export { bulkLoad, DEFAULT_FILL_FACTOR, MIN_FILL_FACTOR } from "./bulk.js";
3
+ export { buildIndex, buildIndexes, entryFor, keyValuesOf, pageCountFor } from "./build.js";
4
+ export { columnsOfIndex, findIndexSpec, indexNameOf, indexSpecsOf, isClusteringIndex, joinIndexFor } from "./spec.js";
5
+ export { keyList, keyText, toIndexDisplay } from "./display.js";
6
+ export { describeKeys, emitIndexLookup, resolveClustered } from "./lookup.js";
7
+ export { emitRangeStart, nextInRange } from "./rangeLookup.js";
8
+ export { validate } from "./validate.js";
9
+ export { describeUniqueKey, uniqueViolation } from "./unique.js";
@@ -0,0 +1,213 @@
1
+ /**
2
+ * A point lookup, narrated. Each node on the root-to-leaf path is fetched
3
+ * through the buffer pool first — index pages are pages, so a traversal costs
4
+ * real reads and shows up in the frame grid alongside the heap.
5
+ *
6
+ * On a table's index the key may repeat (plan.md §25.4 B3): the lookup descends
7
+ * toward the *first* entry with that key, then reads along the leaf for as
8
+ * long as the key holds, hopping to the next leaf only when the run reaches the
9
+ * end of one *and* the leaf's high key says the key continues. A unique key is
10
+ * the same walk that happens to stop after one entry.
11
+ *
12
+ * On a composite index the lookup key may be a *leftmost prefix* — one value of
13
+ * an index over two, say. An entry matches when its first columns equal the
14
+ * prefix, so the run it reads is every entry that starts that way.
15
+ */
16
+ import { childIndexFor, compareKeys, descend, entryWidth, height, highKey, lowerBound, node, probe, valuesOf, } from "./btree.js";
17
+ import { keyList, toIndexDisplay } from "./display.js";
18
+ import { traversalPrompt } from "../predict.js";
19
+ function describe(key) {
20
+ return typeof key === 'string' ? `'${key}'` : String(key);
21
+ }
22
+ /** A lookup key as text: `3`, or `(3, 'ada')` for a composite one. */
23
+ export function describeKeys(keys) {
24
+ return keys.length === 1 ? describe(keys[0]) : `(${keys.map(describe).join(', ')})`;
25
+ }
26
+ /** Does `key`'s leading values equal `keys`? — an entry (or a separator) that starts with the lookup prefix. */
27
+ function startsWith(key, keys) {
28
+ const values = valuesOf(key);
29
+ return keys.every((k, i) => i < values.length && compareKeys(values[i], k) === 0);
30
+ }
31
+ export function emitIndexLookup(emit, tree,
32
+ /** The key to find: one value for a plain index, or the leading values (a leftmost prefix) of a composite one. */
33
+ keys, pool,
34
+ /** The index's name — its column, or its columns joined — for the narration. */
35
+ column, locks,
36
+ /**
37
+ * Every existing caller wants the row, so this defaults to `true`. An
38
+ * index-only scan (plan.md §22.2) passes `false`: the leaf already holds
39
+ * everything the query needs, so fetching and locking the heap page the
40
+ * pointer names would be pure theatre — real disk I/O the query never
41
+ * actually required, which is exactly the discount §12 rules out claiming.
42
+ *
43
+ * With `true`, only the *first* match's heap page is fetched here; the caller
44
+ * fetches each later one as it pulls that row, so a `LIMIT` that stops early
45
+ * never pays for the rest.
46
+ */
47
+ touchHeap = true,
48
+ /** Replaces the "Found …" sentence when the lookup is not answering a query — a `UNIQUE` check's own words. */
49
+ foundNote) {
50
+ const label = describeKeys(keys);
51
+ const wide = tree.tiebreak === true && entryWidth(tree) > 1;
52
+ emit(`Open the B+Tree on \`${column}\`: ${height(tree)} levels, so any key is at most ${height(tree)} node reads away — however many rows the table holds.`, { stage: 'index', action: 'open', tree: toIndexDisplay(tree), nodeId: tree.rootId, height: height(tree) });
53
+ // A table's index orders entries by (values, row address); the lowest entry that could carry this key is what
54
+ // the descent aims for. A sandbox tree just holds the value.
55
+ const target = tree.tiebreak ? probe(keys, 'low') : keys[0];
56
+ const path = descend(tree, target);
57
+ const pagesRead = [];
58
+ let parentPage = null;
59
+ // Ask once, at the first branch. Asking at every level is repetition, not
60
+ // practice — the rule is the same each time.
61
+ let traversalAsked = false;
62
+ for (const nodeId of path) {
63
+ const n = node(tree, nodeId);
64
+ /**
65
+ * Latch crabbing: take the child's lock *before* releasing the parent's,
66
+ * so the subtree cannot be restructured out from under the descent. For a
67
+ * moment two locks are held at once — that overlap is the whole technique.
68
+ */
69
+ locks.acquire(n.pageId, 'shared', parentPage === null
70
+ ? `Lock the root page ${String(n.pageId)} before reading it.`
71
+ : `Lock child page ${String(n.pageId)} before letting go of page ${String(parentPage)} — "crabbing" down the tree, so the branch cannot move while we are following it.`);
72
+ const fetched = pool.fetch(n.pageId);
73
+ if (fetched) {
74
+ pagesRead.push(n.pageId);
75
+ pool.unpin(fetched.frameId);
76
+ }
77
+ if (parentPage !== null) {
78
+ locks.release(parentPage, 'shared', `Now that page ${String(n.pageId)} is locked, page ${String(parentPage)} can be released. The descent never holds the whole path at once.`);
79
+ }
80
+ parentPage = n.pageId;
81
+ if (n.kind === 'internal') {
82
+ const internal = n;
83
+ const childIndex = childIndexFor(internal, target);
84
+ // Equal values can sit on both sides of a separator, so a separator that *starts with* this key sends the search left of it.
85
+ const straddles = tree.tiebreak === true && childIndex < internal.keys.length && startsWith(internal.keys[childIndex], keys);
86
+ // A plain index's separators are single values, which read as "below / between / at or above"; a composite
87
+ // index's are tuples, and the sentence says where the search lands without pretending they are numbers.
88
+ const where = wide
89
+ ? `The first entry starting with ${label} can be no further left than child ${String(childIndex)}, so follow child ${String(childIndex)}.`
90
+ : (() => {
91
+ const separators = internal.keys.map((k) => valuesOf(k)[0]);
92
+ return `${label} ${childIndex === 0
93
+ ? `is below ${describe(separators[0])}`
94
+ : childIndex === separators.length
95
+ ? `is at or above ${describe(separators[separators.length - 1])}`
96
+ : `falls between ${describe(separators[childIndex - 1])} and ${describe(separators[childIndex])}`}, so follow child ${String(childIndex)}.`;
97
+ })();
98
+ emit(`Node ${nodeId} separates on ${keyList(internal.keys)}. ${straddles
99
+ ? `${label} matches a separator, and a run of equal keys can straddle one — so go left of it, to child ${String(childIndex)}, to reach the first ${label} there is.`
100
+ : where}`, { stage: 'index', action: 'traverse', nodeId, key: keys[0], childIndex }, traversalAsked || straddles || wide
101
+ ? undefined
102
+ : (() => {
103
+ traversalAsked = true;
104
+ const prompt = traversalPrompt(nodeId, internal.keys.map((k) => valuesOf(k)[0]), keys[0], childIndex);
105
+ return prompt ? { predictable: prompt } : undefined;
106
+ })());
107
+ }
108
+ else {
109
+ emit(`Reach leaf ${nodeId}, which holds keys ${keyList(n.keys)}.`, {
110
+ stage: 'index',
111
+ action: 'traverse',
112
+ nodeId,
113
+ key: keys[0],
114
+ });
115
+ }
116
+ }
117
+ let leaf = node(tree, path[path.length - 1]);
118
+ const pointers = [];
119
+ const entries = [];
120
+ // Only a clustered table's secondary index ever fills this in (plan.md §25.4 B3e) — parallel to `pointers`.
121
+ const clusterKeys = [];
122
+ let foundLeaf = leaf.id;
123
+ if (tree.tiebreak) {
124
+ // Read along the leaves while the key holds. A hop is taken only when the run reached the end of a leaf and the
125
+ // leaf's high key (its parent's separator) still starts with this key — otherwise the next leaf starts above it.
126
+ let at = lowerBound(leaf.keys, target);
127
+ // A UNIQUE index holds at most one entry per whole key, so the first match is the only one — no run to read, and
128
+ // no reason to look at the next leaf even when the match is the last entry of this one.
129
+ const single = tree.unique === true && keys.length >= entryWidth(tree);
130
+ for (;;) {
131
+ while (at < leaf.keys.length && startsWith(leaf.keys[at], keys) && !(single && pointers.length > 0)) {
132
+ if (pointers.length === 0)
133
+ foundLeaf = leaf.id;
134
+ pointers.push(leaf.pointers[at]);
135
+ entries.push([...valuesOf(leaf.keys[at])]);
136
+ if (tree.clusteredVia !== undefined)
137
+ clusterKeys.push(leaf.keys[at].clusterKey ?? []);
138
+ at += 1;
139
+ }
140
+ if (single && pointers.length > 0)
141
+ break;
142
+ if (at < leaf.keys.length || leaf.next === null)
143
+ break;
144
+ const high = highKey(tree, leaf);
145
+ if (high === null || !startsWith(high, keys))
146
+ break;
147
+ const next = node(tree, leaf.next);
148
+ emit(`Leaf ${leaf.id} ends, and its high key starts with ${label} too — more entries for it continue in the next leaf. Follow the sibling pointer to leaf ${next.id}.`, { stage: 'index', action: 'chain-next', fromNodeId: leaf.id, toNodeId: next.id });
149
+ locks.acquire(next.pageId, 'shared', `Lock the sibling leaf, page ${String(next.pageId)}, before letting go of page ${String(leaf.pageId)}.`);
150
+ const hopped = pool.fetch(next.pageId);
151
+ if (hopped) {
152
+ pagesRead.push(next.pageId);
153
+ pool.unpin(hopped.frameId);
154
+ }
155
+ locks.release(leaf.pageId, 'shared');
156
+ parentPage = next.pageId;
157
+ leaf = next;
158
+ at = 0;
159
+ }
160
+ }
161
+ else {
162
+ const at = leaf.keys.findIndex((k) => compareKeys(k, keys[0]) === 0);
163
+ if (at !== -1) {
164
+ pointers.push(leaf.pointers[at]);
165
+ entries.push([keys[0]]);
166
+ }
167
+ }
168
+ const clustered = tree.clusteredVia !== undefined;
169
+ const pointer = pointers[0];
170
+ if (pointer === undefined) {
171
+ emit(`${label} is not in this leaf, so it is not in the table at all — the search stops here.`, {
172
+ stage: 'index',
173
+ action: 'not-found',
174
+ nodeId: leaf.id,
175
+ key: keys[0],
176
+ });
177
+ if (parentPage !== null)
178
+ locks.release(parentPage, 'shared');
179
+ return { pointer: null, pointers, entries, pagesRead, ...(clustered ? { clusterKeys } : {}) };
180
+ }
181
+ const also = pointers.length > 1
182
+ ? ` ${String(pointers.length - 1)} more ${pointers.length === 2 ? 'entry' : 'entries'} for the same ${wide ? 'key prefix' : 'value'} follow it — a non-unique index keeps one entry per row, ordered by where the row lives, so a repeated ${wide ? 'key' : 'value'} is ${String(pointers.length)} pointers, not one.`
183
+ : '';
184
+ const partial = wide && keys.length < entryWidth(tree) ? ` (a leftmost prefix of the ${String(entryWidth(tree))}-column key)` : '';
185
+ emit(foundNote ??
186
+ (clustered && touchHeap
187
+ ? `Found ${label}${partial}. This table is clustered on \`${tree.clusteredVia}\`, so the leaf holds that key, not the row's address — finding the row costs a second descent, through the clustered index itself.${also}`
188
+ : touchHeap
189
+ ? `Found ${label}${partial}. The leaf stores a pointer to the row, not the row itself: heap page ${pointer.pageId}, slot ${pointer.slot}.${also}`
190
+ : `Found ${label}${partial}. This query only needs \`${column}\` itself, which the leaf already holds — the heap page the pointer names is never read.${also}`), { stage: 'index', action: 'found', nodeId: foundLeaf, key: keys[0] });
191
+ // A clustered table's secondary entry has no address to fetch — `resolveClustered` does the real work, as a
192
+ // second, honest descent, and only the caller knows which of `pointers` it still needs (a `LIMIT` may not need them all).
193
+ if (touchHeap && !clustered) {
194
+ locks.acquire(pointer.pageId, 'shared', `Lock heap page ${String(pointer.pageId)}, where the row itself lives.`);
195
+ const heapFetch = pool.fetch(pointer.pageId);
196
+ if (heapFetch) {
197
+ pagesRead.push(pointer.pageId);
198
+ pool.unpin(heapFetch.frameId);
199
+ }
200
+ locks.release(pointer.pageId, 'shared');
201
+ }
202
+ if (parentPage !== null)
203
+ locks.release(parentPage, 'shared');
204
+ return { pointer, pointers, entries, pagesRead, ...(clustered ? { clusterKeys } : {}) };
205
+ }
206
+ /**
207
+ * The real row address for one of a clustered table's secondary-index matches (plan.md §25.4 B3e): a second, real
208
+ * descent through the clustered index itself, narrated and costed exactly like any other lookup — because that is
209
+ * exactly what it is. `null` only if the row genuinely is not there, which a correctly maintained index never shows.
210
+ */
211
+ export function resolveClustered(emit, clusteredTree, clusteringColumn, clusterKey, pool, locks) {
212
+ return emitIndexLookup(emit, clusteredTree, clusterKey, pool, clusteringColumn, locks).pointer;
213
+ }
@@ -0,0 +1,158 @@
1
+ /**
2
+ * A range scan, narrated. `emitRangeStart` descends the B+Tree once, the same
3
+ * crabbing-locks pattern `lookup.ts`'s point lookup uses, to find the leaf a
4
+ * range's lower bound would start at (or the leftmost leaf, with no lower
5
+ * bound) — the vertical half of plan.md §23.1's leaf-chaining range scan.
6
+ * `nextInRange` is the horizontal half: it walks forward from a position
7
+ * inside that leaf, across `next` sibling pointers as leaves run out, one
8
+ * matching key at a time — so the executor can pull rows through the buffer
9
+ * pool exactly as lazily as every other operator does, rather than
10
+ * materialising the whole range up front.
11
+ */
12
+ import { childIndexFor, compareKeys, descend, height, isEntry, leftmostPath, lowerBound, node, probe, } from "./btree.js";
13
+ import { keyList, toIndexDisplay } from "./display.js";
14
+ import { describeKeys } from "./lookup.js";
15
+ /**
16
+ * Is `key` still inside the range? An entry carries its row address as well as its values, so it is never *equal* to a
17
+ * bare bound — the bound becomes a probe instead: just after every entry with this value (an inclusive `<=`) or just
18
+ * before them all (`<`). Within a composite index's equality `prefix` the probe carries the prefix too, and with no
19
+ * upper bound the end of the range is the end of the prefix's run — the first entry that no longer starts with it.
20
+ */
21
+ function withinHigh(key, high, prefix) {
22
+ if (isEntry(key)) {
23
+ const upper = high ? probe([...prefix, high.value], high.inclusive ? 'high' : 'low') : probe(prefix, 'high');
24
+ return compareKeys(key, upper) < 0;
25
+ }
26
+ if (!high)
27
+ return true;
28
+ const c = compareKeys(key, high.value);
29
+ return high.inclusive ? c <= 0 : c < 0;
30
+ }
31
+ /**
32
+ * Where the walk begins. A plain index descends toward the lower bound (an inclusive `>= v` starts at the first entry
33
+ * *for* v, an exclusive `> v` just past the last). Inside a composite index's equality `prefix` the target carries
34
+ * the prefix; with no lower bound it is the first entry *after the NULLs* of the range column — NULL sorts first,
35
+ * and a range never matches one, so there is no reason to read them.
36
+ */
37
+ function startTarget(tree, low, prefix) {
38
+ if (!tree.tiebreak)
39
+ return low?.value;
40
+ if (low)
41
+ return probe([...prefix, low.value], low.inclusive ? 'low' : 'high');
42
+ return prefix.length > 0 ? probe([...prefix, null], 'high') : undefined;
43
+ }
44
+ /**
45
+ * Descends to the range's starting leaf and narrates it. Unlike a point
46
+ * lookup there is no found/not-found here — the boundary itself is the
47
+ * answer, not a specific key, so `startIndex` may land one past the leaf's
48
+ * last key (an empty range) without that being an error.
49
+ */
50
+ export function emitRangeStart(emit, tree, column, low, pool, locks,
51
+ /** A composite index's equality values for the columns before the range one — the range runs *within* them. */
52
+ prefix = []) {
53
+ const h = height(tree);
54
+ const within = prefix.length > 0 ? ` within ${describeKeys(prefix)}` : '';
55
+ emit(`Open the B+Tree on \`${column}\`: ${String(h)} level${h === 1 ? '' : 's'}. ${low
56
+ ? `Descend once toward ${describeKeys([...prefix, low.value])} — not to find that exact key, but to find where the range begins${within}.`
57
+ : prefix.length > 0
58
+ ? `This range has no lower bound${within}, so descend toward where the entries starting ${describeKeys(prefix)} begin.`
59
+ : "This range has no lower bound, so descend to the tree's leftmost leaf instead."}`, { stage: 'index', action: 'open', tree: toIndexDisplay(tree), nodeId: tree.rootId, height: h });
60
+ const target = startTarget(tree, low, prefix);
61
+ const path = target !== undefined ? descend(tree, target) : leftmostPath(tree);
62
+ let parentPage = null;
63
+ for (const nodeId of path) {
64
+ const n = node(tree, nodeId);
65
+ locks.acquire(n.pageId, 'shared', parentPage === null
66
+ ? `Lock the root page ${String(n.pageId)} before reading it.`
67
+ : `Lock child page ${String(n.pageId)} before letting go of page ${String(parentPage)} — "crabbing" down the tree, so the branch cannot move while we are following it.`);
68
+ const fetched = pool.fetch(n.pageId);
69
+ if (fetched)
70
+ pool.unpin(fetched.frameId);
71
+ if (parentPage !== null) {
72
+ locks.release(parentPage, 'shared', `Now that page ${String(n.pageId)} is locked, page ${String(parentPage)} can be released.`);
73
+ }
74
+ parentPage = n.pageId;
75
+ if (n.kind === 'internal') {
76
+ const childIndex = target !== undefined ? childIndexFor(n, target) : 0;
77
+ emit(`Node ${nodeId} separates on ${keyList(n.keys)}. ${low
78
+ ? `${describeKeys([...prefix, low.value])} routes to child ${String(childIndex)}.`
79
+ : prefix.length > 0
80
+ ? `The entries starting ${describeKeys(prefix)} route to child ${String(childIndex)}.`
81
+ : 'With no lower bound, follow the leftmost child.'}`, { stage: 'index', action: 'traverse', nodeId, ...(low ? { key: low.value } : {}), childIndex });
82
+ }
83
+ else {
84
+ emit(`Reach leaf ${nodeId}, which holds keys ${keyList(n.keys)} — the scan starts here.`, {
85
+ stage: 'index',
86
+ action: 'traverse',
87
+ nodeId,
88
+ ...(low ? { key: low.value } : {}),
89
+ });
90
+ }
91
+ }
92
+ if (parentPage !== null)
93
+ locks.release(parentPage, 'shared');
94
+ const leafId = path[path.length - 1];
95
+ const leaf = node(tree, leafId);
96
+ const at = target !== undefined ? lowerBound(leaf.keys, target) : 0;
97
+ // `lowerBound` finds the first key >= the target — exactly right for an
98
+ // inclusive `>=` bound. In a sandbox tree an exclusive `>` on a key that
99
+ // happens to be present has to skip one further, past the exact match; a
100
+ // table's index already aimed at a probe just past the whole run.
101
+ const startIndex = low && !low.inclusive && !tree.tiebreak && at < leaf.keys.length && compareKeys(leaf.keys[at], low.value) === 0
102
+ ? at + 1
103
+ : at;
104
+ return { leafId, startIndex };
105
+ }
106
+ /**
107
+ * The next matching key from `(leafId, index)` onward, following sibling
108
+ * pointers as leaves run out and narrating each hop. Returns `null` once the
109
+ * chain ends or a key exceeds `high` — a sorted leaf chain means every key
110
+ * past that point is out of range too, so the walk can stop rather than
111
+ * checking the rest.
112
+ *
113
+ * Each hop *reads the sibling leaf through the buffer pool* (`io`): a leaf is a
114
+ * page, and a range that spans ten of them costs ten reads — the same figure
115
+ * the cost model prices (`extraLeaves`), so an `EXPLAIN ANALYZE` estimate and
116
+ * its actual agree on how many leaves the walk touched.
117
+ */
118
+ export function nextInRange(emit, tree, leafId, index, high, io,
119
+ /** A composite index's equality values before the range column — see `emitRangeStart`. */
120
+ prefix = []) {
121
+ let currentLeafId = leafId;
122
+ let currentIndex = index;
123
+ for (;;) {
124
+ if (currentLeafId === null)
125
+ return null;
126
+ const leaf = node(tree, currentLeafId);
127
+ if (currentIndex >= leaf.keys.length) {
128
+ if (!leaf.next)
129
+ return null;
130
+ emit(`Leaf ${currentLeafId} is exhausted; follow its sibling pointer to leaf ${leaf.next}.`, {
131
+ stage: 'index',
132
+ action: 'chain-next',
133
+ fromNodeId: currentLeafId,
134
+ toNodeId: leaf.next,
135
+ });
136
+ if (io) {
137
+ const sibling = node(tree, leaf.next);
138
+ io.locks.acquire(sibling.pageId, 'shared', `Lock the sibling leaf, page ${String(sibling.pageId)}, to read it.`);
139
+ const fetched = io.pool.fetch(sibling.pageId);
140
+ if (fetched)
141
+ io.pool.unpin(fetched.frameId);
142
+ io.locks.release(sibling.pageId, 'shared');
143
+ }
144
+ currentLeafId = leaf.next;
145
+ currentIndex = 0;
146
+ continue;
147
+ }
148
+ const key = leaf.keys[currentIndex];
149
+ if (!withinHigh(key, high, prefix))
150
+ return null;
151
+ return {
152
+ key,
153
+ pointer: leaf.pointers[currentIndex],
154
+ leafId: currentLeafId,
155
+ nextIndex: currentIndex + 1,
156
+ };
157
+ }
158
+ }
@@ -0,0 +1,47 @@
1
+ /**
2
+ * What indexes a table has, as one list (plan.md §25.4 B3).
3
+ *
4
+ * A table declares them two ways: `indexedColumns`, the shorthand for one plain single-column index per name, and
5
+ * `indexes`, for anything richer — today a composite index over several columns. Everything downstream (building
6
+ * the trees, maintaining them on a write, choosing one in the optimizer, pricing a scan of it) wants a single list of
7
+ * indexes, each with a stable name to find its tree by, so this is the one place that reads the two.
8
+ *
9
+ * An index's **name** is its columns joined with `, ` — `id` for a plain one, `lastName, firstName` for a composite
10
+ * one. Column names are identifiers, so the name is unambiguous, and it is what a `Plan` node and `ExecContext.trees`
11
+ * use to say *which* index.
12
+ */
13
+ export const INDEX_NAME_SEPARATOR = ', ';
14
+ export const indexNameOf = (columns) => columns.join(INDEX_NAME_SEPARATOR);
15
+ /** The columns an index name stands for, in key order. */
16
+ export const columnsOfIndex = (name) => name.split(INDEX_NAME_SEPARATOR);
17
+ /** Every index `table` declares: the single-column shorthand first, in declared order, then the richer ones. */
18
+ export function indexSpecsOf(table) {
19
+ const byName = new Map();
20
+ for (const column of table.indexedColumns ?? [])
21
+ byName.set(column, { columns: [column] });
22
+ for (const spec of table.indexes ?? []) {
23
+ byName.set(indexNameOf(spec.columns), { columns: [...spec.columns], ...(spec.unique ? { unique: true } : {}) });
24
+ }
25
+ return [...byName.values()];
26
+ }
27
+ export function findIndexSpec(table, name) {
28
+ return indexSpecsOf(table).find((spec) => indexNameOf(spec.columns) === name);
29
+ }
30
+ /**
31
+ * Is `columns` a table's own CLUSTERING key (plan.md §25.4 B3e)? Only that index's own tree holds real row
32
+ * addresses; every other index on a clustered table indirects through it instead, one extra descent at a time.
33
+ */
34
+ export function isClusteringIndex(clusteredKey, columns) {
35
+ return !!clusteredKey && clusteredKey.length === columns.length && clusteredKey.every((c, i) => c === columns[i]);
36
+ }
37
+ /**
38
+ * The index a `JOIN` can probe for its inner side's join column: one whose *leading* column is that column — a plain
39
+ * index on it, or a composite one that starts with it (a leftmost prefix of one column is enough to enter it).
40
+ * A plain index wins over a composite one, since its entries are narrower. A clustered table's SECONDARY index is
41
+ * never offered (plan.md §25.4 B3e): the extra descent it costs is only built for an equality lookup, not a probe
42
+ * run once per outer row — the clustering key's own index is still offered, since it needs no second descent at all.
43
+ */
44
+ export function joinIndexFor(table, column) {
45
+ const specs = indexSpecsOf(table).filter((spec) => spec.columns[0] === column && (!table.clusteredKey || isClusteringIndex(table.clusteredKey, spec.columns)));
46
+ return specs.find((spec) => spec.columns.length === 1) ?? specs[0];
47
+ }
@@ -0,0 +1,31 @@
1
+ /**
2
+ * `UNIQUE` (plan.md §25.4 B3c), as a fact about a table's rows rather than about a tree.
3
+ *
4
+ * An index is unique when no two rows hold the same key — and, as SQL says, a key with a NULL in it never counts:
5
+ * NULL is not equal to NULL, so any number of rows may hold one. The write path enforces it through the index
6
+ * (`exec/unique.ts` looks the new key up first); this is the same rule applied to a whole table at once, for
7
+ * whoever has to decide whether a set of rows may sit under a unique index at all (the Custom database editor).
8
+ */
9
+ import { indexNameOf, indexSpecsOf } from "./spec.js";
10
+ /** The first pair of rows that break a unique index of `table` — in `rows` (default: its own) — or `null`. */
11
+ export function uniqueViolation(table, rows) {
12
+ for (const spec of indexSpecsOf(table)) {
13
+ if (!spec.unique)
14
+ continue;
15
+ const seen = new Set();
16
+ for (const row of rows) {
17
+ const values = spec.columns.map((c) => row[c] ?? null);
18
+ if (values.some((v) => v === null))
19
+ continue; // NULL never conflicts
20
+ const key = JSON.stringify(values);
21
+ if (seen.has(key))
22
+ return { index: indexNameOf(spec.columns), columns: spec.columns, values };
23
+ seen.add(key);
24
+ }
25
+ }
26
+ return null;
27
+ }
28
+ /** `(email) = ('a@x')` — a key the way PostgreSQL's constraint errors write it. */
29
+ export function describeUniqueKey(columns, values) {
30
+ return `(${columns.join(', ')}) = (${values.map((v) => (typeof v === 'string' ? `'${v}'` : String(v))).join(', ')})`;
31
+ }
@@ -0,0 +1,105 @@
1
+ /**
2
+ * Structural invariants of a B+Tree. plan.md §7.6 requires these to be tested
3
+ * before anything is drawn: a tree with the split rules subtly wrong still
4
+ * *looks* plausible, and silently returns wrong rows.
5
+ *
6
+ * Returns a list of violations — empty means the tree is sound.
7
+ */
8
+ import { compareKeys, leafChain, node } from "./btree.js";
9
+ export function validate(tree) {
10
+ const problems = [];
11
+ const root = node(tree, tree.rootId);
12
+ const minLeafKeys = Math.ceil(tree.maxKeys / 2);
13
+ const minInternalChildren = Math.ceil((tree.maxKeys + 1) / 2);
14
+ const depths = [];
15
+ const visited = new Set();
16
+ function walk(id, depth, low, high) {
17
+ if (visited.has(id)) {
18
+ problems.push(`node ${id} is reachable by more than one path`);
19
+ return;
20
+ }
21
+ visited.add(id);
22
+ const n = node(tree, id);
23
+ const isRoot = id === tree.rootId;
24
+ // Keys within a node are strictly ascending.
25
+ for (let i = 1; i < n.keys.length; i++) {
26
+ if (compareKeys(n.keys[i - 1], n.keys[i]) >= 0) {
27
+ problems.push(`node ${id} keys are not strictly ascending`);
28
+ break;
29
+ }
30
+ }
31
+ // Every key must fall inside the range its ancestors promised.
32
+ for (const key of n.keys) {
33
+ if (low !== null && compareKeys(key, low) < 0) {
34
+ problems.push(`node ${id} holds key below its separator bound`);
35
+ break;
36
+ }
37
+ if (high !== null && compareKeys(key, high) >= 0) {
38
+ problems.push(`node ${id} holds key at or above its separator bound`);
39
+ break;
40
+ }
41
+ }
42
+ if (n.kind === 'leaf') {
43
+ depths.push(depth);
44
+ if (n.keys.length !== n.pointers.length) {
45
+ problems.push(`leaf ${id} has ${n.keys.length} keys but ${n.pointers.length} pointers`);
46
+ }
47
+ // (d) every node except the root is at least half full
48
+ if (!isRoot && n.keys.length < minLeafKeys) {
49
+ problems.push(`leaf ${id} is under half full (${n.keys.length} < ${minLeafKeys})`);
50
+ }
51
+ if (n.keys.length > tree.maxKeys) {
52
+ problems.push(`leaf ${id} overflows (${n.keys.length} > ${tree.maxKeys})`);
53
+ }
54
+ return;
55
+ }
56
+ if (n.children.length !== n.keys.length + 1) {
57
+ problems.push(`internal ${id} has ${n.keys.length} keys but ${n.children.length} children`);
58
+ return;
59
+ }
60
+ if (n.keys.length > tree.maxKeys) {
61
+ problems.push(`internal ${id} overflows (${n.keys.length} > ${tree.maxKeys})`);
62
+ }
63
+ if (!isRoot && n.children.length < minInternalChildren) {
64
+ problems.push(`internal ${id} is under half full (${n.children.length} < ${minInternalChildren})`);
65
+ }
66
+ if (isRoot && n.children.length < 2) {
67
+ problems.push(`root ${id} has fewer than two children`);
68
+ }
69
+ n.children.forEach((childId, i) => {
70
+ walk(childId, depth + 1, i === 0 ? low : n.keys[i - 1], i === n.keys.length ? high : n.keys[i]);
71
+ });
72
+ }
73
+ walk(root.id, 0, null, null);
74
+ // (c) all leaves at equal depth
75
+ if (new Set(depths).size > 1) {
76
+ problems.push(`leaves sit at differing depths: ${[...new Set(depths)].sort().join(', ')}`);
77
+ }
78
+ // (b) the sibling chain is sorted and reaches every leaf
79
+ const chain = leafChain(tree);
80
+ const chainKeys = chain.flatMap((id) => node(tree, id).keys);
81
+ for (let i = 1; i < chainKeys.length; i++) {
82
+ if (compareKeys(chainKeys[i - 1], chainKeys[i]) >= 0) {
83
+ problems.push('the leaf sibling chain is not in ascending key order');
84
+ break;
85
+ }
86
+ }
87
+ const leavesInTree = [...visited].filter((id) => node(tree, id).kind === 'leaf');
88
+ if (chain.length !== leavesInTree.length) {
89
+ problems.push(`the sibling chain reaches ${chain.length} leaves but the tree has ${leavesInTree.length}`);
90
+ }
91
+ // Every separator in an internal node must actually exist as a leaf key
92
+ // (the copy-up rule) — a pushed-up leaf separator would break lookups.
93
+ const leafKeySet = new Set(chainKeys.map((k) => JSON.stringify(k)));
94
+ for (const id of visited) {
95
+ const n = node(tree, id);
96
+ if (n.kind !== 'internal')
97
+ continue;
98
+ for (const key of n.keys) {
99
+ if (!leafKeySet.has(JSON.stringify(key))) {
100
+ problems.push(`separator ${JSON.stringify(key)} in ${id} is not present in any leaf`);
101
+ }
102
+ }
103
+ }
104
+ return problems;
105
+ }
@@ -0,0 +1,16 @@
1
+ export { runQuery } from "./runQuery.js";
2
+ export { MAX_REFERENCES, parseReferenceString, randomReferenceString, runBufferTrace, } from "./bufferTrace.js";
3
+ export { createJoinCompareDatabase, createJoinDemoDatabase, createJoinOrderDemoDatabase, createSeedDatabase, } from "./seed.js";
4
+ export { DATASETS, DATASET_IDS, DEFAULT_DATASET, isDatasetId, } from "./datasets.js";
5
+ export { createTracer } from "./trace.js";
6
+ export { parse, toDisplayTree, columnsIn, tokenize } from "./parser/index.js";
7
+ export { buildPlan, estimatePlan, exprToSql, optimize, OPTIMIZER_RULES, rootEstimate, selectivityOf, toPlanDisplay, } from "./planner/index.js";
8
+ export { costOfOrder, isReorderableChain, joinGraphFromAst, rebuildJoinsForOrder, reorderJoinsIfCheaper, runJoinOrderDP, writtenOrderFromAst, } from "./planner/joinOrder.js";
9
+ export { buildHeap, createBufferPool, policyFor, REPLACEMENT_POLICY_LABELS, REPLACEMENT_POLICY_NAMES, USAGE_COUNT_CAP, } from "./storage/index.js";
10
+ export { allKeys, buildIndex, createTree, DEFAULT_MAX_KEYS, deleteKey, descend, height, columnsOfIndex, indexNameOf, indexSpecsOf, insert, keyList, search, toIndexDisplay, uniqueViolation, validate, } from "./index/index.js";
11
+ export { evaluatePredicate, externalMergeSort, passes, sortPassCount, sortTempPageCount } from "./exec/index.js";
12
+ export { createLockManager } from "./locks/index.js";
13
+ export { applyEvent, createTraceProjector, initialViewState, stateAt, } from "./viewState.js";
14
+ export { DEFAULT_ENGINE_OPTIONS, IMPLEMENTED_STAGES, STAGES } from "./types.js";
15
+ export { analyzeTable, statsToRows } from "./stats.js";
16
+ export { finalPlan, nodeKey } from "./explain.js";
@@ -0,0 +1 @@
1
+ export { createLockManager } from "./lockManager.js";