querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { resultPrompt } from "../predict.js";
|
|
2
|
+
import { rootEstimate } from "../planner/cost.js";
|
|
3
|
+
import { buildOperator } from "./operators.js";
|
|
4
|
+
/** Pulls the operator tree to exhaustion and emits the result. */
|
|
5
|
+
export function execute(plan, ctx) {
|
|
6
|
+
const root = buildOperator(plan, ctx);
|
|
7
|
+
const rows = [];
|
|
8
|
+
root.open();
|
|
9
|
+
for (;;) {
|
|
10
|
+
const row = root.next();
|
|
11
|
+
if (!row)
|
|
12
|
+
break;
|
|
13
|
+
rows.push(row);
|
|
14
|
+
}
|
|
15
|
+
root.close();
|
|
16
|
+
const { hits, misses, evictions } = ctx.pool.stats;
|
|
17
|
+
// Every table this query names, including `ctx.table` itself — a chain's own cost-model recursion (plan.md
|
|
18
|
+
// §25.4 C3 slice b) needs to resolve it by name again once it reaches the bottom of the chain, not just at the
|
|
19
|
+
// top (see emit.ts's identical build for the fuller explanation).
|
|
20
|
+
const otherTables = ctx.joinPartners
|
|
21
|
+
? {
|
|
22
|
+
...Object.fromEntries(Object.entries(ctx.joinPartners).map(([name, partner]) => [name, partner.table])),
|
|
23
|
+
[ctx.table.name]: ctx.table,
|
|
24
|
+
}
|
|
25
|
+
: undefined;
|
|
26
|
+
const estimate = rootEstimate(plan, ctx.table, ctx.options.rowsPerPage, otherTables);
|
|
27
|
+
const reconcile = estimate.estRows === rows.length
|
|
28
|
+
? ` The cost model guessed ~${estimate.estRows} — on the nose (${estimate.basis}).`
|
|
29
|
+
: ` The cost model guessed ~${estimate.estRows} (${estimate.basis}); the count is exact, the estimate is a heuristic.`;
|
|
30
|
+
ctx.emit(`${rows.length} row${rows.length === 1 ? '' : 's'} returned, from ${misses} disk read${misses === 1 ? '' : 's'}, ${hits} buffer hit${hits === 1 ? '' : 's'} and ${evictions} eviction${evictions === 1 ? '' : 's'}.${reconcile}`, { stage: 'result', rows }, (() => {
|
|
31
|
+
const prompt = resultPrompt(rows.length, ctx.table.rows.length);
|
|
32
|
+
return prompt ? { predictable: prompt } : undefined;
|
|
33
|
+
})());
|
|
34
|
+
return rows;
|
|
35
|
+
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `Sort` operator's own driver: external merge sort over real `Row[]`,
|
|
3
|
+
* with every page it writes or re-reads flowing through the real buffer pool
|
|
4
|
+
* (plan.md §22.2) — competing for the same frames a heap scan uses, which is
|
|
5
|
+
* the whole point of building this rather than calling `Array.prototype.sort`
|
|
6
|
+
* with a comment.
|
|
7
|
+
*
|
|
8
|
+
* The algorithm mirrors the worked example already built for `/tools/sort`
|
|
9
|
+
* (`src/ui/externalSort.ts`, Ramakrishnan & Gehrke §13.3): `memPages` pages
|
|
10
|
+
* of records sorted in memory become one run each (pass 0), then every later
|
|
11
|
+
* pass merges `memPages − 1` runs at a time until one remains.
|
|
12
|
+
*
|
|
13
|
+
* Page ids are never reused: each pass writes into its own fresh block, so a
|
|
14
|
+
* page a later step still needs to read is never silently overwritten by an
|
|
15
|
+
* earlier one — the property a real spill file gets for free by simply being
|
|
16
|
+
* a different file. `sortTempPageCount` sizes that reservation; `runQuery.ts`
|
|
17
|
+
* adds it to the buffer pool's page count before the pool is built, so every
|
|
18
|
+
* id this module hands out is valid.
|
|
19
|
+
*/
|
|
20
|
+
/** Runs merged per merge step — one buffer page is reserved for the output. */
|
|
21
|
+
function fanInFor(memPages) {
|
|
22
|
+
return Math.max(2, memPages - 1);
|
|
23
|
+
}
|
|
24
|
+
/** Passes a sort of `pageCount` pages needs: run generation, then merges until one run remains. */
|
|
25
|
+
export function sortPassCount(pageCount, memPages) {
|
|
26
|
+
const fanIn = fanInFor(memPages);
|
|
27
|
+
let runCount = Math.max(1, Math.ceil(pageCount / memPages));
|
|
28
|
+
let passes = 1;
|
|
29
|
+
while (runCount > 1) {
|
|
30
|
+
runCount = Math.ceil(runCount / fanIn);
|
|
31
|
+
passes += 1;
|
|
32
|
+
}
|
|
33
|
+
return passes;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Distinct page ids to reserve for sorting `rowCount` rows: a fresh block per
|
|
37
|
+
* pass, sized generously above the perfectly-packed page count. Merging
|
|
38
|
+
* several runs whose own last page is only partly full needs a handful more
|
|
39
|
+
* once each merge group repacks its own output — the same last-page waste a
|
|
40
|
+
* real spill file has — so the block leaves `memPages` pages of slack rather
|
|
41
|
+
* than claim an exact figure it cannot defend (plan.md §12).
|
|
42
|
+
*/
|
|
43
|
+
export function sortTempPageCount(rowCount, rowsPerPage, memPages) {
|
|
44
|
+
const pageCount = Math.max(1, Math.ceil(rowCount / rowsPerPage));
|
|
45
|
+
const passes = sortPassCount(pageCount, memPages);
|
|
46
|
+
return (pageCount + memPages) * passes;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Total order over `SqlValue`, ascending: numbers numerically, everything
|
|
50
|
+
* else lexicographically, NULLs last — Postgres's own default, and the
|
|
51
|
+
* reason `direction: 'desc'` below just negates this rather than re-deriving
|
|
52
|
+
* a second rule (negating "NULLs last" gives "NULLs first", Postgres's
|
|
53
|
+
* default there too).
|
|
54
|
+
*/
|
|
55
|
+
function compareAsc(a, b) {
|
|
56
|
+
if (a === null && b === null)
|
|
57
|
+
return 0;
|
|
58
|
+
if (a === null)
|
|
59
|
+
return 1;
|
|
60
|
+
if (b === null)
|
|
61
|
+
return -1;
|
|
62
|
+
if (typeof a === 'number' && typeof b === 'number')
|
|
63
|
+
return a - b;
|
|
64
|
+
const sa = typeof a === 'boolean' ? String(Number(a)) : String(a);
|
|
65
|
+
const sb = typeof b === 'boolean' ? String(Number(b)) : String(b);
|
|
66
|
+
return sa < sb ? -1 : sa > sb ? 1 : 0;
|
|
67
|
+
}
|
|
68
|
+
/** Writes `rows` to a fresh page, through the pool for the frame bookkeeping and narration. */
|
|
69
|
+
function spillPage(io, spill, nextId, rows) {
|
|
70
|
+
const pageId = nextId();
|
|
71
|
+
const result = io.pool.write(pageId);
|
|
72
|
+
if (result)
|
|
73
|
+
io.pool.unpin(result.frameId);
|
|
74
|
+
spill.set(pageId, rows);
|
|
75
|
+
return pageId;
|
|
76
|
+
}
|
|
77
|
+
/** Reads a spilled page's rows back, through the pool for the same reason. */
|
|
78
|
+
function readPage(io, spill, pageId) {
|
|
79
|
+
const result = io.pool.fetch(pageId);
|
|
80
|
+
if (result)
|
|
81
|
+
io.pool.unpin(result.frameId);
|
|
82
|
+
return spill.get(pageId) ?? [];
|
|
83
|
+
}
|
|
84
|
+
/** Writes one run's already-sorted rows out, `rowsPerPage` at a time (at least one page, even if empty). */
|
|
85
|
+
function writeRun(io, spill, nextId, id, rows) {
|
|
86
|
+
if (rows.length === 0)
|
|
87
|
+
return { id, pages: [spillPage(io, spill, nextId, [])] };
|
|
88
|
+
const pages = [];
|
|
89
|
+
for (let i = 0; i < rows.length; i += io.rowsPerPage) {
|
|
90
|
+
pages.push(spillPage(io, spill, nextId, rows.slice(i, i + io.rowsPerPage)));
|
|
91
|
+
}
|
|
92
|
+
return { id, pages };
|
|
93
|
+
}
|
|
94
|
+
/** k-way merge of already-sorted runs, read back page by page through the pool. */
|
|
95
|
+
function mergeRuns(sources, spill, io, cmp) {
|
|
96
|
+
const rows = sources.map((run) => run.pages.flatMap((pid) => readPage(io, spill, pid)));
|
|
97
|
+
const heads = rows.map(() => 0);
|
|
98
|
+
const merged = [];
|
|
99
|
+
for (;;) {
|
|
100
|
+
let pick = -1;
|
|
101
|
+
for (let i = 0; i < rows.length; i++) {
|
|
102
|
+
if (heads[i] >= rows[i].length)
|
|
103
|
+
continue;
|
|
104
|
+
if (pick === -1 || cmp(rows[i][heads[i]], rows[pick][heads[pick]]) < 0)
|
|
105
|
+
pick = i;
|
|
106
|
+
}
|
|
107
|
+
if (pick === -1)
|
|
108
|
+
break;
|
|
109
|
+
merged.push(rows[pick][heads[pick]]);
|
|
110
|
+
heads[pick] += 1;
|
|
111
|
+
}
|
|
112
|
+
return merged;
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Runs a full external merge sort and returns the rows in final order. Runs
|
|
116
|
+
* to completion before returning a single row — `Sort` is the one operator
|
|
117
|
+
* in this engine that cannot stream, and the narration says so.
|
|
118
|
+
*
|
|
119
|
+
* `keyOf` returns a *tuple* — compared lexicographically, left column first —
|
|
120
|
+
* rather than a single `SqlValue`, so this same driver serves `Sort`'s
|
|
121
|
+
* one-column `ORDER BY` (a one-element tuple) and `SortAggregate`'s
|
|
122
|
+
* potentially multi-column `GROUP BY` key without duplicating the algorithm.
|
|
123
|
+
* `subject` and `op` name the caller in the narration and in the emitted
|
|
124
|
+
* `execute` events (`'Sort'` by default; `SortAggregate` passes its own name
|
|
125
|
+
* so its internal sort phase is not mistaken for a top-level `Sort` node).
|
|
126
|
+
* `table` is `sortMergeJoin`'s own disambiguator (its `Join`'s `rightTable`) passed through so these internal,
|
|
127
|
+
* zero-row progress pings carry the same `table` as that `Join`'s real `execute` events — otherwise one of these
|
|
128
|
+
* could land in `explain.ts`'s per-node lookup without the table that identifies which `Join`, in a chain with
|
|
129
|
+
* more than one, it belongs to. `undefined` for `Sort`/`SortAggregate`/`SortDistinct`, which never repeat.
|
|
130
|
+
*/
|
|
131
|
+
export function externalMergeSort(rows, keyOf, direction, io, subject = 'Sort', op = 'Sort', table) {
|
|
132
|
+
const cmp = (a, b) => {
|
|
133
|
+
const ka = keyOf(a);
|
|
134
|
+
const kb = keyOf(b);
|
|
135
|
+
for (let i = 0; i < ka.length; i++) {
|
|
136
|
+
const c = compareAsc(ka[i] ?? null, kb[i] ?? null);
|
|
137
|
+
if (c !== 0)
|
|
138
|
+
return direction === 'desc' ? -c : c;
|
|
139
|
+
}
|
|
140
|
+
return 0;
|
|
141
|
+
};
|
|
142
|
+
const spill = new Map();
|
|
143
|
+
let cursor = io.base;
|
|
144
|
+
const nextId = () => cursor++;
|
|
145
|
+
const fanIn = fanInFor(io.memPages);
|
|
146
|
+
const chunk = io.memPages * io.rowsPerPage;
|
|
147
|
+
io.emit(`${subject} has to see every row before it can produce one — the one operator here that cannot stream. ${plural(rows.length, 'row')} to sort, ${plural(io.memPages, 'page')} of memory at a time.`, { stage: 'execute', op, rowsProduced: 0, ...(table === undefined ? {} : { table }) });
|
|
148
|
+
let runs = [];
|
|
149
|
+
let runId = 0;
|
|
150
|
+
for (let i = 0; i < rows.length; i += chunk) {
|
|
151
|
+
const sorted = [...rows.slice(i, i + chunk)].sort(cmp);
|
|
152
|
+
runs.push(writeRun(io, spill, nextId, runId++, sorted));
|
|
153
|
+
}
|
|
154
|
+
if (runs.length === 0)
|
|
155
|
+
runs.push(writeRun(io, spill, nextId, runId++, []));
|
|
156
|
+
const pagesWritten = runs.reduce((n, r) => n + r.pages.length, 0);
|
|
157
|
+
io.emit(`Run generation: ${plural(runs.length, 'sorted run')} written, ${plural(pagesWritten, 'page')} total.`, { stage: 'execute', op, rowsProduced: 0, ...(table === undefined ? {} : { table }) });
|
|
158
|
+
let passIndex = 1;
|
|
159
|
+
while (runs.length > 1) {
|
|
160
|
+
const next = [];
|
|
161
|
+
for (let g = 0; g < runs.length; g += fanIn) {
|
|
162
|
+
const merged = mergeRuns(runs.slice(g, g + fanIn), spill, io, cmp);
|
|
163
|
+
next.push(writeRun(io, spill, nextId, runId++, merged));
|
|
164
|
+
}
|
|
165
|
+
io.emit(`Merge pass ${String(passIndex)}: ${plural(runs.length, 'run')} become ${plural(next.length, 'run')}.`, { stage: 'execute', op, rowsProduced: 0, ...(table === undefined ? {} : { table }) });
|
|
166
|
+
runs = next;
|
|
167
|
+
passIndex += 1;
|
|
168
|
+
}
|
|
169
|
+
return runs[0].pages.flatMap((pid) => readPage(io, spill, pid));
|
|
170
|
+
}
|
|
171
|
+
const plural = (n, noun) => `${String(n)} ${noun}${n === 1 ? '' : 's'}`;
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Enforcing `UNIQUE` (plan.md §25.4 B3c). A unique constraint is not a rule checked somewhere else — it *is* an
|
|
3
|
+
* index lookup: before a new key goes in, the index is asked whether it is already there, and if it is the statement
|
|
4
|
+
* is rejected. That is why every unique constraint is backed by an index, and why an `INSERT` into a table with one
|
|
5
|
+
* reads index pages even though it never scans anything. Each check below narrates that lookup, real page reads and
|
|
6
|
+
* all; on a violation it throws, and `runQuery` turns the throw into the statement's error — no change is made,
|
|
7
|
+
* because a write's rows only reach the table when the whole statement succeeds.
|
|
8
|
+
*/
|
|
9
|
+
import { rowAt } from "./operators.js";
|
|
10
|
+
import { describeUniqueKey, emitIndexLookup, indexNameOf, indexSpecsOf, keyValuesOf, resolveClustered } from "../index/index.js";
|
|
11
|
+
/** A statement tried to put a key into a unique index that already holds it. */
|
|
12
|
+
export class UniqueViolation extends Error {
|
|
13
|
+
index;
|
|
14
|
+
columns;
|
|
15
|
+
values;
|
|
16
|
+
holder;
|
|
17
|
+
constructor(index, columns, values, holder) {
|
|
18
|
+
super(`duplicate key value violates unique index ${index}`);
|
|
19
|
+
this.index = index;
|
|
20
|
+
this.columns = columns;
|
|
21
|
+
this.values = values;
|
|
22
|
+
this.holder = holder;
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
/** Asks the unique index whether `values` is already there; throws `UniqueViolation` if it is. */
|
|
26
|
+
function checkKey(ctx, spec, values, statement) {
|
|
27
|
+
const name = indexNameOf(spec.columns);
|
|
28
|
+
const tree = ctx.trees[name];
|
|
29
|
+
if (!tree)
|
|
30
|
+
return;
|
|
31
|
+
// A NULL in the key is never equal to another NULL, so it can conflict with nothing: there is nothing to look up.
|
|
32
|
+
if (values.some((v) => v === null)) {
|
|
33
|
+
ctx.emit(`The UNIQUE index on \`${name}\` has a NULL in this key, and NULL is not equal to NULL, so it cannot conflict with any entry — no lookup is needed.`, {
|
|
34
|
+
stage: 'execute',
|
|
35
|
+
op: statement === 'INSERT' ? 'Insert' : 'Update',
|
|
36
|
+
rowsProduced: 0,
|
|
37
|
+
});
|
|
38
|
+
return;
|
|
39
|
+
}
|
|
40
|
+
const key = describeUniqueKey(spec.columns, values);
|
|
41
|
+
ctx.emit(`${statement} checks the UNIQUE index on \`${name}\` before it changes anything: is there already an entry for ${key}?`, {
|
|
42
|
+
stage: 'execute',
|
|
43
|
+
op: statement === 'INSERT' ? 'Insert' : 'Update',
|
|
44
|
+
rowsProduced: 0,
|
|
45
|
+
});
|
|
46
|
+
const outcome = emitIndexLookup(ctx.emit, tree, values, ctx.pool, name, ctx.locks, false, `Found ${key} — the key is already taken, so ${statement} will be rejected.`);
|
|
47
|
+
if (outcome.pointer) {
|
|
48
|
+
// On a clustered table's secondary UNIQUE index, `pointer` is a sentinel — the row that has the key needs the
|
|
49
|
+
// same second descent any other match through this index would (plan.md §25.4 B3e), so the error can still say
|
|
50
|
+
// which row holds it, not just that one does.
|
|
51
|
+
const clusterKey = outcome.clusterKeys?.[0];
|
|
52
|
+
const clusteredTree = tree.clusteredVia !== undefined ? ctx.trees[tree.clusteredVia] : undefined;
|
|
53
|
+
const holder = clusterKey && clusteredTree
|
|
54
|
+
? (resolveClustered(ctx.emit, clusteredTree, tree.clusteredVia, clusterKey, ctx.pool, ctx.locks) ?? outcome.pointer)
|
|
55
|
+
: outcome.pointer;
|
|
56
|
+
throw new UniqueViolation(name, spec.columns, values, holder);
|
|
57
|
+
}
|
|
58
|
+
ctx.emit(`No entry for ${key}, so the key is free and ${statement} goes on.`, {
|
|
59
|
+
stage: 'execute',
|
|
60
|
+
op: statement === 'INSERT' ? 'Insert' : 'Update',
|
|
61
|
+
rowsProduced: 0,
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
/** Every unique index of the table, checked for the row an `INSERT` is about to add. */
|
|
65
|
+
export function enforceUniqueOnInsert(ctx, row) {
|
|
66
|
+
for (const spec of indexSpecsOf(ctx.table)) {
|
|
67
|
+
if (spec.unique)
|
|
68
|
+
checkKey(ctx, spec, keyValuesOf(row, spec.columns), 'INSERT');
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
/** The one unique index `spec`, checked for the values an `UPDATE` is about to give a row (its old entry is already out of the tree). */
|
|
72
|
+
export function enforceUniqueOnUpdate(ctx, spec, values) {
|
|
73
|
+
if (spec.unique)
|
|
74
|
+
checkKey(ctx, spec, values, 'UPDATE');
|
|
75
|
+
}
|
|
76
|
+
/** What the row holding a taken key looks like, for the error's hint. */
|
|
77
|
+
export function holderOf(ctx, violation) {
|
|
78
|
+
return rowAt(ctx, violation.holder);
|
|
79
|
+
}
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Executes an `UPDATE`'s access plan (`SeqScan` or `IndexScan`, from
|
|
3
|
+
* `emitUpdatePlanEvents`) against the real heap and index: finds every
|
|
4
|
+
* matching row exactly the way `DELETE` does, applies the SET assignments to
|
|
5
|
+
* produce its replacement, and — if any assignment touches the indexed
|
|
6
|
+
* column — drops the old key and adds the new one through the real
|
|
7
|
+
* `deleteKey`/`insert` algorithms. Delete-then-reinsert is the only honest
|
|
8
|
+
* way to keep a B+Tree's keys in order when the value they index changes;
|
|
9
|
+
* there is no in-place "rekey" (plan.md §22.2, the write path's last
|
|
10
|
+
* statement).
|
|
11
|
+
*
|
|
12
|
+
* Like `DELETE`, this finds every match before it changes anything — a
|
|
13
|
+
* `SELECT`'s Volcano operators pull one row at a time, but an update mid-scan
|
|
14
|
+
* could shift what a later step reads. It narrates through the same
|
|
15
|
+
* `execute`-stage shape (`SeqScan`/`IndexScan` while finding, then `Update`
|
|
16
|
+
* while applying) that `DELETE` and `INSERT` already use.
|
|
17
|
+
*/
|
|
18
|
+
import { enforceUniqueOnUpdate } from "./unique.js";
|
|
19
|
+
import { indexLookupMatches, indexRangeMatches } from "./writeScan.js";
|
|
20
|
+
import { evaluateValue, passes } from "./evaluate.js";
|
|
21
|
+
import { deleteKey, entryFor, height, indexNameOf, indexSpecsOf, insert, keyText, keyValuesOf } from "../index/index.js";
|
|
22
|
+
import { summarizeSteps } from "./delete.js";
|
|
23
|
+
import { summarizeInsert } from "./insert.js";
|
|
24
|
+
const UPDATE = { name: 'UPDATE', does: 'modify' };
|
|
25
|
+
/**
|
|
26
|
+
* `row` with every SET assignment applied — a new object; `row` itself is
|
|
27
|
+
* untouched. Every value is evaluated against the row *as it was*, so
|
|
28
|
+
* `SET a = b, b = a` swaps the two rather than copying one over the other
|
|
29
|
+
* (standard SQL, and what plan.md §25.4 B1b's `SET n = n + 1` relies on).
|
|
30
|
+
*/
|
|
31
|
+
function applyAssignments(row, assignments) {
|
|
32
|
+
const next = { ...row };
|
|
33
|
+
for (const a of assignments)
|
|
34
|
+
next[a.column] = evaluateValue(a.value, row);
|
|
35
|
+
return next;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Moves the row's entry from its old values to its new ones in each index whose own columns actually changed.
|
|
39
|
+
* A NULL leading column is never indexed, so on either side it just drops the delete or the insert half. A table
|
|
40
|
+
* can have several indexes; an assignment only reindexes the ones whose columns it touched, one sentence each.
|
|
41
|
+
*/
|
|
42
|
+
function reindexRow(ctx, before, after, pointer) {
|
|
43
|
+
const sentences = [];
|
|
44
|
+
for (const spec of indexSpecsOf(ctx.table)) {
|
|
45
|
+
const name = indexNameOf(spec.columns);
|
|
46
|
+
const tree = ctx.trees[name];
|
|
47
|
+
if (!tree)
|
|
48
|
+
continue;
|
|
49
|
+
const beforeValues = keyValuesOf(before, spec.columns);
|
|
50
|
+
const afterValues = keyValuesOf(after, spec.columns);
|
|
51
|
+
if (beforeValues.every((v, i) => v === afterValues[i]))
|
|
52
|
+
continue; // none of this index's columns changed
|
|
53
|
+
const parts = [];
|
|
54
|
+
if (beforeValues[0] !== null) {
|
|
55
|
+
const { key } = entryFor(spec.columns, before, pointer, ctx.table.clusteredKey);
|
|
56
|
+
const removed = deleteKey(tree, key);
|
|
57
|
+
if (removed.found) {
|
|
58
|
+
parts.push(`drops key ${keyText(beforeValues)} — ${summarizeSteps(removed.steps)}`);
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
if (afterValues[0] !== null) {
|
|
62
|
+
// Its old entry is out, so what remains is everyone else's: a UNIQUE index must not already hold the new key.
|
|
63
|
+
enforceUniqueOnUpdate(ctx, spec, afterValues);
|
|
64
|
+
const nodesBefore = Object.keys(tree.nodes).length;
|
|
65
|
+
const heightBefore = height(tree);
|
|
66
|
+
const { key, pointer: stored } = entryFor(spec.columns, after, pointer, ctx.table.clusteredKey);
|
|
67
|
+
insert(tree, key, stored);
|
|
68
|
+
const nodesAfter = Object.keys(tree.nodes).length;
|
|
69
|
+
const heightAfter = height(tree);
|
|
70
|
+
parts.push(`adds key ${keyText(afterValues)} — ${summarizeInsert(nodesBefore, nodesAfter, heightBefore, heightAfter)}`);
|
|
71
|
+
}
|
|
72
|
+
if (parts.length > 0) {
|
|
73
|
+
sentences.push(`Because \`${name}\` changed, its B+Tree ${parts.join(', then ')}.`);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return sentences.length > 0 ? sentences.join(' ') : null;
|
|
77
|
+
}
|
|
78
|
+
/** Every heap page, filtered by `filter` if present — the SeqScan half of an update. */
|
|
79
|
+
function scanForMatches(ctx, filter) {
|
|
80
|
+
const matched = [];
|
|
81
|
+
let produced = 0;
|
|
82
|
+
for (const page of ctx.heap.pages) {
|
|
83
|
+
ctx.locks.acquire(page.pageId, 'exclusive', `Take an exclusive lock on page ${String(page.pageId)} — UPDATE may modify a row on it, so a shared lock is not enough.`);
|
|
84
|
+
const fetched = ctx.pool.fetch(page.pageId);
|
|
85
|
+
if (fetched)
|
|
86
|
+
ctx.pool.unpin(fetched.frameId);
|
|
87
|
+
let touchedPage = false;
|
|
88
|
+
page.rows.forEach((row, slot) => {
|
|
89
|
+
if (!passes(filter, row))
|
|
90
|
+
return;
|
|
91
|
+
matched.push({ row, pointer: { pageId: page.pageId, slot } });
|
|
92
|
+
touchedPage = true;
|
|
93
|
+
produced += 1;
|
|
94
|
+
ctx.emit(`SeqScan finds a matching row on page ${page.pageId} — UPDATE will modify it.`, {
|
|
95
|
+
stage: 'execute',
|
|
96
|
+
op: 'SeqScan',
|
|
97
|
+
rowsProduced: produced,
|
|
98
|
+
});
|
|
99
|
+
});
|
|
100
|
+
if (touchedPage)
|
|
101
|
+
ctx.pool.markDirty(page.pageId);
|
|
102
|
+
ctx.locks.release(page.pageId, 'exclusive');
|
|
103
|
+
}
|
|
104
|
+
return matched;
|
|
105
|
+
}
|
|
106
|
+
export function executeUpdate(plan, assignments, ctx) {
|
|
107
|
+
const matches = plan.op === 'IndexScan'
|
|
108
|
+
? indexLookupMatches(ctx, plan, UPDATE)
|
|
109
|
+
: plan.op === 'IndexRangeScan'
|
|
110
|
+
? indexRangeMatches(ctx, plan, UPDATE)
|
|
111
|
+
: scanForMatches(ctx, plan.op === 'SeqScan' ? plan.filter : undefined);
|
|
112
|
+
const before = [];
|
|
113
|
+
const after = [];
|
|
114
|
+
let produced = 0;
|
|
115
|
+
for (const { row, pointer } of matches) {
|
|
116
|
+
const updated = applyAssignments(row, assignments);
|
|
117
|
+
before.push(row);
|
|
118
|
+
after.push(updated);
|
|
119
|
+
produced += 1;
|
|
120
|
+
const indexDetail = reindexRow(ctx, row, updated, pointer);
|
|
121
|
+
ctx.emit(`Update changes row ${JSON.stringify(row)} to ${JSON.stringify(updated)}.${indexDetail ? ` ${indexDetail}` : ''}`, { stage: 'execute', op: 'Update', rowsProduced: produced });
|
|
122
|
+
}
|
|
123
|
+
return { before, after };
|
|
124
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How a `DELETE` or `UPDATE` finds its rows through an index (plan.md §22.2's write path; ranges since §25.4 B3b).
|
|
3
|
+
*
|
|
4
|
+
* A write cannot stream: it has to find *every* match before it changes anything, or a change mid-scan could shift what
|
|
5
|
+
* a later step reads (an `UPDATE` that moves a row's key would find it again). So each of these walks the index to
|
|
6
|
+
* the end, takes an exclusive lock on the heap page of every row it will touch, marks that page dirty, and hands back
|
|
7
|
+
* the rows with their addresses — the address is what lets the executor drop and re-add exactly *that* row's index entry.
|
|
8
|
+
* `exec/delete.ts` and `exec/update.ts` share them, and differ only in the verb the narration uses.
|
|
9
|
+
*/
|
|
10
|
+
import { lookupKeysOf } from "../planner/index.js";
|
|
11
|
+
import { fetchHeapPage, rangeHolds, rowAt } from "./operators.js";
|
|
12
|
+
import { passes } from "./evaluate.js";
|
|
13
|
+
import { emitIndexLookup, emitRangeStart, nextInRange, resolveClustered } from "../index/index.js";
|
|
14
|
+
/** Takes the exclusive lock a write needs on the row's heap page, marks it dirty, and narrates the find. */
|
|
15
|
+
function claim(ctx, match, verb, op, found, of) {
|
|
16
|
+
ctx.locks.acquire(match.pointer.pageId, 'exclusive', `Upgrade to an exclusive lock on heap page ${String(match.pointer.pageId)} — ${verb.name} is about to ${verb.does} the row it holds.`);
|
|
17
|
+
ctx.pool.markDirty(match.pointer.pageId);
|
|
18
|
+
ctx.emit(of, { stage: 'execute', op, rowsProduced: found });
|
|
19
|
+
ctx.locks.release(match.pointer.pageId, 'exclusive');
|
|
20
|
+
}
|
|
21
|
+
const rejected = (ctx, verb, op, found) => {
|
|
22
|
+
ctx.emit(`The row the index found fails the rest of the predicate, so ${verb.name} leaves it alone.`, {
|
|
23
|
+
stage: 'execute',
|
|
24
|
+
op,
|
|
25
|
+
rowsProduced: found,
|
|
26
|
+
});
|
|
27
|
+
};
|
|
28
|
+
/** Every row an equality lookup points at that passes any residual predicate — one, or one per index entry. */
|
|
29
|
+
export function indexLookupMatches(ctx, plan, verb) {
|
|
30
|
+
const tree = ctx.trees[plan.column];
|
|
31
|
+
if (!tree)
|
|
32
|
+
return [];
|
|
33
|
+
const outcome = emitIndexLookup(ctx.emit, tree, lookupKeysOf(plan), ctx.pool, plan.column, ctx.locks);
|
|
34
|
+
const { pointers, clusterKeys } = outcome;
|
|
35
|
+
const clusteredTree = tree.clusteredVia !== undefined ? ctx.trees[tree.clusteredVia] : undefined;
|
|
36
|
+
const matched = [];
|
|
37
|
+
pointers.forEach((pointer, at) => {
|
|
38
|
+
// A clustered table's secondary entry has no address (plan.md §25.4 B3e): every match costs a real second
|
|
39
|
+
// descent, through the clustered index. Otherwise the lookup already fetched the first match's page, and the
|
|
40
|
+
// rest are read now, as their rows are examined.
|
|
41
|
+
const clusterKey = clusterKeys?.[at];
|
|
42
|
+
const real = clusterKey && clusteredTree ? resolveClustered(ctx.emit, clusteredTree, tree.clusteredVia, clusterKey, ctx.pool, ctx.locks) : pointer;
|
|
43
|
+
if (!real)
|
|
44
|
+
return;
|
|
45
|
+
if (!clusterKey && at > 0)
|
|
46
|
+
fetchHeapPage(ctx, pointer);
|
|
47
|
+
const row = rowAt(ctx, real);
|
|
48
|
+
if (!row)
|
|
49
|
+
return;
|
|
50
|
+
if (!passes(plan.residual, row)) {
|
|
51
|
+
rejected(ctx, verb, 'IndexScan', matched.length);
|
|
52
|
+
return;
|
|
53
|
+
}
|
|
54
|
+
matched.push({ row, pointer: real });
|
|
55
|
+
claim(ctx, { row, pointer: real }, verb, 'IndexScan', matched.length, pointers.length === 1
|
|
56
|
+
? `IndexScan finds the one row ${verb.name} will ${verb.does}.`
|
|
57
|
+
: `IndexScan finds a row ${verb.name} will ${verb.does} — ${String(matched.length)} so far, of ${String(pointers.length)} entries for this value.`);
|
|
58
|
+
});
|
|
59
|
+
return matched;
|
|
60
|
+
}
|
|
61
|
+
/** Every row a range walk of the index reaches that passes the range and any residual predicate. */
|
|
62
|
+
export function indexRangeMatches(ctx, plan, verb) {
|
|
63
|
+
const tree = ctx.trees[plan.column];
|
|
64
|
+
if (!tree)
|
|
65
|
+
return [];
|
|
66
|
+
const start = emitRangeStart(ctx.emit, tree, plan.column, plan.low, ctx.pool, ctx.locks, plan.prefix);
|
|
67
|
+
let leafId = start.leafId;
|
|
68
|
+
let index = start.startIndex;
|
|
69
|
+
const matched = [];
|
|
70
|
+
while (leafId !== null) {
|
|
71
|
+
const found = nextInRange(ctx.emit, tree, leafId, index, plan.high, { pool: ctx.pool, locks: ctx.locks }, plan.prefix);
|
|
72
|
+
if (!found)
|
|
73
|
+
break;
|
|
74
|
+
leafId = found.leafId;
|
|
75
|
+
index = found.nextIndex;
|
|
76
|
+
fetchHeapPage(ctx, found.pointer);
|
|
77
|
+
const row = rowAt(ctx, found.pointer);
|
|
78
|
+
if (!row || !rangeHolds(row, plan))
|
|
79
|
+
continue;
|
|
80
|
+
if (!passes(plan.residual, row)) {
|
|
81
|
+
rejected(ctx, verb, 'IndexRangeScan', matched.length);
|
|
82
|
+
continue;
|
|
83
|
+
}
|
|
84
|
+
matched.push({ row, pointer: found.pointer });
|
|
85
|
+
claim(ctx, { row, pointer: found.pointer }, verb, 'IndexRangeScan', matched.length, `IndexRangeScan finds a row ${verb.name} will ${verb.does} via leaf ${found.leafId} — ${String(matched.length)} so far.`);
|
|
86
|
+
}
|
|
87
|
+
return matched;
|
|
88
|
+
}
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `EXPLAIN [ANALYZE] SELECT …` as a statement (plan.md §25.4 B6) — the in-app
|
|
3
|
+
* EXPLAIN that used to be only a button, now something you can *type*, and
|
|
4
|
+
* whose answer is a result you can read, copy and share like any other.
|
|
5
|
+
*
|
|
6
|
+
* EXPLAIN plans and optimises the query and shows the plan with the
|
|
7
|
+
* cost model's estimate for every operator. **Nothing is
|
|
8
|
+
* read** — the trace stops before the executor, exactly as
|
|
9
|
+
* a real EXPLAIN never touches the data.
|
|
10
|
+
* EXPLAIN ANALYZE runs the query for real (the trace *is* that run) and puts
|
|
11
|
+
* the executor's actual row count beside each estimate, then
|
|
12
|
+
* the buffer pool's hits, misses and evictions. It reports
|
|
13
|
+
* no timings: this engine makes no claim about latency (§12).
|
|
14
|
+
*
|
|
15
|
+
* Built on the ordinary run of the inner query, so the plan shown is exactly
|
|
16
|
+
* the plan that would execute. The `EXPLAIN [ANALYZE]` prefix is replaced by
|
|
17
|
+
* spaces before that run, which keeps every span the inner statement reports
|
|
18
|
+
* (the editor's squiggles, the parse tree's highlights) pointing at the text
|
|
19
|
+
* the user actually typed.
|
|
20
|
+
*/
|
|
21
|
+
import { createTracer } from "./trace.js";
|
|
22
|
+
export const EXPLAIN_COLUMN = 'QUERY PLAN';
|
|
23
|
+
/** The optimizer's final plan: the last rewrite's `after`, or the canonical plan if nothing rewrote it. */
|
|
24
|
+
export function finalPlan(events) {
|
|
25
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
26
|
+
const e = events[i];
|
|
27
|
+
if (e.stage === 'optimize')
|
|
28
|
+
return e.after;
|
|
29
|
+
if (e.stage === 'plan')
|
|
30
|
+
return e.planTree;
|
|
31
|
+
}
|
|
32
|
+
return null;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* `max(est/act, act/est)`, Leis et al.'s own metric for cardinality-estimation error (plan.md §25.4 C2;
|
|
36
|
+
* RESEARCH.md §41) — `1` is a perfect guess, `10` an estimate ten times too high or too low, whichever direction
|
|
37
|
+
* it missed. Both sides get the literature's own `+1` smoothing, so a `0` estimate or a `0` actual row count is
|
|
38
|
+
* still a real, finite number rather than a division by zero.
|
|
39
|
+
*/
|
|
40
|
+
function qError(estRows, actualRows) {
|
|
41
|
+
return Math.max((estRows + 1) / (actualRows + 1), (actualRows + 1) / (estRows + 1));
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* The key `actual` (below) is built and read by: `op` alone collides whenever a plan has two nodes of the same
|
|
45
|
+
* kind — a multi-join chain's several `SeqScan`s or several `Join`s (plan.md §25.4 C3 slice b flagged this as a
|
|
46
|
+
* real, pre-existing bug) — so every row this engine ever produces for one of them would be misreported onto
|
|
47
|
+
* all the others sharing that `op`. `table` (a scan's own table, or a `Join`'s `rightTable`) is always enough to
|
|
48
|
+
* tell them apart, since the parser never lets a query name the same table twice.
|
|
49
|
+
*/
|
|
50
|
+
export function nodeKey(op, table) {
|
|
51
|
+
return table === undefined ? op : `${op}:${table}`;
|
|
52
|
+
}
|
|
53
|
+
/** One line per operator, PostgreSQL-shaped: the root, then `->` for each level beneath it. */
|
|
54
|
+
function planLines(root, actual) {
|
|
55
|
+
const lines = [];
|
|
56
|
+
const walk = (node, depth) => {
|
|
57
|
+
const parts = [];
|
|
58
|
+
if (node.estRows !== undefined)
|
|
59
|
+
parts.push(`est ~${String(node.estRows)} rows`);
|
|
60
|
+
const rows = actual?.get(nodeKey(node.op, node.table));
|
|
61
|
+
if (rows !== undefined) {
|
|
62
|
+
parts.push(`actual ${String(rows)} rows`);
|
|
63
|
+
// Below 2× is well within the noise a real optimizer lives with every day — naming it would just be clutter.
|
|
64
|
+
if (node.estRows !== undefined) {
|
|
65
|
+
const q = qError(node.estRows, rows);
|
|
66
|
+
if (q >= 2)
|
|
67
|
+
parts.push(`q-error ${q.toFixed(1)}×`);
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
const indent = depth === 0 ? '' : `${' '.repeat(depth - 1)} -> `;
|
|
71
|
+
const text = node.detail ? `${node.label} ${node.detail}` : node.label;
|
|
72
|
+
lines.push(`${indent}${text}${parts.length > 0 ? ` (${parts.join('; ')})` : ''}`);
|
|
73
|
+
(node.children ?? []).forEach((child) => { walk(child, depth + 1); });
|
|
74
|
+
};
|
|
75
|
+
walk(root, 0);
|
|
76
|
+
return lines;
|
|
77
|
+
}
|
|
78
|
+
export function explainResult(sql, ast, run) {
|
|
79
|
+
const inner = run(' '.repeat(ast.select.span.from) + sql.slice(ast.select.span.from));
|
|
80
|
+
if (inner.error)
|
|
81
|
+
return inner;
|
|
82
|
+
const plan = finalPlan(inner.events);
|
|
83
|
+
if (!plan)
|
|
84
|
+
return { events: inner.events, rows: [] };
|
|
85
|
+
// What ran, per operator: the largest count it reported. Bookkeeping events (a discarded candidate, a hash build
|
|
86
|
+
// announcement) can carry a 0 that lands after the real count has grown, so the last event is not always the total.
|
|
87
|
+
const actual = new Map();
|
|
88
|
+
for (const e of inner.events) {
|
|
89
|
+
if (e.stage !== 'execute')
|
|
90
|
+
continue;
|
|
91
|
+
const key = nodeKey(e.op, e.table);
|
|
92
|
+
actual.set(key, Math.max(actual.get(key) ?? 0, e.rowsProduced));
|
|
93
|
+
}
|
|
94
|
+
const lines = planLines(plan, ast.analyze ? actual : null);
|
|
95
|
+
if (ast.analyze) {
|
|
96
|
+
const count = (action) => inner.events.filter((e) => e.stage === 'buffer' && e.action === action).length;
|
|
97
|
+
lines.push(`Result: ${String(inner.rows.length)} row${inner.rows.length === 1 ? '' : 's'}`, `Buffer pool: ${String(count('hit'))} hit${count('hit') === 1 ? '' : 's'}, ${String(count('miss'))} miss${count('miss') === 1 ? '' : 'es'}, ${String(count('evict'))} eviction${count('evict') === 1 ? '' : 's'}`);
|
|
98
|
+
}
|
|
99
|
+
const rows = lines.map((line) => ({ [EXPLAIN_COLUMN]: line }));
|
|
100
|
+
// Plain EXPLAIN stops before the executor; EXPLAIN ANALYZE keeps the whole run but replaces its result.
|
|
101
|
+
const kept = inner.events.filter((e) => ast.analyze ? e.stage !== 'result' : e.stage === 'parse' || e.stage === 'plan' || e.stage === 'optimize');
|
|
102
|
+
const { emit, drain } = createTracer();
|
|
103
|
+
for (const e of kept) {
|
|
104
|
+
const { id: _id, label, predictable, dwell, ...body } = e;
|
|
105
|
+
emit(label, body, {
|
|
106
|
+
...(predictable ? { predictable } : {}),
|
|
107
|
+
...(dwell === undefined ? {} : { dwell }),
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
emit(ast.analyze
|
|
111
|
+
? `EXPLAIN ANALYZE ran the query for real — the trace above is that run — and reports the cost model's estimate next to the executor's actual row count for every operator, then what the buffer pool did.`
|
|
112
|
+
: `EXPLAIN shows the plan the optimizer chose and the cost model's estimate for each operator. Nothing was executed and no page was read — that is what separates EXPLAIN from EXPLAIN ANALYZE.`, { stage: 'result', rows });
|
|
113
|
+
return { events: drain(), rows };
|
|
114
|
+
}
|