querylens 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +35 -0
- package/dist/bin/querylens.js +208 -0
- package/dist/src/engine/bufferTrace.js +67 -0
- package/dist/src/engine/datasets.js +139 -0
- package/dist/src/engine/exec/delete.js +95 -0
- package/dist/src/engine/exec/evaluate.js +174 -0
- package/dist/src/engine/exec/index.js +4 -0
- package/dist/src/engine/exec/insert.js +75 -0
- package/dist/src/engine/exec/operators.js +1290 -0
- package/dist/src/engine/exec/run.js +35 -0
- package/dist/src/engine/exec/sort.js +171 -0
- package/dist/src/engine/exec/unique.js +79 -0
- package/dist/src/engine/exec/update.js +124 -0
- package/dist/src/engine/exec/writeScan.js +88 -0
- package/dist/src/engine/explain.js +114 -0
- package/dist/src/engine/index/btree.js +481 -0
- package/dist/src/engine/index/build.js +99 -0
- package/dist/src/engine/index/bulk.js +107 -0
- package/dist/src/engine/index/display.js +38 -0
- package/dist/src/engine/index/index.js +9 -0
- package/dist/src/engine/index/lookup.js +213 -0
- package/dist/src/engine/index/rangeLookup.js +158 -0
- package/dist/src/engine/index/spec.js +47 -0
- package/dist/src/engine/index/unique.js +31 -0
- package/dist/src/engine/index/validate.js +105 -0
- package/dist/src/engine/index.js +16 -0
- package/dist/src/engine/locks/index.js +1 -0
- package/dist/src/engine/locks/lockManager.js +46 -0
- package/dist/src/engine/parser/ast.js +77 -0
- package/dist/src/engine/parser/display.js +404 -0
- package/dist/src/engine/parser/index.js +4 -0
- package/dist/src/engine/parser/parser.js +1108 -0
- package/dist/src/engine/parser/print.js +74 -0
- package/dist/src/engine/parser/tokenizer.js +146 -0
- package/dist/src/engine/planner/buildPlan.js +208 -0
- package/dist/src/engine/planner/cost.js +582 -0
- package/dist/src/engine/planner/emit.js +267 -0
- package/dist/src/engine/planner/emitDelete.js +57 -0
- package/dist/src/engine/planner/emitUpdate.js +51 -0
- package/dist/src/engine/planner/index.js +8 -0
- package/dist/src/engine/planner/joinOrder.js +252 -0
- package/dist/src/engine/planner/optimize.js +906 -0
- package/dist/src/engine/planner/plan.js +445 -0
- package/dist/src/engine/predict.js +120 -0
- package/dist/src/engine/runQuery.js +393 -0
- package/dist/src/engine/seed.js +165 -0
- package/dist/src/engine/stats.js +118 -0
- package/dist/src/engine/storage/bufferPool.js +194 -0
- package/dist/src/engine/storage/index.js +3 -0
- package/dist/src/engine/storage/page.js +46 -0
- package/dist/src/engine/storage/policy.js +360 -0
- package/dist/src/engine/subquery.js +88 -0
- package/dist/src/engine/trace.js +17 -0
- package/dist/src/engine/types.js +39 -0
- package/dist/src/engine/value.js +80 -0
- package/dist/src/engine/viewState.js +187 -0
- package/package.json +40 -0
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `ANALYZE` (plan.md §25.4 C1) — a real scan of a real table, producing the statistics a query planner actually
|
|
3
|
+
* reads instead of assuming every value in a column is equally likely and every column independent of every
|
|
4
|
+
* other. Shaped after PostgreSQL's own `pg_stats`: distinct count, null fraction, most-common values (with their
|
|
5
|
+
* real frequencies), an equi-depth histogram over what is left, and the column's physical/logical correlation.
|
|
6
|
+
*
|
|
7
|
+
* Pure and synchronous, on the same rules as the rest of `src/engine`. A full scan, not a sample — a real
|
|
8
|
+
* `ANALYZE` samples a fraction of the table for speed and lives with the noise that introduces; a toy table is
|
|
9
|
+
* small enough to just look at every row, so this engine's statistics are as accurate as a scan gets. What they
|
|
10
|
+
* are *not* is fresh forever: `Table.stats` is a snapshot of one `ANALYZE` run, and stays exactly that — even
|
|
11
|
+
* once wrong — until `ANALYZE` runs again (plan.md §25.4 C2 is where that staleness is made to matter).
|
|
12
|
+
*/
|
|
13
|
+
import { compareValues } from "./value.js";
|
|
14
|
+
/** How many of a column's most frequent values get their own entry, PostgreSQL's own default. */
|
|
15
|
+
const MOST_COMMON_LIMIT = 5;
|
|
16
|
+
/** Buckets in the equi-depth histogram over whatever `mostCommonValues` does not already cover. */
|
|
17
|
+
const HISTOGRAM_BUCKETS = 10;
|
|
18
|
+
/** Below this many non-MCV rows, a histogram is noise, not signal — left empty instead. */
|
|
19
|
+
const MIN_ROWS_FOR_HISTOGRAM = 20;
|
|
20
|
+
/** `sorted`'s `buckets`-quantile boundaries: `n + 1` values bounding `n` roughly-equal-depth buckets. */
|
|
21
|
+
function equiDepthBounds(sorted, buckets) {
|
|
22
|
+
const n = Math.max(1, Math.min(buckets, sorted.length - 1));
|
|
23
|
+
const bounds = [];
|
|
24
|
+
for (let i = 0; i <= n; i++) {
|
|
25
|
+
const at = Math.round((i * (sorted.length - 1)) / n);
|
|
26
|
+
bounds.push(sorted[at]);
|
|
27
|
+
}
|
|
28
|
+
return bounds;
|
|
29
|
+
}
|
|
30
|
+
/** Pearson correlation between two equal-length number series — 0 when either has no variation to correlate. */
|
|
31
|
+
function pearson(xs, ys) {
|
|
32
|
+
const n = xs.length;
|
|
33
|
+
const meanX = xs.reduce((s, x) => s + x, 0) / n;
|
|
34
|
+
const meanY = ys.reduce((s, y) => s + y, 0) / n;
|
|
35
|
+
let num = 0;
|
|
36
|
+
let denX = 0;
|
|
37
|
+
let denY = 0;
|
|
38
|
+
for (let i = 0; i < n; i++) {
|
|
39
|
+
const dx = xs[i] - meanX;
|
|
40
|
+
const dy = ys[i] - meanY;
|
|
41
|
+
num += dx * dy;
|
|
42
|
+
denX += dx * dx;
|
|
43
|
+
denY += dy * dy;
|
|
44
|
+
}
|
|
45
|
+
return denX === 0 || denY === 0 ? 0 : num / Math.sqrt(denX * denY);
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* How much `values`' order tracks physical row order: each value's *rank* among all of them (tied values share
|
|
49
|
+
* the average of their positions, same as Spearman's), correlated against its row index. `compareValues`'s
|
|
50
|
+
* NULL-first order decides ties among NULLs the same way it decides everything else here — there is no second
|
|
51
|
+
* rule for them.
|
|
52
|
+
*/
|
|
53
|
+
function correlationOf(values) {
|
|
54
|
+
const n = values.length;
|
|
55
|
+
if (n < 2)
|
|
56
|
+
return 0;
|
|
57
|
+
const byValue = values.map((_, i) => i).sort((a, b) => compareValues(values[a], values[b]));
|
|
58
|
+
const rank = new Array(n);
|
|
59
|
+
let i = 0;
|
|
60
|
+
while (i < n) {
|
|
61
|
+
let j = i;
|
|
62
|
+
while (j + 1 < n && compareValues(values[byValue[j + 1]], values[byValue[i]]) === 0)
|
|
63
|
+
j += 1;
|
|
64
|
+
const averageRank = (i + j) / 2;
|
|
65
|
+
for (let k = i; k <= j; k++)
|
|
66
|
+
rank[byValue[k]] = averageRank;
|
|
67
|
+
i = j + 1;
|
|
68
|
+
}
|
|
69
|
+
const physicalIndex = Array.from({ length: n }, (_, i) => i);
|
|
70
|
+
return pearson(physicalIndex, rank);
|
|
71
|
+
}
|
|
72
|
+
function analyzeColumn(rows, column) {
|
|
73
|
+
const values = rows.map((row) => row[column] ?? null);
|
|
74
|
+
const nonNull = values.filter((v) => v !== null);
|
|
75
|
+
const nullFraction = values.length === 0 ? 0 : (values.length - nonNull.length) / values.length;
|
|
76
|
+
const counts = new Map();
|
|
77
|
+
for (const v of nonNull)
|
|
78
|
+
counts.set(v, (counts.get(v) ?? 0) + 1);
|
|
79
|
+
// "Most common" means it actually repeats — a column of all-distinct values (a key) has none, honestly.
|
|
80
|
+
const byFrequency = [...counts.entries()]
|
|
81
|
+
.filter(([, frequency]) => frequency > 1)
|
|
82
|
+
.sort((a, b) => b[1] - a[1] || compareValues(a[0], b[0]));
|
|
83
|
+
const mostCommonValues = byFrequency.slice(0, MOST_COMMON_LIMIT).map(([value, frequency]) => ({ value, frequency }));
|
|
84
|
+
const covered = new Set(mostCommonValues.map((m) => m.value));
|
|
85
|
+
const remaining = nonNull.filter((v) => !covered.has(v)).sort(compareValues);
|
|
86
|
+
const histogramBounds = remaining.length < MIN_ROWS_FOR_HISTOGRAM ? [] : equiDepthBounds(remaining, HISTOGRAM_BUCKETS);
|
|
87
|
+
return {
|
|
88
|
+
nDistinct: counts.size,
|
|
89
|
+
nullFraction,
|
|
90
|
+
mostCommonValues,
|
|
91
|
+
histogramBounds,
|
|
92
|
+
correlation: correlationOf(values),
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
/** The table's statistics, as of right now — what `ANALYZE <table>` computes. */
|
|
96
|
+
export function analyzeTable(table) {
|
|
97
|
+
const columns = {};
|
|
98
|
+
for (const column of table.columns)
|
|
99
|
+
columns[column] = analyzeColumn(table.rows, column);
|
|
100
|
+
return { rowCount: table.rows.length, columns };
|
|
101
|
+
}
|
|
102
|
+
function describeValue(value) {
|
|
103
|
+
if (value === null)
|
|
104
|
+
return 'NULL';
|
|
105
|
+
return typeof value === 'string' ? `'${value}'` : String(value);
|
|
106
|
+
}
|
|
107
|
+
/** `ANALYZE`'s own result, one row per column — what the Playground's result table (and the CLI) already know
|
|
108
|
+
* how to render, so a learner opens the statistics the same way they open any other query's answer. */
|
|
109
|
+
export function statsToRows(stats) {
|
|
110
|
+
return Object.entries(stats.columns).map(([column, s]) => ({
|
|
111
|
+
column,
|
|
112
|
+
n_distinct: s.nDistinct,
|
|
113
|
+
null_frac: Math.round(s.nullFraction * 1000) / 1000,
|
|
114
|
+
most_common: s.mostCommonValues.length === 0 ? null : s.mostCommonValues.map((m) => `${describeValue(m.value)} (${String(m.frequency)})`).join(', '),
|
|
115
|
+
histogram_buckets: Math.max(0, s.histogramBounds.length - 1),
|
|
116
|
+
correlation: Math.round(s.correlation * 1000) / 1000,
|
|
117
|
+
}));
|
|
118
|
+
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The buffer pool: a fixed set of frames standing between the executor and
|
|
3
|
+
* "disk". Every page the engine reads goes through `fetch`, which is the only
|
|
4
|
+
* place a disk read can happen — so the hit/miss numbers in the UI are exactly
|
|
5
|
+
* the ones the engine experienced.
|
|
6
|
+
*/
|
|
7
|
+
import { policyFor } from "./policy.js";
|
|
8
|
+
import { evictionPrompt } from "../predict.js";
|
|
9
|
+
/** Playback pacing weight — a disk read should *feel* slower than a hit. */
|
|
10
|
+
export const DISK_READ_DWELL = 2.5;
|
|
11
|
+
function emptyState(frameCount) {
|
|
12
|
+
return {
|
|
13
|
+
frames: Array.from({ length: frameCount }, (_, frameId) => ({
|
|
14
|
+
frameId,
|
|
15
|
+
pageId: null,
|
|
16
|
+
usageCount: 0,
|
|
17
|
+
pinCount: 0,
|
|
18
|
+
lastUsedAt: -1,
|
|
19
|
+
loadedAt: -1,
|
|
20
|
+
})),
|
|
21
|
+
now: 0,
|
|
22
|
+
hand: 0,
|
|
23
|
+
lookaheadCursor: 0,
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
/** How the `evict` event narrates the choice, per policy. */
|
|
27
|
+
function evictionReason(policy) {
|
|
28
|
+
switch (policy) {
|
|
29
|
+
case 'lru':
|
|
30
|
+
return 'LRU picked it because it was used least recently.';
|
|
31
|
+
case 'clock':
|
|
32
|
+
return 'Clock sweep picked it because the hand reached it with a usage count of zero.';
|
|
33
|
+
case 'fifo':
|
|
34
|
+
return 'FIFO picked it because it has been resident longest — first in, first out, however much it was used.';
|
|
35
|
+
case 'second-chance':
|
|
36
|
+
return 'Second-chance picked it: the oldest frame whose reference bit was already clear.';
|
|
37
|
+
case 'optimal':
|
|
38
|
+
return "The optimal policy picked it because its next use is farthest away — the choice no real policy can make, since it needs the future.";
|
|
39
|
+
case 'midpoint':
|
|
40
|
+
return 'Midpoint LRU picked the least-recently-used page in the old sublist — a page read once and never touched again, which is exactly what it is built to evict first.';
|
|
41
|
+
case 'two-q':
|
|
42
|
+
return '2Q picked it: the oldest page in the small queue of pages read only once, which is where a one-pass scan leaves its pages.';
|
|
43
|
+
case 'lru-k':
|
|
44
|
+
return 'LRU-2 picked it because its second-most-recent reference is the oldest — a page touched only once has no second reference at all, so it goes first.';
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* `pageCount` covers every page in the database — heap pages first, then index
|
|
49
|
+
* pages. The pool manages frames, not contents: what a page *holds* is the
|
|
50
|
+
* caller's business, which is why heap and index pages can share one pool and
|
|
51
|
+
* compete for the same frames.
|
|
52
|
+
*/
|
|
53
|
+
export function createBufferPool(pageCount, options, emit,
|
|
54
|
+
/**
|
|
55
|
+
* The full page-reference string, when the caller has one (`runBufferTrace`
|
|
56
|
+
* does; `runQuery` does not — its page accesses emerge from execution). Only
|
|
57
|
+
* the `optimal` policy reads it; every other policy ignores it.
|
|
58
|
+
*/
|
|
59
|
+
lookahead) {
|
|
60
|
+
const state = emptyState(options.bufferFrames);
|
|
61
|
+
if (lookahead !== undefined)
|
|
62
|
+
state.lookahead = lookahead;
|
|
63
|
+
const policy = policyFor(options.policy);
|
|
64
|
+
const stats = { hits: 0, misses: 0, evictions: 0, writes: 0 };
|
|
65
|
+
// Which access `fetch` is resolving, so `optimal` can look past it. Advances
|
|
66
|
+
// once per successful `fetch`, staying aligned with the reference index.
|
|
67
|
+
let accessIndex = -1;
|
|
68
|
+
// Only the first eviction asks. After that the learner has the rule, and a
|
|
69
|
+
// prompt on every one of them would be nagging rather than teaching.
|
|
70
|
+
let evictionAsked = false;
|
|
71
|
+
// Pages a DELETE (or a future write) has modified since they were read —
|
|
72
|
+
// resident or not, since a page can be marked dirty and evicted later in
|
|
73
|
+
// the same query. Cleared once the writeback happens.
|
|
74
|
+
const dirty = new Set();
|
|
75
|
+
const resident = (pageId) => state.frames.find((f) => f.pageId === pageId);
|
|
76
|
+
/**
|
|
77
|
+
* A free frame, or the policy's eviction victim — shared between a disk
|
|
78
|
+
* read's miss and a fresh write, since both need "a frame for this page"
|
|
79
|
+
* and both may have to evict for it under the same policy. `incoming` is
|
|
80
|
+
* narration only: what the emitted `evict` event says is displacing the
|
|
81
|
+
* victim.
|
|
82
|
+
*/
|
|
83
|
+
function claimFrame(incoming, forWrite) {
|
|
84
|
+
const free = state.frames.find((f) => f.pageId === null);
|
|
85
|
+
if (free)
|
|
86
|
+
return free;
|
|
87
|
+
const choice = policy.selectVictim(state, emit);
|
|
88
|
+
if (choice === null)
|
|
89
|
+
return null; // every frame pinned
|
|
90
|
+
const victimId = choice.frameId;
|
|
91
|
+
const victim = state.frames.find((f) => f.frameId === victimId);
|
|
92
|
+
const evicted = victim.pageId;
|
|
93
|
+
stats.evictions += 1;
|
|
94
|
+
// A clean page's eviction is free — the disk already holds the same
|
|
95
|
+
// bytes. A dirty one costs a write first, charged and narrated here,
|
|
96
|
+
// at the moment it is actually paid, not when the page was marked.
|
|
97
|
+
const writtenBack = dirty.delete(evicted);
|
|
98
|
+
if (writtenBack)
|
|
99
|
+
stats.writes += 1;
|
|
100
|
+
// The eviction question is asked once regardless of which path triggers
|
|
101
|
+
// it — a write can just as easily be the first eviction a trace shows.
|
|
102
|
+
const prompt = evictionAsked ? null : evictionPrompt(state.frames, victimId, incoming, options.policy);
|
|
103
|
+
evictionAsked = true;
|
|
104
|
+
emit(`The pool is full, so page ${evicted} is evicted from frame ${victimId} to make room ${forWrite ? 'for a freshly written page' : ''}. ${writtenBack
|
|
105
|
+
? `Page ${evicted} is dirty — a row was deleted from it — so evicting it costs a write back to disk first. `
|
|
106
|
+
: ''}${evictionReason(options.policy)}`, {
|
|
107
|
+
stage: 'buffer',
|
|
108
|
+
action: 'evict',
|
|
109
|
+
frameId: victimId,
|
|
110
|
+
victimPageId: evicted,
|
|
111
|
+
pageId: incoming,
|
|
112
|
+
...(choice.clockHand === undefined ? {} : { clockHand: choice.clockHand }),
|
|
113
|
+
...(writtenBack ? { writtenBack } : {}),
|
|
114
|
+
}, prompt ? { predictable: prompt } : undefined);
|
|
115
|
+
victim.pageId = null;
|
|
116
|
+
victim.usageCount = 0;
|
|
117
|
+
victim.loadedAt = -1;
|
|
118
|
+
return victim;
|
|
119
|
+
}
|
|
120
|
+
function markDirty(pageId) {
|
|
121
|
+
if (Number.isInteger(pageId) && pageId >= 0 && pageId < pageCount)
|
|
122
|
+
dirty.add(pageId);
|
|
123
|
+
}
|
|
124
|
+
function fetch(pageId) {
|
|
125
|
+
if (!Number.isInteger(pageId) || pageId < 0 || pageId >= pageCount)
|
|
126
|
+
return null;
|
|
127
|
+
accessIndex += 1;
|
|
128
|
+
state.lookaheadCursor = accessIndex;
|
|
129
|
+
const hitFrame = resident(pageId);
|
|
130
|
+
if (hitFrame) {
|
|
131
|
+
stats.hits += 1;
|
|
132
|
+
hitFrame.pinCount += 1;
|
|
133
|
+
policy.onAccess(state, hitFrame.frameId);
|
|
134
|
+
emit(`Page ${pageId} is already in frame ${hitFrame.frameId} — a buffer hit. No disk read.`, { stage: 'buffer', action: 'hit', pageId, frameId: hitFrame.frameId });
|
|
135
|
+
return { frameId: hitFrame.frameId, hit: true };
|
|
136
|
+
}
|
|
137
|
+
stats.misses += 1;
|
|
138
|
+
emit(`Page ${pageId} is not in the pool — a buffer miss.`, {
|
|
139
|
+
stage: 'buffer',
|
|
140
|
+
action: 'miss',
|
|
141
|
+
pageId,
|
|
142
|
+
});
|
|
143
|
+
const target = claimFrame(pageId, false);
|
|
144
|
+
if (!target)
|
|
145
|
+
return null;
|
|
146
|
+
// Bump the clock on every load, not only in `onAccess` — FIFO and
|
|
147
|
+
// second-chance order frames by `loadedAt`, and FIFO's `onAccess` is a
|
|
148
|
+
// no-op, so nothing else would advance it.
|
|
149
|
+
state.now += 1;
|
|
150
|
+
target.pageId = pageId;
|
|
151
|
+
target.usageCount = 0;
|
|
152
|
+
target.loadedAt = state.now;
|
|
153
|
+
target.pinCount += 1;
|
|
154
|
+
// A load is a miss, not a hit: `onLoad` (default `onAccess`) lets a policy
|
|
155
|
+
// treat the two differently — midpoint-LRU inserts at the old sublist.
|
|
156
|
+
(policy.onLoad ?? policy.onAccess)(state, target.frameId);
|
|
157
|
+
emit(`Read page ${pageId} from disk into frame ${target.frameId}. This is the expensive part — everything else is memory.`, { stage: 'buffer', action: 'load', pageId, frameId: target.frameId }, { dwell: DISK_READ_DWELL });
|
|
158
|
+
return { frameId: target.frameId, hit: false };
|
|
159
|
+
}
|
|
160
|
+
/**
|
|
161
|
+
* A page a caller is *producing* — a `Sort` run or merge-pass output — not
|
|
162
|
+
* reading off disk. Always a fresh page id (a `Sort` operator never reuses
|
|
163
|
+
* one within a run), so unlike `fetch` there is no hit path: it either
|
|
164
|
+
* takes a free frame or evicts, exactly the way a disk read does, and the
|
|
165
|
+
* two kinds of page genuinely compete for the same pool.
|
|
166
|
+
*/
|
|
167
|
+
function write(pageId) {
|
|
168
|
+
if (!Number.isInteger(pageId) || pageId < 0 || pageId >= pageCount)
|
|
169
|
+
return null;
|
|
170
|
+
stats.writes += 1;
|
|
171
|
+
const target = claimFrame(pageId, true);
|
|
172
|
+
if (!target)
|
|
173
|
+
return null;
|
|
174
|
+
state.now += 1;
|
|
175
|
+
target.pageId = pageId;
|
|
176
|
+
target.usageCount = 0;
|
|
177
|
+
target.loadedAt = state.now;
|
|
178
|
+
target.pinCount += 1;
|
|
179
|
+
(policy.onLoad ?? policy.onAccess)(state, target.frameId);
|
|
180
|
+
emit(`Write page ${pageId} into frame ${target.frameId} — a fresh page, not a disk read.`, {
|
|
181
|
+
stage: 'buffer',
|
|
182
|
+
action: 'write',
|
|
183
|
+
pageId,
|
|
184
|
+
frameId: target.frameId,
|
|
185
|
+
});
|
|
186
|
+
return { frameId: target.frameId, hit: false };
|
|
187
|
+
}
|
|
188
|
+
function unpin(frameId) {
|
|
189
|
+
const f = state.frames.find((x) => x.frameId === frameId);
|
|
190
|
+
if (f && f.pinCount > 0)
|
|
191
|
+
f.pinCount -= 1;
|
|
192
|
+
}
|
|
193
|
+
return { fetch, write, markDirty, unpin, state, stats };
|
|
194
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pages. Real engines use ~4–16 KB pages holding hundreds of rows; we hold a
|
|
3
|
+
* handful, because a page you can see the contents of teaches more than a
|
|
4
|
+
* realistic one you cannot. The disclaimer banner covers the difference.
|
|
5
|
+
*/
|
|
6
|
+
import { compareValues } from "../value.js";
|
|
7
|
+
function compareRows(a, b, columns) {
|
|
8
|
+
for (const column of columns) {
|
|
9
|
+
const cmp = compareValues(a[column] ?? null, b[column] ?? null);
|
|
10
|
+
if (cmp !== 0)
|
|
11
|
+
return cmp;
|
|
12
|
+
}
|
|
13
|
+
return 0;
|
|
14
|
+
}
|
|
15
|
+
export function buildHeap(table, rowsPerPage, firstPageId = 0) {
|
|
16
|
+
const pages = [];
|
|
17
|
+
const firstRowIndex = [];
|
|
18
|
+
// A CLUSTERED (index-organized) table (plan.md §25.4 B3e) is stored in clustering-key order, not insertion order
|
|
19
|
+
// — its whole reason to exist: rows near each other in key order sit near each other on disk, so a range on that
|
|
20
|
+
// key reads pages in sequence instead of wherever insertion happened to leave them.
|
|
21
|
+
const rows = table.clusteredKey ? [...table.rows].sort((a, b) => compareRows(a, b, table.clusteredKey)) : table.rows;
|
|
22
|
+
for (let i = 0; i < rows.length; i += rowsPerPage) {
|
|
23
|
+
firstRowIndex.push(i);
|
|
24
|
+
pages.push({
|
|
25
|
+
pageId: firstPageId + pages.length,
|
|
26
|
+
table: table.name,
|
|
27
|
+
rows: rows.slice(i, i + rowsPerPage),
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
// An empty table still has one (empty) page, so a scan has something to read.
|
|
31
|
+
if (pages.length === 0) {
|
|
32
|
+
firstRowIndex.push(0);
|
|
33
|
+
pages.push({ pageId: firstPageId, table: table.name, rows: [] });
|
|
34
|
+
}
|
|
35
|
+
return { table: table.name, pages, firstRowIndex };
|
|
36
|
+
}
|
|
37
|
+
/** Which page holds the row at `rowIndex`. */
|
|
38
|
+
export function pageOfRow(heap, rowIndex, rowsPerPage) {
|
|
39
|
+
return heap.pages[Math.floor(rowIndex / rowsPerPage)];
|
|
40
|
+
}
|
|
41
|
+
export function pageById(heap, pageId) {
|
|
42
|
+
return heap.pages.find((p) => p.pageId === pageId);
|
|
43
|
+
}
|
|
44
|
+
export function rowsOnPage(page) {
|
|
45
|
+
return page.rows;
|
|
46
|
+
}
|