querylens 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +35 -0
  3. package/dist/bin/querylens.js +208 -0
  4. package/dist/src/engine/bufferTrace.js +67 -0
  5. package/dist/src/engine/datasets.js +139 -0
  6. package/dist/src/engine/exec/delete.js +95 -0
  7. package/dist/src/engine/exec/evaluate.js +174 -0
  8. package/dist/src/engine/exec/index.js +4 -0
  9. package/dist/src/engine/exec/insert.js +75 -0
  10. package/dist/src/engine/exec/operators.js +1290 -0
  11. package/dist/src/engine/exec/run.js +35 -0
  12. package/dist/src/engine/exec/sort.js +171 -0
  13. package/dist/src/engine/exec/unique.js +79 -0
  14. package/dist/src/engine/exec/update.js +124 -0
  15. package/dist/src/engine/exec/writeScan.js +88 -0
  16. package/dist/src/engine/explain.js +114 -0
  17. package/dist/src/engine/index/btree.js +481 -0
  18. package/dist/src/engine/index/build.js +99 -0
  19. package/dist/src/engine/index/bulk.js +107 -0
  20. package/dist/src/engine/index/display.js +38 -0
  21. package/dist/src/engine/index/index.js +9 -0
  22. package/dist/src/engine/index/lookup.js +213 -0
  23. package/dist/src/engine/index/rangeLookup.js +158 -0
  24. package/dist/src/engine/index/spec.js +47 -0
  25. package/dist/src/engine/index/unique.js +31 -0
  26. package/dist/src/engine/index/validate.js +105 -0
  27. package/dist/src/engine/index.js +16 -0
  28. package/dist/src/engine/locks/index.js +1 -0
  29. package/dist/src/engine/locks/lockManager.js +46 -0
  30. package/dist/src/engine/parser/ast.js +77 -0
  31. package/dist/src/engine/parser/display.js +404 -0
  32. package/dist/src/engine/parser/index.js +4 -0
  33. package/dist/src/engine/parser/parser.js +1108 -0
  34. package/dist/src/engine/parser/print.js +74 -0
  35. package/dist/src/engine/parser/tokenizer.js +146 -0
  36. package/dist/src/engine/planner/buildPlan.js +208 -0
  37. package/dist/src/engine/planner/cost.js +582 -0
  38. package/dist/src/engine/planner/emit.js +267 -0
  39. package/dist/src/engine/planner/emitDelete.js +57 -0
  40. package/dist/src/engine/planner/emitUpdate.js +51 -0
  41. package/dist/src/engine/planner/index.js +8 -0
  42. package/dist/src/engine/planner/joinOrder.js +252 -0
  43. package/dist/src/engine/planner/optimize.js +906 -0
  44. package/dist/src/engine/planner/plan.js +445 -0
  45. package/dist/src/engine/predict.js +120 -0
  46. package/dist/src/engine/runQuery.js +393 -0
  47. package/dist/src/engine/seed.js +165 -0
  48. package/dist/src/engine/stats.js +118 -0
  49. package/dist/src/engine/storage/bufferPool.js +194 -0
  50. package/dist/src/engine/storage/index.js +3 -0
  51. package/dist/src/engine/storage/page.js +46 -0
  52. package/dist/src/engine/storage/policy.js +360 -0
  53. package/dist/src/engine/subquery.js +88 -0
  54. package/dist/src/engine/trace.js +17 -0
  55. package/dist/src/engine/types.js +39 -0
  56. package/dist/src/engine/value.js +80 -0
  57. package/dist/src/engine/viewState.js +187 -0
  58. package/package.json +40 -0
@@ -0,0 +1,1108 @@
1
+ /**
2
+ * Recursive-descent parser for the v0 SQL subset (plan.md §4):
3
+ *
4
+ * statement := explain | analyze | select | delete | insert | update
5
+ * explain := EXPLAIN [ANALYZE] select
6
+ * analyze := ANALYZE identifier [';'] EOF
7
+ * select := SELECT [DISTINCT] selectList FROM identifier [join]* [WHERE expr]
8
+ * [GROUP BY identifier (',' identifier)*] [HAVING expr]
9
+ * [ORDER BY identifier [ASC|DESC]] [LIMIT number [OFFSET number]] [';'] EOF
10
+ * join := [INNER | LEFT [OUTER]] JOIN identifier ON qualifiedColumn '=' qualifiedColumn
11
+ * delete := DELETE FROM identifier [WHERE expr] [';'] EOF
12
+ * insert := INSERT INTO identifier [columnList] VALUES valueList [';'] EOF
13
+ * update := UPDATE identifier SET assignment (',' assignment)* [WHERE expr] [';'] EOF
14
+ * assignment := identifier '=' expr
15
+ * columnList := '(' identifier (',' identifier)* ')'
16
+ * valueList := '(' value (',' value)* ')'
17
+ * value := an expr built from literals only — number | string | NULL | TRUE | FALSE,
18
+ * negated or combined with arithmetic, folded to one value at parse time
19
+ * selectList := '*' | selectItem (',' selectItem)*
20
+ * selectItem := aggregateCall | expr [AS identifier]
21
+ * aggregateCall := ('COUNT'|'SUM'|'AVG'|'MIN'|'MAX') '(' ('*' | qualifiedColumn) ')'
22
+ * expr := orExpr
23
+ * orExpr := andExpr (OR andExpr)*
24
+ * andExpr := notExpr (AND notExpr)*
25
+ * notExpr := NOT notExpr | predicate
26
+ * predicate := operand ( compareOp operand
27
+ * | [NOT] BETWEEN operand AND operand
28
+ * | IS [NOT] NULL
29
+ * | [NOT] IN '(' (operand (',' operand)* | inSubquery) ')'
30
+ * | [NOT] LIKE operand )?
31
+ * inSubquery := SELECT identifier FROM identifier [WHERE expr]
32
+ * compareOp := '=' | '<>' | '!=' | '<' | '>' | '<=' | '>='
33
+ * operand := additive
34
+ * additive := multiplicative (('+' | '-') multiplicative)*
35
+ * multiplicative := unary (('*' | '/' | '%') unary)*
36
+ * unary := '-' unary | primary
37
+ * primary := qualifiedColumn | number | string | NULL | TRUE | FALSE | '(' expr ')'
38
+ * | aggregateCall -- only inside HAVING
39
+ * qualifiedColumn := identifier ['.' identifier]
40
+ *
41
+ * `GROUP BY` and `ORDER BY` are mutually exclusive — the parser rejects a
42
+ * query with both, so despite both appearing under `select` above, no query
43
+ * this grammar accepts ever has both clauses at once (see `checkGroupingValidity`).
44
+ *
45
+ * `[join]` is a bare or `INNER JOIN` (an implicit inner join), or `LEFT [OUTER] JOIN` (plan.md §25.4 C3) —
46
+ * `RIGHT`/`FULL`/`CROSS` stay out of scope. Zero or more may chain (plan.md §25.4 C3 slice b), left-deep: each
47
+ * one's `ON` names the table it introduces plus *some* table the query already knows — `FROM`'s, or an earlier
48
+ * `JOIN`'s — never a fresh third table on both sides, and never repeats a table name (no aliasing, so two same-named
49
+ * tables could never be told apart). None of this combines with `GROUP BY`, `ORDER BY`, or an aggregate
50
+ * `selectItem` — each rejected at parse time (`checkJoinValidity`). A `qualifiedColumn`'s `table` half is required
51
+ * on every column reference once any `join` is present (there is no ambiguity resolution — v0 asks the query to say
52
+ * which table), and validated against the tables the query actually names.
53
+ *
54
+ * `inSubquery` (plan.md §25.4 C3 slice c, closing C3) is the one shape `IN` accepts besides a literal list —
55
+ * always exactly one plain column, no `JOIN`/`GROUP BY`/`HAVING`/`ORDER BY`/`LIMIT`/`DISTINCT`/aggregate, and
56
+ * always uncorrelated (its own `WHERE`, if any, can only name a column on its own `FROM` table — the same rule
57
+ * that already applies to a bare non-`JOIN` query). `runQuery.ts`'s `resolveWhereSubqueries` runs it exactly once,
58
+ * before the outer query is planned, replacing it with the literal-list shape everything downstream already knew.
59
+ *
60
+ * Hand-written rather than generated: the error messages are half the teaching
61
+ * value, and they need to name the v0 scope explicitly.
62
+ *
63
+ * `parse` stays the SELECT-only entry point every existing caller (the
64
+ * Playground, the standalone tools, the CLI) already expects. `parseStatement`
65
+ * is the broader one `runQuery.ts` needs now that `DELETE`, `INSERT` and
66
+ * `UPDATE` are real — the four share every helper below; nothing about
67
+ * SELECT parsing changed.
68
+ */
69
+ import { selectItemKey, walkExpr } from "./ast.js";
70
+ import { exprToSql } from "./print.js";
71
+ import { arithmetic, negate } from "../value.js";
72
+ import { TokenizeError, tokenize } from "./tokenizer.js";
73
+ class ParseError extends Error {
74
+ from;
75
+ to;
76
+ hint;
77
+ constructor(message, at, hint) {
78
+ super(message);
79
+ this.name = 'ParseError';
80
+ this.from = at.from;
81
+ this.to = Math.max(at.to, at.from + 1);
82
+ if (hint !== undefined)
83
+ this.hint = hint;
84
+ }
85
+ }
86
+ /** Keywords that tokenise cleanly but land outside the v0 subset. */
87
+ const OUT_OF_SCOPE = {
88
+ RIGHT: 'RIGHT JOIN is not in the v0 subset — write it the other way round: swap the two tables and use a LEFT JOIN.',
89
+ FULL: 'FULL (OUTER) JOIN is not in the v0 subset — an inner or a LEFT JOIN only.',
90
+ CROSS: 'CROSS JOIN is not in the v0 subset — JOIN ... ON needs an equality.',
91
+ USING: 'USING is not in the v0 subset — write the equality out: JOIN t ON a.x = t.y.',
92
+ DISTINCT: 'DISTINCT applies to the whole SELECT list — write `SELECT DISTINCT a, b …`. Inside an aggregate, `COUNT(DISTINCT x)` is not supported yet.',
93
+ UNION: 'UNION is not in the v0 subset.',
94
+ CREATE: 'QueryLens only runs SELECT, DELETE, INSERT and UPDATE. Edit the seed data to change the table.',
95
+ DROP: 'QueryLens only runs SELECT, DELETE, INSERT and UPDATE. Edit the seed data to change the table.',
96
+ };
97
+ const describe = (t) => t.type === 'eof' ? 'the end of the query' : `\`${t.value}\``;
98
+ function makeCursor(tokens) {
99
+ let pos = 0;
100
+ const peek = () => tokens[pos];
101
+ const peekAt = (n) => tokens[Math.min(pos + n, tokens.length - 1)];
102
+ const next = () => tokens[pos++];
103
+ const atEnd = () => peek().type === 'eof';
104
+ function guardScope(token) {
105
+ if (token.type === 'dot') {
106
+ throw new ParseError('Unexpected `.` here.', token, 'A qualified name is `table.column` — exactly one dot, right after the table name.');
107
+ }
108
+ const reason = token.type === 'keyword' ? OUT_OF_SCOPE[token.upper] : undefined;
109
+ if (reason) {
110
+ throw new ParseError(`\`${token.value.toUpperCase()}\` is not supported yet.`, token, reason);
111
+ }
112
+ }
113
+ function expectKeyword(word, what) {
114
+ const token = peek();
115
+ if (token.type === 'keyword' && token.upper === word)
116
+ return next();
117
+ guardScope(token);
118
+ throw new ParseError(`Expected \`${word}\` ${what}, found ${describe(token)}.`, token);
119
+ }
120
+ function expectIdentifier(what) {
121
+ const token = peek();
122
+ if (token.type === 'identifier')
123
+ return next();
124
+ guardScope(token);
125
+ throw new ParseError(`Expected ${what}, found ${describe(token)}.`, token);
126
+ }
127
+ return { allowAggregates: false, peek, peekAt, next, atEnd, guardScope, expectKeyword, expectIdentifier };
128
+ }
129
+ /** The v0 aggregate functions. Not tokenizer keywords — they parse as identifiers, recognised by name here. */
130
+ export const AGGREGATE_FNS = new Set(['COUNT', 'SUM', 'AVG', 'MIN', 'MAX']);
131
+ /**
132
+ * `identifier ['.' identifier]` — `first` is the identifier token already
133
+ * consumed by the caller; this only looks for an optional `.column` after it.
134
+ * Shared by every place a column reference can appear: a `selectItem`, an
135
+ * `operand`, and an aggregate's argument.
136
+ */
137
+ function parseQualifiedColumnTail(c, first) {
138
+ if (c.peek().type !== 'dot') {
139
+ return { name: first.value, from: first.from, to: first.to };
140
+ }
141
+ c.next(); // '.'
142
+ const col = c.expectIdentifier('a column name after the table qualifier');
143
+ return { table: first.value, name: col.value, from: first.from, to: col.to };
144
+ }
145
+ /**
146
+ * One SELECT item: an aggregate call — `COUNT(*)`, `SUM(age)` — or any
147
+ * expression: a plain column, `price * qty`, `age > 30`. An expression may be
148
+ * named with `AS`; a bare column with no alias stays a plain `column` item.
149
+ */
150
+ function parseSelectItem(c) {
151
+ const token = c.peek();
152
+ const upper = token.value.toUpperCase();
153
+ if (token.type === 'identifier' && AGGREGATE_FNS.has(upper) && c.peekAt(1).type === 'lparen') {
154
+ c.next(); // the function name
155
+ c.next(); // '('
156
+ return parseAggregateCall(c, token, upper);
157
+ }
158
+ const expr = parseExpr(c, true);
159
+ let alias;
160
+ let end = expr.span.to;
161
+ if (isKeyword(c.peek(), 'AS')) {
162
+ c.next();
163
+ const name = c.expectIdentifier('a name for the column after AS');
164
+ alias = name.value;
165
+ end = name.to;
166
+ }
167
+ if (expr.kind === 'column' && alias === undefined) {
168
+ return {
169
+ kind: 'column',
170
+ name: expr.name,
171
+ ...(expr.table === undefined ? {} : { table: expr.table }),
172
+ span: expr.span,
173
+ };
174
+ }
175
+ return { kind: 'expression', expr, ...(alias === undefined ? {} : { alias }), span: { from: expr.span.from, to: end } };
176
+ }
177
+ /** The argument and closing parenthesis of an aggregate call, after `FN(` has been consumed. */
178
+ function parseAggregateArgs(c, upper) {
179
+ const argToken = c.peek();
180
+ let arg;
181
+ if (argToken.type === 'star') {
182
+ c.next();
183
+ if (upper !== 'COUNT') {
184
+ throw new ParseError(`\`${upper}(*)\` is not supported — \`*\` only works with COUNT.`, argToken, `Give ${upper} a column to aggregate, e.g. ${upper}(age).`);
185
+ }
186
+ arg = { kind: 'star', span: { from: argToken.from, to: argToken.to } };
187
+ }
188
+ else {
189
+ if (isKeyword(argToken, 'DISTINCT')) {
190
+ throw new ParseError('`COUNT(DISTINCT …)` is not supported yet.', argToken, '`SELECT DISTINCT` is supported; inside an aggregate it is not — count the distinct rows another way.');
191
+ }
192
+ const colToken = c.expectIdentifier(`a column name, or \`*\` for COUNT`);
193
+ const ref = parseQualifiedColumnTail(c, colToken);
194
+ arg = {
195
+ kind: 'column',
196
+ name: ref.name,
197
+ ...(ref.table === undefined ? {} : { table: ref.table }),
198
+ span: { from: ref.from, to: ref.to },
199
+ };
200
+ }
201
+ const close = c.peek();
202
+ if (close.type !== 'rparen') {
203
+ c.guardScope(close);
204
+ throw new ParseError(`Expected \`)\` after the aggregate's argument, found ${describe(close)}.`, close);
205
+ }
206
+ c.next();
207
+ return { arg, close };
208
+ }
209
+ /** The part of an aggregate SELECT item after `FN(`. */
210
+ function parseAggregateCall(c, token, upper) {
211
+ const { arg, close } = parseAggregateArgs(c, upper);
212
+ const after = c.peek();
213
+ if ((after.type === 'operator' && ['+', '-', '/', '%'].includes(after.value)) || after.type === 'star') {
214
+ throw new ParseError('Arithmetic on an aggregate is not supported yet.', after, `Compute \`${upper}(...)\` on its own; do the arithmetic on plain columns.`);
215
+ }
216
+ if (isKeyword(after, 'AS')) {
217
+ throw new ParseError('`AS` is not supported on an aggregate yet.', after, 'An aggregate is named by its own text, like `COUNT(*)`. `AS` works on plain columns and computed ones.');
218
+ }
219
+ return { kind: 'aggregate', fn: upper, arg, span: { from: token.from, to: close.to } };
220
+ }
221
+ function parseSelectList(c) {
222
+ const first = c.peek();
223
+ if (first.type === 'star') {
224
+ c.next();
225
+ return { kind: 'star', span: { from: first.from, to: first.to } };
226
+ }
227
+ const items = [];
228
+ for (;;) {
229
+ items.push(parseSelectItem(c));
230
+ if (c.peek().type !== 'comma')
231
+ break;
232
+ c.next();
233
+ }
234
+ const last = items[items.length - 1];
235
+ return { kind: 'columns', items, span: { from: first.from, to: last.span.to } };
236
+ }
237
+ /** `GROUP BY col [, col ...]`, if present. */
238
+ function parseGroupBy(c) {
239
+ if (!(c.peek().type === 'keyword' && c.peek().upper === 'GROUP'))
240
+ return undefined;
241
+ const groupToken = c.next();
242
+ c.expectKeyword('BY', 'after GROUP');
243
+ const columns = [];
244
+ for (;;) {
245
+ const token = c.expectIdentifier('a column name to group by');
246
+ columns.push({ name: token.value, span: { from: token.from, to: token.to } });
247
+ if (c.peek().type !== 'comma')
248
+ break;
249
+ c.next();
250
+ }
251
+ const last = columns[columns.length - 1];
252
+ return { columns, span: { from: groupToken.from, to: last.span.to } };
253
+ }
254
+ /**
255
+ * Standard SQL's rule: a plain column in the SELECT list must be one the
256
+ * query groups by, or wrapped in an aggregate — otherwise which of the many
257
+ * rows in a group would it come from is not defined. Applies whether or not
258
+ * `GROUP BY` is written at all: `SELECT dept, COUNT(*) FROM t` is invalid for
259
+ * the same reason `SELECT dept, COUNT(*) FROM t GROUP BY status` is.
260
+ */
261
+ function checkGroupingValidity(select, groupBy) {
262
+ if (select.kind === 'star') {
263
+ if (groupBy) {
264
+ throw new ParseError('`SELECT *` cannot be combined with GROUP BY.', groupBy.span, 'Name the columns you want — each one either grouped or wrapped in an aggregate.');
265
+ }
266
+ return;
267
+ }
268
+ const hasAggregate = select.items.some((item) => item.kind === 'aggregate');
269
+ if (!hasAggregate && !groupBy)
270
+ return;
271
+ const computed = select.items.find((item) => item.kind === 'expression');
272
+ if (computed) {
273
+ throw new ParseError(`\`${selectItemKey(computed)}\` is a computed column, which cannot be combined with GROUP BY or an aggregate yet.`, computed.span, 'Aggregate plain columns, or compute the expression in a query without GROUP BY.');
274
+ }
275
+ const groupNames = new Set(groupBy?.columns.map((c) => c.name) ?? []);
276
+ const ungrouped = select.items.find((item) => item.kind === 'column' && !groupNames.has(item.name));
277
+ if (ungrouped) {
278
+ throw new ParseError(`\`${ungrouped.name}\` is not aggregated${groupBy ? ' and is not in the GROUP BY list' : ''}.`, ungrouped.span, groupBy
279
+ ? 'Every plain column in the SELECT list must be one you GROUP BY, or wrapped in an aggregate like COUNT(...).'
280
+ : 'Mixing a plain column with an aggregate needs a GROUP BY — add one, or wrap the column in an aggregate too.');
281
+ }
282
+ }
283
+ /**
284
+ * A HAVING predicate filters *groups*, so what it may mention is what a group
285
+ * has: the grouped columns and aggregates. A plain column that is neither is as
286
+ * undefined here as it is in the SELECT list — which of a group's many rows
287
+ * would it come from? And HAVING only means something where there are groups:
288
+ * with `GROUP BY`, or an aggregate that collapses the table into one.
289
+ */
290
+ function checkHavingValidity(having, select, groupBy, havingToken) {
291
+ let hasAggregate = select.kind === 'columns' && select.items.some((item) => item.kind === 'aggregate');
292
+ const columns = [];
293
+ walkExpr(having, (e) => {
294
+ if (e.kind === 'aggregate')
295
+ hasAggregate = true;
296
+ if (e.kind === 'column')
297
+ columns.push(e);
298
+ });
299
+ if (!groupBy && !hasAggregate) {
300
+ throw new ParseError('HAVING needs GROUP BY or an aggregate.', havingToken, 'HAVING filters groups. To filter individual rows, use WHERE.');
301
+ }
302
+ const groupNames = new Set(groupBy?.columns.map((c) => c.name) ?? []);
303
+ const ungrouped = columns.find((col) => !groupNames.has(col.name));
304
+ if (ungrouped) {
305
+ throw new ParseError(`\`${ungrouped.name}\` is not aggregated and is not in the GROUP BY list.`, ungrouped.span, groupBy
306
+ ? 'A HAVING predicate may test grouped columns and aggregates like COUNT(*) — not a plain column.'
307
+ : 'Without a GROUP BY the whole table is one group, so HAVING can only test aggregates.');
308
+ }
309
+ }
310
+ function parseTableRef(c) {
311
+ const token = c.expectIdentifier('a table name');
312
+ return { name: token.value, span: { from: token.from, to: token.to } };
313
+ }
314
+ /** `[INNER | LEFT [OUTER]] JOIN identifier ON qualifiedColumn '=' qualifiedColumn`, if present (plan.md §25.4 C3). */
315
+ function parseJoin(c) {
316
+ const first = c.peek();
317
+ const isLeft = isKeyword(first, 'LEFT');
318
+ const isInner = isKeyword(first, 'INNER');
319
+ if (!isLeft && !isInner && !isKeyword(first, 'JOIN'))
320
+ return undefined;
321
+ let joinType;
322
+ if (isLeft) {
323
+ c.next();
324
+ if (isKeyword(c.peek(), 'OUTER'))
325
+ c.next();
326
+ joinType = 'left';
327
+ }
328
+ else if (isInner) {
329
+ c.next();
330
+ }
331
+ c.expectKeyword('JOIN', isLeft || isInner ? 'after LEFT/INNER' : 'to begin a join');
332
+ const table = parseTableRef(c);
333
+ c.expectKeyword('ON', 'after the joined table name');
334
+ const on = parseComparison(c);
335
+ checkJoinOn(on);
336
+ return { table, on, ...(joinType ? { joinType } : {}), span: { from: first.from, to: on.span.to } };
337
+ }
338
+ /** `ON` is always a single equality between one qualified column from each side — an equi-join. */
339
+ function checkJoinOn(on) {
340
+ if (on.op !== '=') {
341
+ throw new ParseError('JOIN ... ON only supports an equality between two columns.', on.span, 'v0 joins are equi-joins: `table.column = table.column`.');
342
+ }
343
+ if (on.left.kind !== 'column' || on.left.table === undefined) {
344
+ throw new ParseError('The left side of ON must be a qualified column, like `authors.id`.', on.left.span, 'Every column in a JOIN needs its table name, so it is never ambiguous which side it comes from.');
345
+ }
346
+ if (on.right.kind !== 'column' || on.right.table === undefined) {
347
+ throw new ParseError('The right side of ON must be a qualified column, like `posts.authorId`.', on.right.span, 'Every column in a JOIN needs its table name, so it is never ambiguous which side it comes from.');
348
+ }
349
+ }
350
+ /**
351
+ * Where a `JOIN` sits in the chain (plan.md §25.4 C3 slice b): `checkJoinOn` already guarantees `on`'s two operands
352
+ * are qualified columns, but not *which* tables they name. This checks the one thing that actually depends on the
353
+ * chain built so far — one side must be the table this `JOIN` introduces, and the other must already be in scope
354
+ * (`FROM`, or an earlier `JOIN`) — closing a gap the single-join grammar never had to close itself (a single join's
355
+ * `ON` was never actually checked against `FROM`'s and its own table name; `buildJoinPlan` just assumed it).
356
+ */
357
+ function checkJoinChain(join, known) {
358
+ const { left, right } = join.on;
359
+ if (left.kind !== 'column' || right.kind !== 'column')
360
+ return; // unreachable: checkJoinOn already guarantees this
361
+ const newTable = join.table.name;
362
+ if (known.has(newTable)) {
363
+ throw new ParseError(`\`${newTable}\` is already in this query.`, join.table.span, 'Joining the same table twice needs an alias, which v0 does not have — give the table a different name in a copy of the data instead.');
364
+ }
365
+ const leftIsNew = left.table === newTable;
366
+ const rightIsNew = right.table === newTable;
367
+ if (!leftIsNew && !rightIsNew) {
368
+ throw new ParseError(`Neither side of ON names \`${newTable}\` — the table this JOIN introduces.`, join.on.span, `One side of ON must be a column on \`${newTable}\`; the other, on a table already in the query (${[...known].join(', ')}).`);
369
+ }
370
+ if (leftIsNew && rightIsNew) {
371
+ throw new ParseError(`Both sides of ON name \`${newTable}\` — a JOIN needs one column from the table it introduces and one from a table already in the query.`, join.on.span);
372
+ }
373
+ const otherTable = (leftIsNew ? right.table : left.table);
374
+ if (!known.has(otherTable)) {
375
+ throw new ParseError(`There is no table called \`${otherTable}\` yet in this query.`, join.on.span, `\`${otherTable}\` has to be named in FROM or an earlier JOIN before this ON clause can reference it. Known so far: ${[...known].join(', ')}.`);
376
+ }
377
+ }
378
+ /** Every column reference anywhere in the SELECT list or WHERE, in source order. */
379
+ function collectColumnRefs(select, where) {
380
+ const refs = [];
381
+ if (select.kind === 'columns') {
382
+ for (const item of select.items) {
383
+ if (item.kind === 'column')
384
+ refs.push(item);
385
+ else if (item.kind === 'expression')
386
+ walkExpr(item.expr, (e) => { if (e.kind === 'column')
387
+ refs.push(e); });
388
+ else if (item.arg.kind === 'column')
389
+ refs.push(item.arg);
390
+ }
391
+ }
392
+ if (where)
393
+ walkExpr(where, (e) => { if (e.kind === 'column')
394
+ refs.push(e); });
395
+ return refs;
396
+ }
397
+ /**
398
+ * Two checks, both about column qualifiers: a `join` combined with an
399
+ * aggregate SELECT item is not supported yet (v1 restriction), and every
400
+ * column reference's `table` — required at all once a `join` is present, so
401
+ * no reference can ever be ambiguous between the tables — must actually
402
+ * name a table this query knows about (`FROM` or any `JOIN`).
403
+ */
404
+ function checkJoinValidity(select, joins, where, from) {
405
+ const hasJoin = joins.length > 0;
406
+ if (hasJoin && select.kind === 'columns') {
407
+ const aggregate = select.items.find((item) => item.kind === 'aggregate');
408
+ if (aggregate) {
409
+ throw new ParseError('An aggregate function is not supported yet in a query with JOIN.', aggregate.span, 'JOIN queries return plain rows for now — aggregate the result in a later version.');
410
+ }
411
+ }
412
+ const known = [from.name, ...joins.map((j) => j.table.name)];
413
+ const knownSet = new Set(known);
414
+ for (const ref of collectColumnRefs(select, where)) {
415
+ if (hasJoin && ref.table === undefined) {
416
+ throw new ParseError(`\`${ref.name}\` needs a table name now that this query joins ${String(known.length)} tables.`, ref.span, `Write ${known.map((t) => `\`${t}.${ref.name}\``).join(' or ')}, whichever table it is actually on.`);
417
+ }
418
+ if (!hasJoin && ref.table !== undefined) {
419
+ // A single-table query's rows are never qualified — `authors.id` would
420
+ // look up a key that does not exist, since the row is just `{id, ...}`.
421
+ // Rather than special-case a self-qualification that happens to name
422
+ // the one table right, v0 asks for the plain name instead.
423
+ throw new ParseError(`\`${ref.table}.${ref.name}\` does not need a table name — this query only has one table.`, ref.span, `Write \`${ref.name}\` on its own.`);
424
+ }
425
+ if (hasJoin && ref.table !== undefined && !knownSet.has(ref.table)) {
426
+ throw new ParseError(`There is no table called \`${ref.table}\` in this query.`, ref.span, `This query only knows about ${known.map((t) => `\`${t}\``).join(', ')}.`);
427
+ }
428
+ }
429
+ }
430
+ const isKeyword = (t, ...words) => t.type === 'keyword' && words.includes(t.upper);
431
+ /** A literal, a column, or a parenthesised expression — the leaves every expression is built from. */
432
+ function parsePrimary(c) {
433
+ const token = c.peek();
434
+ if (token.type === 'identifier' && AGGREGATE_FNS.has(token.value.toUpperCase()) && c.peekAt(1).type === 'lparen') {
435
+ if (!c.allowAggregates) {
436
+ throw new ParseError(`\`${token.value.toUpperCase()}(…)\` cannot be used here.`, token, 'An aggregate has no group to read in a WHERE clause — WHERE filters rows *before* they are grouped. Put it in HAVING, which filters the groups.');
437
+ }
438
+ c.next();
439
+ c.next(); // '('
440
+ const upper = token.value.toUpperCase();
441
+ const { arg, close } = parseAggregateArgs(c, upper);
442
+ return { kind: 'aggregate', fn: upper, arg, span: { from: token.from, to: close.to } };
443
+ }
444
+ if (token.type === 'identifier') {
445
+ c.next();
446
+ const ref = parseQualifiedColumnTail(c, token);
447
+ return {
448
+ kind: 'column',
449
+ name: ref.name,
450
+ ...(ref.table === undefined ? {} : { table: ref.table }),
451
+ span: { from: ref.from, to: ref.to },
452
+ };
453
+ }
454
+ if (token.type === 'number') {
455
+ c.next();
456
+ return {
457
+ kind: 'literal',
458
+ value: Number(token.value),
459
+ raw: token.value,
460
+ span: { from: token.from, to: token.to },
461
+ };
462
+ }
463
+ if (token.type === 'string') {
464
+ c.next();
465
+ return {
466
+ kind: 'literal',
467
+ value: token.value,
468
+ raw: `'${token.value}'`,
469
+ span: { from: token.from, to: token.to },
470
+ };
471
+ }
472
+ if (token.type === 'keyword' && ['NULL', 'TRUE', 'FALSE'].includes(token.upper)) {
473
+ c.next();
474
+ const value = token.upper === 'NULL' ? null : token.upper === 'TRUE';
475
+ return {
476
+ kind: 'literal',
477
+ value,
478
+ raw: token.upper,
479
+ span: { from: token.from, to: token.to },
480
+ };
481
+ }
482
+ if (token.type === 'lparen') {
483
+ c.next();
484
+ // Inside parentheses a bare value is fine — `(a + 1) * 2`; the caller decides whether it is a valid predicate.
485
+ const inner = parseExpr(c, true);
486
+ const close = c.peek();
487
+ if (close.type !== 'rparen') {
488
+ c.guardScope(close);
489
+ throw new ParseError(`Expected \`)\` to close the parenthesis, found ${describe(close)}.`, close);
490
+ }
491
+ c.next();
492
+ // The span takes in the parentheses, so the editor highlights what was actually written.
493
+ return { ...inner, span: { from: token.from, to: close.to } };
494
+ }
495
+ if (isKeyword(token, 'SELECT')) {
496
+ throw new ParseError('A subquery is not supported here.', token, 'v2 has no subqueries yet — write the values out, or join the two tables.');
497
+ }
498
+ c.guardScope(token);
499
+ throw new ParseError(`Expected a column name or a value, found ${describe(token)}.`, token);
500
+ }
501
+ const numericLiteral = (e) => e.kind === 'literal' && typeof e.value === 'number';
502
+ /** `-x`. A minus in front of a numeric literal is folded into the literal, so `id = -5` is still an index equality. */
503
+ function parseUnary(c) {
504
+ const token = c.peek();
505
+ if (token.type === 'operator' && token.value === '-') {
506
+ c.next();
507
+ const operand = parseUnary(c);
508
+ const span = { from: token.from, to: operand.span.to };
509
+ if (numericLiteral(operand)) {
510
+ return { kind: 'literal', value: negate(operand.value) ?? 0, raw: operand.raw.startsWith('-') ? operand.raw.slice(1) : `-${operand.raw}`, span };
511
+ }
512
+ return { kind: 'neg', operand, span };
513
+ }
514
+ if (token.type === 'operator' && token.value === '+') {
515
+ c.next(); // unary plus changes nothing
516
+ return parseUnary(c);
517
+ }
518
+ return parsePrimary(c);
519
+ }
520
+ function parseMultiplicative(c) {
521
+ let left = parseUnary(c);
522
+ for (;;) {
523
+ const t = c.peek();
524
+ const isMul = t.type === 'star' || (t.type === 'operator' && (t.value === '/' || t.value === '%'));
525
+ if (!isMul)
526
+ return left;
527
+ c.next();
528
+ const right = parseUnary(c);
529
+ left = { kind: 'arith', op: t.value, left, right, span: { from: left.span.from, to: right.span.to } };
530
+ }
531
+ }
532
+ /** The operand of a comparison or a test: `+ -` over `* / %` over unary minus over a primary. */
533
+ function parseOperand(c) {
534
+ let left = parseMultiplicative(c);
535
+ for (;;) {
536
+ const t = c.peek();
537
+ if (!(t.type === 'operator' && (t.value === '+' || t.value === '-')))
538
+ return left;
539
+ c.next();
540
+ const right = parseMultiplicative(c);
541
+ left = { kind: 'arith', op: t.value, left, right, span: { from: left.span.from, to: right.span.to } };
542
+ }
543
+ }
544
+ const COMPARE_OPS = ['=', '<>', '!=', '<', '>', '<=', '>='];
545
+ /** `left op operand`, given `left` already parsed — shared by `parseComparison` and the plain (non-BETWEEN) half of `parsePredicate`. */
546
+ function finishComparison(c, left) {
547
+ const opToken = c.peek();
548
+ if (opToken.type !== 'operator') {
549
+ c.guardScope(opToken);
550
+ throw new ParseError(`Expected \`=\`, \`<>\`, \`<\`, \`>\`, \`<=\`, \`>=\`, \`BETWEEN\`, \`IN\`, \`LIKE\` or \`IS\`, found ${describe(opToken)}.`, opToken);
551
+ }
552
+ if (!COMPARE_OPS.includes(opToken.value)) {
553
+ throw new ParseError(`\`${opToken.value}\` is not supported yet.`, opToken, 'QueryLens compares with `=`, `<>`, `<`, `>`, `<=`, `>=`, and also has BETWEEN, IN, LIKE and IS NULL.');
554
+ }
555
+ c.next();
556
+ const right = parseOperand(c);
557
+ // `a < b < c` is legal in some engines and means something nobody intends; say what to write instead.
558
+ const again = c.peek();
559
+ if (again.type === 'operator' && COMPARE_OPS.includes(again.value)) {
560
+ throw new ParseError('A comparison compares two things — it cannot be chained.', again, 'Write `a < b AND b < c`, or use `b BETWEEN a AND c`.');
561
+ }
562
+ return {
563
+ kind: 'compare',
564
+ op: (opToken.value === '!=' ? '<>' : opToken.value),
565
+ left,
566
+ right,
567
+ span: { from: left.span.from, to: right.span.to },
568
+ };
569
+ }
570
+ /** Used only by `JOIN ... ON`, which needs exactly one equality — no BETWEEN there. */
571
+ function parseComparison(c) {
572
+ return finishComparison(c, parseOperand(c));
573
+ }
574
+ /** True for a node that is a truth value on its own, so `WHERE (a = 1 OR b = 2)` needs no operator after it. */
575
+ const isBooleanNode = (e) => e.kind !== 'column' && e.kind !== 'literal' && e.kind !== 'arith' && e.kind !== 'neg';
576
+ /**
577
+ * One predicate: an operand followed by whatever it is being tested against.
578
+ *
579
+ * `x BETWEEN lo AND hi` desugars to `x >= lo AND x <= hi` right here, rather
580
+ * than adding a `between` `Expr` kind — the two comparisons that come out are
581
+ * indistinguishable from ones written by hand, so `index-selection`'s range
582
+ * rule, the cost model's selectivity estimate, `evaluate.ts`, and every other
583
+ * conjunct-walking piece of the engine all handle a `BETWEEN` for free.
584
+ * `NOT BETWEEN` is that, wrapped in `not`.
585
+ */
586
+ function parsePredicate(c, bare = false) {
587
+ const left = parseOperand(c);
588
+ const first = c.peek();
589
+ // No test follows. A parenthesised predicate is already a truth value — `(a = 1 OR b = 2)` — and where a
590
+ // *value* is wanted (a SELECT item, a SET value, inside parentheses) a bare column or sum is one too.
591
+ if ((isBooleanNode(left) || bare) && !(first.type === 'operator' || isKeyword(first, 'IS', 'NOT', 'IN', 'LIKE', 'BETWEEN'))) {
592
+ return left;
593
+ }
594
+ if (isKeyword(first, 'IS')) {
595
+ c.next();
596
+ const negated = isKeyword(c.peek(), 'NOT') ? (c.next(), true) : false;
597
+ const nullToken = c.expectKeyword('NULL', 'after IS' + (negated ? ' NOT' : ''));
598
+ return { kind: 'isNull', operand: left, negated, span: { from: left.span.from, to: nullToken.to } };
599
+ }
600
+ const negated = isKeyword(first, 'NOT') && isKeyword(c.peekAt(1), 'BETWEEN', 'IN', 'LIKE');
601
+ if (negated)
602
+ c.next();
603
+ const word = c.peek();
604
+ if (isKeyword(word, 'BETWEEN')) {
605
+ c.next();
606
+ const low = parseOperand(c);
607
+ c.expectKeyword('AND', 'between the two bounds of a BETWEEN range');
608
+ const high = parseOperand(c);
609
+ const between = {
610
+ kind: 'and',
611
+ left: { kind: 'compare', op: '>=', left, right: low, span: { from: left.span.from, to: low.span.to } },
612
+ right: { kind: 'compare', op: '<=', left, right: high, span: { from: left.span.from, to: high.span.to } },
613
+ span: { from: left.span.from, to: high.span.to },
614
+ };
615
+ return negated ? { kind: 'not', operand: between, span: between.span } : between;
616
+ }
617
+ if (isKeyword(word, 'IN')) {
618
+ c.next();
619
+ const open = c.peek();
620
+ if (open.type !== 'lparen') {
621
+ c.guardScope(open);
622
+ throw new ParseError(`Expected \`(\` after IN, found ${describe(open)}.`, open, 'IN takes a list of values: `x IN (1, 2, 3)`.');
623
+ }
624
+ // `IN (SELECT …)` (plan.md §25.4 C3 slice c) vs. `IN (1, 2, 3)` — both start with `(`, so the token right after
625
+ // it is the only thing that tells them apart.
626
+ if (isKeyword(c.peekAt(1), 'SELECT')) {
627
+ c.next(); // '('
628
+ const { subquery, close } = parseInSubquery(c);
629
+ return { kind: 'in', operand: left, items: [], subquery, negated, span: { from: left.span.from, to: close.to } };
630
+ }
631
+ const list = parseParenList(c, 'IN list', parseOperand);
632
+ return { kind: 'in', operand: left, items: list.items, negated, span: { from: left.span.from, to: list.to } };
633
+ }
634
+ if (isKeyword(word, 'LIKE')) {
635
+ c.next();
636
+ const pattern = parseOperand(c);
637
+ return { kind: 'like', operand: left, pattern, negated, span: { from: left.span.from, to: pattern.span.to } };
638
+ }
639
+ return finishComparison(c, left);
640
+ }
641
+ /**
642
+ * `SELECT column FROM table [WHERE expr]` — the whole grammar `IN`'s subquery form accepts (plan.md §25.4 C3
643
+ * slice c), starting right after the `(` `parsePredicate`'s IN branch already consumed. Deliberately much smaller
644
+ * than `parseSelectBody`: a subquery here is always uncorrelated and collapses to a set of one column's values,
645
+ * so `*`/an aggregate/a computed column, `JOIN`, `GROUP BY`, `HAVING`, `ORDER BY`, `LIMIT` and `DISTINCT` are all
646
+ * rejected outright rather than parsed and later restricted — none of them could change *which* values end up in
647
+ * that set. `checkJoinValidity` (with no `JOIN`) does the same "no table name needed" check on `where`'s columns
648
+ * that any other single-table query already gets, which is also what makes this uncorrelated: there is no path
649
+ * for a column here to name a table the outer query knows about.
650
+ */
651
+ function parseInSubquery(c) {
652
+ const selectToken = c.expectKeyword('SELECT', 'to begin the subquery');
653
+ if (isKeyword(c.peek(), 'DISTINCT')) {
654
+ throw new ParseError('DISTINCT is not needed inside an IN subquery.', c.peek(), "IN already treats the subquery's rows as a set of values — a repeat makes no difference either way.");
655
+ }
656
+ const first = c.peek();
657
+ if (first.type === 'star') {
658
+ throw new ParseError('`SELECT *` is not supported inside an IN subquery.', first, 'Name the one column IN needs to compare against.');
659
+ }
660
+ if (first.type === 'identifier' && AGGREGATE_FNS.has(first.value.toUpperCase()) && c.peekAt(1).type === 'lparen') {
661
+ throw new ParseError('An aggregate is not supported yet inside an IN subquery.', first, 'Select the plain column IN needs to compare against instead.');
662
+ }
663
+ const colToken = c.expectIdentifier('the one column IN needs to compare against');
664
+ const tail = parseQualifiedColumnTail(c, colToken);
665
+ const columnItem = {
666
+ kind: 'column',
667
+ name: tail.name,
668
+ ...(tail.table === undefined ? {} : { table: tail.table }),
669
+ span: { from: tail.from, to: tail.to },
670
+ };
671
+ if (c.peek().type === 'comma') {
672
+ throw new ParseError('An IN subquery can only select one column.', c.peek(), 'IN compares against a single value per row — select just the column you need.');
673
+ }
674
+ c.expectKeyword('FROM', "after the subquery's column");
675
+ const from = parseTableRef(c);
676
+ const afterFrom = c.peek();
677
+ if (isKeyword(afterFrom, 'JOIN') || isKeyword(afterFrom, 'LEFT') || isKeyword(afterFrom, 'INNER')) {
678
+ throw new ParseError('JOIN is not supported yet inside an IN subquery.', afterFrom, 'Query the one table IN needs, or join the tables outside the subquery instead.');
679
+ }
680
+ const where = parseOptionalWhere(c);
681
+ const after = c.peek();
682
+ if (isKeyword(after, 'GROUP')) {
683
+ throw new ParseError('GROUP BY is not supported inside an IN subquery.', after, 'Select the plain column IN needs to compare against instead.');
684
+ }
685
+ if (isKeyword(after, 'ORDER')) {
686
+ throw new ParseError('ORDER BY makes no difference inside an IN subquery.', after, 'IN treats the result as a set of values — order never matters here. Drop it.');
687
+ }
688
+ if (isKeyword(after, 'LIMIT')) {
689
+ throw new ParseError('LIMIT is not supported yet inside an IN subquery.', after, 'IN needs every matching value, not just some of them.');
690
+ }
691
+ checkJoinValidity({ kind: 'columns', items: [columnItem], span: columnItem.span }, [], where, from);
692
+ const close = c.peek();
693
+ if (close.type !== 'rparen') {
694
+ c.guardScope(close);
695
+ throw new ParseError(`Expected \`)\` to close the subquery, found ${describe(close)}.`, close);
696
+ }
697
+ c.next();
698
+ return {
699
+ subquery: {
700
+ column: { name: tail.name, span: { from: tail.from, to: tail.to } },
701
+ from,
702
+ ...(where ? { where } : {}),
703
+ span: { from: selectToken.from, to: close.from },
704
+ },
705
+ close,
706
+ };
707
+ }
708
+ function parseNotExpr(c, bare = false) {
709
+ const token = c.peek();
710
+ if (isKeyword(token, 'NOT')) {
711
+ c.next();
712
+ const operand = parseNotExpr(c);
713
+ return { kind: 'not', operand, span: { from: token.from, to: operand.span.to } };
714
+ }
715
+ return parsePredicate(c, bare);
716
+ }
717
+ function parseAndExpr(c, bare = false) {
718
+ let left = parseNotExpr(c, bare);
719
+ while (isKeyword(c.peek(), 'AND')) {
720
+ c.next();
721
+ const right = parseNotExpr(c);
722
+ left = { kind: 'and', left, right, span: { from: left.span.from, to: right.span.to } };
723
+ }
724
+ return left;
725
+ }
726
+ /**
727
+ * `OR` binds loosest, then `AND`, then `NOT` — so `a OR b AND c` is `a OR (b AND c)`.
728
+ * `bare` says a lone value (a column, `a + 1`) is acceptable as the whole
729
+ * expression — true for SELECT items, `SET` values and inside parentheses,
730
+ * false for a `WHERE`, where a bare column is not a predicate.
731
+ */
732
+ function parseExpr(c, bare = false) {
733
+ let left = parseAndExpr(c, bare);
734
+ while (isKeyword(c.peek(), 'OR')) {
735
+ c.next();
736
+ const right = parseAndExpr(c);
737
+ left = { kind: 'or', left, right, span: { from: left.span.from, to: right.span.to } };
738
+ }
739
+ return left;
740
+ }
741
+ /** `WHERE expr`, if present. Shared by every statement form that allows one. */
742
+ function parseOptionalWhere(c) {
743
+ if (c.peek().type === 'keyword' && c.peek().upper === 'WHERE') {
744
+ c.next();
745
+ return parseExpr(c);
746
+ }
747
+ return undefined;
748
+ }
749
+ /** The value of an expression built only from literals, or `undefined` if it depends on anything else. */
750
+ function constantValue(expr) {
751
+ if (expr.kind === 'literal')
752
+ return { value: expr.value };
753
+ if (expr.kind === 'neg') {
754
+ const inner = constantValue(expr.operand);
755
+ return inner && { value: negate(inner.value) };
756
+ }
757
+ if (expr.kind === 'arith') {
758
+ const left = constantValue(expr.left);
759
+ const right = constantValue(expr.right);
760
+ return left && right && { value: arithmetic(expr.op, left.value, right.value) };
761
+ }
762
+ return undefined;
763
+ }
764
+ /**
765
+ * A value for an INSERT's VALUES list: a literal, or arithmetic over literals
766
+ * (`-5`, `2 * 3`), folded to one literal here — the row a statement writes is
767
+ * known before it runs, so there is nothing to compute later. A column
768
+ * reference is rejected: an INSERT has no row to read it from.
769
+ */
770
+ function parseLiteralValue(c) {
771
+ const token = c.peek();
772
+ if (token.type === 'rparen' || token.type === 'comma' || token.type === 'eof') {
773
+ throw new ParseError(`Expected a value, found ${describe(token)}.`, token);
774
+ }
775
+ const expr = parseExpr(c, true);
776
+ const folded = constantValue(expr);
777
+ if (folded) {
778
+ // A lone literal keeps its source text; a folded expression is shown as what it computes.
779
+ const raw = expr.kind === 'literal' ? expr.raw : exprToSql(expr);
780
+ return { kind: 'literal', value: folded.value, raw, span: expr.span };
781
+ }
782
+ let column;
783
+ walkExpr(expr, (e) => { if (!column && e.kind === 'column')
784
+ column = e; });
785
+ if (column) {
786
+ throw new ParseError(`Expected a value, found \`${column.name}\`.`, column.span, 'INSERT takes literal values, or arithmetic on them (`-5`, `2 * 3`) — no column references.');
787
+ }
788
+ c.guardScope(token);
789
+ throw new ParseError(`Expected a value, found ${describe(token)}.`, token);
790
+ }
791
+ /** `( item (',' item)* )` — shared by INSERT's column list and its VALUES list. */
792
+ function parseParenList(c, what, parseItem) {
793
+ const open = c.peek();
794
+ if (open.type !== 'lparen') {
795
+ c.guardScope(open);
796
+ throw new ParseError(`Expected \`(\` before the ${what}, found ${describe(open)}.`, open);
797
+ }
798
+ c.next();
799
+ const items = [parseItem(c)];
800
+ while (c.peek().type === 'comma') {
801
+ c.next();
802
+ items.push(parseItem(c));
803
+ }
804
+ const close = c.peek();
805
+ if (close.type !== 'rparen') {
806
+ c.guardScope(close);
807
+ throw new ParseError(`Expected \`)\` after the ${what}, found ${describe(close)}.`, close);
808
+ }
809
+ c.next();
810
+ return { items, from: open.from, to: close.to };
811
+ }
812
+ /** Trailing `;` (optional) then EOF, or a "trailing junk" error naming why. */
813
+ function expectStatementEnd(c) {
814
+ if (c.peek().type === 'semicolon')
815
+ c.next();
816
+ if (!c.atEnd()) {
817
+ const token = c.peek();
818
+ c.guardScope(token);
819
+ throw new ParseError(`Unexpected ${describe(token)} after the end of the query.`, token);
820
+ }
821
+ }
822
+ /* -------------------------------------------------------------------------- */
823
+ /* SELECT */
824
+ /* -------------------------------------------------------------------------- */
825
+ function parseSelectBody(c) {
826
+ const selectToken = c.expectKeyword('SELECT', 'to begin the query');
827
+ const distinct = isKeyword(c.peek(), 'DISTINCT') ? (c.next(), true) : false;
828
+ const select = parseSelectList(c);
829
+ c.expectKeyword('FROM', 'after the column list');
830
+ const from = parseTableRef(c);
831
+ const joins = [];
832
+ const known = new Set([from.name]);
833
+ for (;;) {
834
+ const join = parseJoin(c);
835
+ if (!join)
836
+ break;
837
+ checkJoinChain(join, known);
838
+ known.add(join.table.name);
839
+ joins.push(join);
840
+ }
841
+ const hasJoin = joins.length > 0;
842
+ const where = parseOptionalWhere(c);
843
+ const groupBy = parseGroupBy(c);
844
+ if (hasJoin && groupBy) {
845
+ throw new ParseError('GROUP BY is not supported yet in a query with JOIN.', groupBy.span, 'JOIN queries return plain rows for now — group the result in a later version.');
846
+ }
847
+ checkGroupingValidity(select, groupBy);
848
+ checkJoinValidity(select, joins, where, from);
849
+ let having;
850
+ if (isKeyword(c.peek(), 'HAVING')) {
851
+ const havingToken = c.next();
852
+ if (hasJoin) {
853
+ throw new ParseError('HAVING is not supported yet in a query with JOIN.', havingToken, 'JOIN queries return plain rows for now — group and filter the result in a later version.');
854
+ }
855
+ c.allowAggregates = true;
856
+ try {
857
+ having = parseExpr(c);
858
+ }
859
+ finally {
860
+ c.allowAggregates = false;
861
+ }
862
+ checkHavingValidity(having, select, groupBy, havingToken);
863
+ }
864
+ let orderBy;
865
+ if (c.peek().type === 'keyword' && c.peek().upper === 'ORDER') {
866
+ if (groupBy) {
867
+ throw new ParseError('ORDER BY with GROUP BY is not supported yet.', c.peek(), 'GROUP BY on its own still works — sort the grouped results yourself for now.');
868
+ }
869
+ if (hasJoin) {
870
+ throw new ParseError('ORDER BY is not supported yet in a query with JOIN.', c.peek(), 'JOIN queries return plain rows for now — sort the result in a later version.');
871
+ }
872
+ const orderToken = c.next();
873
+ c.expectKeyword('BY', 'after ORDER');
874
+ const col = c.expectIdentifier('a column name to sort by');
875
+ // SELECT DISTINCT returns each row once, so a row has no single value of a column that is not in the list
876
+ // to sort it by — PostgreSQL's rule, and the only one that keeps DISTINCT's answer well defined.
877
+ if (distinct && select.kind === 'columns' && !select.items.some((item) => selectItemKey(item) === col.value)) {
878
+ throw new ParseError(`For SELECT DISTINCT, ORDER BY \`${col.value}\` must appear in the select list.`, col, 'Add it to the list, or drop DISTINCT: each distinct row stands for many source rows, so a column that is not selected has no one value to sort it by.');
879
+ }
880
+ let direction = 'asc';
881
+ let end = col.to;
882
+ if (c.peek().type === 'keyword' && (c.peek().upper === 'ASC' || c.peek().upper === 'DESC')) {
883
+ const dir = c.next();
884
+ direction = dir.upper === 'ASC' ? 'asc' : 'desc';
885
+ end = dir.to;
886
+ }
887
+ // Spans the whole clause including the keyword, like LIMIT — unlike
888
+ // WHERE, whose span is its Expr's own and never includes "WHERE" itself.
889
+ orderBy = { column: col.value, direction, span: { from: orderToken.from, to: end } };
890
+ }
891
+ const limitToken = c.peek().type === 'keyword' && c.peek().upper === 'LIMIT' ? c.next() : undefined;
892
+ let limit;
893
+ if (limitToken) {
894
+ const token = c.peek();
895
+ if (token.type !== 'number' || token.value.includes('.')) {
896
+ c.guardScope(token);
897
+ throw new ParseError(`Expected a whole number after LIMIT, found ${describe(token)}.`, token);
898
+ }
899
+ c.next();
900
+ limit = { value: Number(token.value), span: { from: limitToken.from, to: token.to } };
901
+ // `OFFSET m` rides on a LIMIT (plan.md §25.4 B1c) — SQLite's rule; an OFFSET alone is rejected below.
902
+ if (isKeyword(c.peek(), 'OFFSET')) {
903
+ c.next();
904
+ const skip = c.peek();
905
+ if (skip.type !== 'number' || skip.value.includes('.')) {
906
+ c.guardScope(skip);
907
+ throw new ParseError(`Expected a whole number after OFFSET, found ${describe(skip)}.`, skip);
908
+ }
909
+ c.next();
910
+ limit = { ...limit, offset: Number(skip.value), span: { from: limitToken.from, to: skip.to } };
911
+ }
912
+ }
913
+ if (!limit && isKeyword(c.peek(), 'OFFSET')) {
914
+ throw new ParseError('OFFSET needs a LIMIT before it.', c.peek(), 'Write `LIMIT n OFFSET m` — skip m rows, then return the next n.');
915
+ }
916
+ expectStatementEnd(c);
917
+ const end = limit?.span.to ??
918
+ orderBy?.span.to ??
919
+ having?.span.to ??
920
+ groupBy?.span.to ??
921
+ where?.span.to ??
922
+ joins.at(-1)?.span.to ??
923
+ from.span.to;
924
+ return {
925
+ kind: 'select',
926
+ select,
927
+ ...(distinct ? { distinct } : {}),
928
+ from,
929
+ ...(hasJoin ? { joins } : {}),
930
+ ...(where === undefined ? {} : { where }),
931
+ ...(groupBy === undefined ? {} : { groupBy }),
932
+ ...(having === undefined ? {} : { having }),
933
+ ...(orderBy === undefined ? {} : { orderBy }),
934
+ ...(limit === undefined ? {} : { limit }),
935
+ span: { from: selectToken.from, to: end },
936
+ };
937
+ }
938
+ /** The SELECT-only entry point — every existing caller's contract, unchanged. */
939
+ export function parse(sql) {
940
+ return runParse(sql, (c) => {
941
+ if (c.atEnd())
942
+ throw new ParseError('Nothing to run — the query is empty.', c.peek());
943
+ return parseSelectBody(c);
944
+ });
945
+ }
946
+ /* -------------------------------------------------------------------------- */
947
+ /* DELETE */
948
+ /* -------------------------------------------------------------------------- */
949
+ function parseDeleteBody(c) {
950
+ const deleteToken = c.expectKeyword('DELETE', 'to begin a delete');
951
+ c.expectKeyword('FROM', 'after DELETE');
952
+ const from = parseTableRef(c);
953
+ const where = parseOptionalWhere(c);
954
+ expectStatementEnd(c);
955
+ const end = where?.span.to ?? from.span.to;
956
+ return {
957
+ kind: 'delete',
958
+ from,
959
+ ...(where === undefined ? {} : { where }),
960
+ span: { from: deleteToken.from, to: end },
961
+ };
962
+ }
963
+ /* -------------------------------------------------------------------------- */
964
+ /* INSERT */
965
+ /* -------------------------------------------------------------------------- */
966
+ function parseInsertBody(c) {
967
+ const insertToken = c.expectKeyword('INSERT', 'to begin an insert');
968
+ c.expectKeyword('INTO', 'after INSERT');
969
+ const into = parseTableRef(c);
970
+ let columns;
971
+ if (c.peek().type === 'lparen') {
972
+ const parsed = parseParenList(c, 'column list', (cc) => {
973
+ const token = cc.expectIdentifier('a column name');
974
+ return { name: token.value, span: { from: token.from, to: token.to } };
975
+ });
976
+ columns = { names: parsed.items, span: { from: parsed.from, to: parsed.to } };
977
+ }
978
+ c.expectKeyword('VALUES', columns ? 'after the column list' : 'after the table name');
979
+ const parsedValues = parseParenList(c, 'value list', parseLiteralValue);
980
+ const values = { items: parsedValues.items, span: { from: parsedValues.from, to: parsedValues.to } };
981
+ if (columns && columns.names.length !== values.items.length) {
982
+ throw new ParseError(`${columns.names.length} column${columns.names.length === 1 ? '' : 's'} but ${values.items.length} value${values.items.length === 1 ? '' : 's'} — INSERT needs exactly one value per column.`, values.span);
983
+ }
984
+ expectStatementEnd(c);
985
+ return {
986
+ kind: 'insert',
987
+ into,
988
+ ...(columns === undefined ? {} : { columns }),
989
+ values,
990
+ span: { from: insertToken.from, to: values.span.to },
991
+ };
992
+ }
993
+ /* -------------------------------------------------------------------------- */
994
+ /* UPDATE */
995
+ /* -------------------------------------------------------------------------- */
996
+ function parseAssignment(c) {
997
+ const column = c.expectIdentifier('a column name');
998
+ const eq = c.peek();
999
+ if (eq.type !== 'operator' || eq.value !== '=') {
1000
+ c.guardScope(eq);
1001
+ throw new ParseError(`Expected \`=\` after the column name, found ${describe(eq)}.`, eq);
1002
+ }
1003
+ c.next();
1004
+ const value = parseExpr(c, true);
1005
+ return { column: column.value, columnSpan: { from: column.from, to: column.to }, value };
1006
+ }
1007
+ /** `SET assignment (',' assignment)*` — no parentheses, unlike INSERT's lists. */
1008
+ function parseSetClause(c) {
1009
+ const first = parseAssignment(c);
1010
+ const assignments = [first];
1011
+ let last = first;
1012
+ while (c.peek().type === 'comma') {
1013
+ c.next();
1014
+ last = parseAssignment(c);
1015
+ assignments.push(last);
1016
+ }
1017
+ return { assignments, span: { from: first.columnSpan.from, to: last.value.span.to } };
1018
+ }
1019
+ function parseUpdateBody(c) {
1020
+ const updateToken = c.expectKeyword('UPDATE', 'to begin an update');
1021
+ const table = parseTableRef(c);
1022
+ c.expectKeyword('SET', 'after the table name');
1023
+ const set = parseSetClause(c);
1024
+ const where = parseOptionalWhere(c);
1025
+ expectStatementEnd(c);
1026
+ const end = where?.span.to ?? set.span.to;
1027
+ return {
1028
+ kind: 'update',
1029
+ table,
1030
+ set,
1031
+ ...(where === undefined ? {} : { where }),
1032
+ span: { from: updateToken.from, to: end },
1033
+ };
1034
+ }
1035
+ /* -------------------------------------------------------------------------- */
1036
+ /* EXPLAIN */
1037
+ /* -------------------------------------------------------------------------- */
1038
+ function parseExplainBody(c) {
1039
+ const explainToken = c.expectKeyword('EXPLAIN', 'to begin an explain');
1040
+ const analyze = isKeyword(c.peek(), 'ANALYZE') ? (c.next(), true) : false;
1041
+ const next = c.peek();
1042
+ if (next.type === 'eof' || next.type === 'semicolon') {
1043
+ throw new ParseError('Nothing to explain.', next.type === 'eof' ? explainToken : next, 'Follow EXPLAIN with a SELECT: `EXPLAIN SELECT * FROM users WHERE id = 5`.');
1044
+ }
1045
+ if (isKeyword(next, 'DELETE', 'INSERT', 'UPDATE')) {
1046
+ throw new ParseError(`EXPLAIN${analyze ? ' ANALYZE' : ''} of a ${next.upper} is not supported.`, next, analyze
1047
+ ? 'EXPLAIN ANALYZE runs what it explains, and running a write would perform it. EXPLAIN a SELECT.'
1048
+ : 'EXPLAIN takes a SELECT — a write is planned the same way, but not shown yet.');
1049
+ }
1050
+ if (isKeyword(next, 'EXPLAIN')) {
1051
+ throw new ParseError('EXPLAIN cannot be nested.', next, 'One EXPLAIN, then the SELECT.');
1052
+ }
1053
+ const select = parseSelectBody(c);
1054
+ return { kind: 'explain', analyze, select, span: { from: explainToken.from, to: select.span.to } };
1055
+ }
1056
+ /* -------------------------------------------------------------------------- */
1057
+ /* ANALYZE */
1058
+ /* -------------------------------------------------------------------------- */
1059
+ function parseAnalyzeBody(c) {
1060
+ const analyzeToken = c.expectKeyword('ANALYZE', 'to begin an analyze');
1061
+ const table = parseTableRef(c);
1062
+ expectStatementEnd(c);
1063
+ return { kind: 'analyze', table, span: { from: analyzeToken.from, to: table.span.to } };
1064
+ }
1065
+ /**
1066
+ * The broader entry point: SELECT, DELETE, INSERT or UPDATE, dispatched on
1067
+ * the first keyword. `runQuery.ts` uses this one; every SELECT-only caller
1068
+ * keeps using `parse`.
1069
+ */
1070
+ export function parseStatement(sql) {
1071
+ return runParse(sql, (c) => {
1072
+ if (c.atEnd())
1073
+ throw new ParseError('Nothing to run — the query is empty.', c.peek());
1074
+ const first = c.peek();
1075
+ if (first.type === 'keyword' && first.upper === 'EXPLAIN')
1076
+ return parseExplainBody(c);
1077
+ if (first.type === 'keyword' && first.upper === 'ANALYZE')
1078
+ return parseAnalyzeBody(c);
1079
+ if (first.type === 'keyword' && first.upper === 'DELETE')
1080
+ return parseDeleteBody(c);
1081
+ if (first.type === 'keyword' && first.upper === 'INSERT')
1082
+ return parseInsertBody(c);
1083
+ if (first.type === 'keyword' && first.upper === 'UPDATE')
1084
+ return parseUpdateBody(c);
1085
+ return parseSelectBody(c);
1086
+ });
1087
+ }
1088
+ /* -------------------------------------------------------------------------- */
1089
+ function runParse(sql, body) {
1090
+ try {
1091
+ const tokens = tokenize(sql);
1092
+ return { ok: true, ast: body(makeCursor(tokens)) };
1093
+ }
1094
+ catch (err) {
1095
+ if (err instanceof ParseError || err instanceof TokenizeError) {
1096
+ return {
1097
+ ok: false,
1098
+ error: {
1099
+ message: err.message,
1100
+ from: err.from,
1101
+ to: err.to,
1102
+ ...(err.hint === undefined ? {} : { hint: err.hint }),
1103
+ },
1104
+ };
1105
+ }
1106
+ throw err;
1107
+ }
1108
+ }