@impetik/xeer-mcp 0.2.5 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/dist/dev-session.d.ts +1 -1
- package/dist/network-policy.js +1 -1
- package/dist/server.d.ts +1 -1
- package/dist/server.js +4 -4
- package/dist/test-run.d.ts +1 -1
- package/dist/xeer-cli.d.ts +1 -1
- package/package.json +8 -5
- package/vendor/spec/actions.d.ts +1250 -0
- package/vendor/spec/actions.js +805 -0
- package/vendor/spec/admin-sql.d.ts +59 -0
- package/vendor/spec/admin-sql.js +147 -0
- package/vendor/spec/admin.d.ts +110 -0
- package/vendor/spec/admin.js +58 -0
- package/vendor/spec/canonical.d.ts +3 -0
- package/vendor/spec/canonical.js +36 -0
- package/vendor/spec/diagnostics.d.ts +49 -0
- package/vendor/spec/diagnostics.js +500 -0
- package/vendor/spec/docs.d.ts +21 -0
- package/vendor/spec/docs.js +57 -0
- package/vendor/spec/events.d.ts +8 -0
- package/vendor/spec/events.js +21 -0
- package/vendor/spec/identity-keys.d.ts +36 -0
- package/vendor/spec/identity-keys.js +72 -0
- package/vendor/spec/index.d.ts +20 -0
- package/vendor/spec/index.js +20 -0
- package/vendor/spec/local-identity.d.ts +69 -0
- package/vendor/spec/local-identity.js +132 -0
- package/vendor/spec/network-policy.d.ts +16 -0
- package/vendor/spec/network-policy.js +50 -0
- package/vendor/spec/public-assets.d.ts +153 -0
- package/vendor/spec/public-assets.js +166 -0
- package/vendor/spec/review.d.ts +82 -0
- package/vendor/spec/review.js +175 -0
- package/vendor/spec/route.d.ts +43 -0
- package/vendor/spec/route.js +87 -0
- package/vendor/spec/schema-lifecycle.d.ts +6 -0
- package/vendor/spec/schema-lifecycle.js +59 -0
- package/vendor/spec/schema-plan.d.ts +98 -0
- package/vendor/spec/schema-plan.js +194 -0
- package/vendor/spec/schema.d.ts +166 -0
- package/vendor/spec/schema.js +409 -0
- package/vendor/spec/sql-expression.d.ts +91 -0
- package/vendor/spec/sql-expression.js +650 -0
- package/vendor/spec/state-export.d.ts +143 -0
- package/vendor/spec/state-export.js +341 -0
- package/vendor/spec/storage.d.ts +61 -0
- package/vendor/spec/storage.js +120 -0
- package/vendor/spec/table-ddl.d.ts +162 -0
- package/vendor/spec/table-ddl.js +508 -0
- package/vendor/spec/types.d.ts +275 -0
- package/vendor/spec/types.js +11 -0
- package/vendor/spec/value.d.ts +22 -0
- package/vendor/spec/value.js +72 -0
|
@@ -0,0 +1,650 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The SQL text primitives every generated statement is built out of, and the one place an
|
|
3
|
+
* author-written expression is allowed to become SQL.
|
|
4
|
+
*
|
|
5
|
+
* A manifest may carry expressions — a table CHECK, a generated column, a partial index's WHERE, an
|
|
6
|
+
* index over `lower(title)`. There is no way to bind those as parameters: they are DDL, not values,
|
|
7
|
+
* so they reach SQLite as text. That makes them the only part of the manifest that could smuggle
|
|
8
|
+
* syntax into a statement, and the reason this module exists is to make "expression" a closed
|
|
9
|
+
* vocabulary rather than a passthrough.
|
|
10
|
+
*
|
|
11
|
+
* The expression is lexed into tokens and re-emitted from them. Nothing survives that the lexer did
|
|
12
|
+
* not recognize, which buys three things at once:
|
|
13
|
+
*
|
|
14
|
+
* - **Escaping.** A column reference is re-emitted through {@link sqlIdentifier} and a string
|
|
15
|
+
* through {@link sqlStringLiteral}, so quoting is structural rather than a rule someone has to
|
|
16
|
+
* remember. There is no path by which raw author text reaches the statement.
|
|
17
|
+
* - **Scope.** Every bare word is either a keyword, an allowlisted function, or a column of *this*
|
|
18
|
+
* table. A statement separator, a comment, a bind parameter, a qualified `other.column`, and the
|
|
19
|
+
* word `SELECT` all have no token, so a subquery cannot be written at all.
|
|
20
|
+
* - **Determinism.** Whitespace and keyword case are normalized on the way out, so the same
|
|
21
|
+
* expression always produces byte-identical DDL.
|
|
22
|
+
*
|
|
23
|
+
* The function allowlist holds only *deterministic* functions. That is not caution: SQLite refuses a
|
|
24
|
+
* non-deterministic function in a CHECK, a generated column, and an index, so `datetime('now')` in
|
|
25
|
+
* any of those positions is an error at CREATE time. Rejecting it here turns a failure when the
|
|
26
|
+
* schema is applied into a diagnostic when it is written.
|
|
27
|
+
*/
|
|
28
|
+
/** Every function an expression may call. Deterministic only — see the module comment. */
|
|
29
|
+
export const SQL_EXPRESSION_FUNCTIONS = Object.freeze([
|
|
30
|
+
'abs', 'char', 'coalesce', 'concat', 'concat_ws', 'format', 'hex', 'ifnull', 'iif', 'instr',
|
|
31
|
+
'json_array_length', 'json_extract', 'json_quote', 'json_type', 'json_valid', 'length',
|
|
32
|
+
'likelihood', 'likely', 'lower', 'ltrim', 'max', 'min', 'nullif', 'printf', 'quote', 'replace',
|
|
33
|
+
'round', 'rtrim', 'sign', 'substr', 'substring', 'trim', 'typeof', 'unhex', 'unicode', 'unlikely',
|
|
34
|
+
'upper', 'zeroblob',
|
|
35
|
+
]);
|
|
36
|
+
/**
|
|
37
|
+
* How many arguments each allowlisted function takes, as `[minimum, maximum]`.
|
|
38
|
+
*
|
|
39
|
+
* SQLite reports a bad count as `wrong number of arguments to function lower()` when the statement
|
|
40
|
+
* is prepared, which for DDL means when the schema is applied rather than when it is written. The
|
|
41
|
+
* table is exhaustive over {@link SQL_EXPRESSION_FUNCTIONS} and a test holds it that way, because a
|
|
42
|
+
* function reachable without an entry would be a function nobody counts.
|
|
43
|
+
*
|
|
44
|
+
* `max` and `min` are the reason this is not merely tidiness. Both are scalar with two or more
|
|
45
|
+
* arguments and *aggregates* with one, and SQLite answers the aggregate form with `misuse of
|
|
46
|
+
* aggregate function max()` in a CHECK, a generated column and an index alike. Requiring two
|
|
47
|
+
* arguments keeps `max(position, 0)` and turns `max(position)` into a diagnostic.
|
|
48
|
+
*/
|
|
49
|
+
export const SQL_FUNCTION_ARITY = Object.freeze({
|
|
50
|
+
abs: [1, 1],
|
|
51
|
+
char: [0, Number.POSITIVE_INFINITY],
|
|
52
|
+
coalesce: [2, Number.POSITIVE_INFINITY],
|
|
53
|
+
concat: [1, Number.POSITIVE_INFINITY],
|
|
54
|
+
concat_ws: [2, Number.POSITIVE_INFINITY],
|
|
55
|
+
format: [1, Number.POSITIVE_INFINITY],
|
|
56
|
+
hex: [1, 1],
|
|
57
|
+
ifnull: [2, 2],
|
|
58
|
+
iif: [3, 3],
|
|
59
|
+
instr: [2, 2],
|
|
60
|
+
json_array_length: [1, 2],
|
|
61
|
+
json_extract: [2, Number.POSITIVE_INFINITY],
|
|
62
|
+
json_quote: [1, 1],
|
|
63
|
+
json_type: [1, 2],
|
|
64
|
+
json_valid: [1, 2],
|
|
65
|
+
length: [1, 1],
|
|
66
|
+
likelihood: [2, 2],
|
|
67
|
+
likely: [1, 1],
|
|
68
|
+
lower: [1, 1],
|
|
69
|
+
ltrim: [1, 2],
|
|
70
|
+
max: [2, Number.POSITIVE_INFINITY],
|
|
71
|
+
min: [2, Number.POSITIVE_INFINITY],
|
|
72
|
+
nullif: [2, 2],
|
|
73
|
+
printf: [1, Number.POSITIVE_INFINITY],
|
|
74
|
+
quote: [1, 1],
|
|
75
|
+
replace: [3, 3],
|
|
76
|
+
round: [1, 2],
|
|
77
|
+
rtrim: [1, 2],
|
|
78
|
+
sign: [1, 1],
|
|
79
|
+
substr: [2, 3],
|
|
80
|
+
substring: [2, 3],
|
|
81
|
+
trim: [1, 2],
|
|
82
|
+
typeof: [1, 1],
|
|
83
|
+
unhex: [1, 2],
|
|
84
|
+
unicode: [1, 1],
|
|
85
|
+
unlikely: [1, 1],
|
|
86
|
+
upper: [1, 1],
|
|
87
|
+
zeroblob: [1, 1],
|
|
88
|
+
});
|
|
89
|
+
/**
|
|
90
|
+
* Words that are grammar rather than data. A field that happens to be named `end` or `text` is
|
|
91
|
+
* reachable — as `"end"` — but only when written quoted, because an unquoted one is grammar here.
|
|
92
|
+
*/
|
|
93
|
+
const KEYWORDS = new Set([
|
|
94
|
+
'AND', 'AS', 'BETWEEN', 'BINARY', 'BLOB', 'CASE', 'CAST', 'COLLATE', 'ELSE', 'END', 'ESCAPE',
|
|
95
|
+
'FALSE', 'GLOB', 'IN', 'INTEGER', 'IS', 'ISNULL', 'LIKE', 'MATCH', 'NOCASE', 'NOT', 'NOTNULL',
|
|
96
|
+
'NULL', 'NUMERIC', 'OR', 'REAL', 'RTRIM', 'TEXT', 'THEN', 'TRUE', 'WHEN',
|
|
97
|
+
]);
|
|
98
|
+
/**
|
|
99
|
+
* Words that open a statement or a subquery. They would otherwise be read as column references and
|
|
100
|
+
* rejected with a message about an unknown column, which describes the symptom rather than the rule.
|
|
101
|
+
*/
|
|
102
|
+
const BANNED_WORDS = new Set([
|
|
103
|
+
'ALTER', 'ATTACH', 'CREATE', 'DELETE', 'DETACH', 'DROP', 'EXISTS', 'FROM', 'GROUP', 'INSERT',
|
|
104
|
+
'JOIN', 'LIMIT', 'ORDER', 'PRAGMA', 'RAISE', 'SELECT', 'TABLE', 'UNION', 'UPDATE', 'VACUUM',
|
|
105
|
+
'VALUES', 'WHERE', 'WITH',
|
|
106
|
+
]);
|
|
107
|
+
/** Longest first, so `<=` is never lexed as `<` followed by `=`. */
|
|
108
|
+
const OPERATORS = Object.freeze([
|
|
109
|
+
'||', '<<', '>>', '<=', '>=', '<>', '!=', '==',
|
|
110
|
+
'+', '-', '*', '/', '%', '&', '|', '<', '>', '=', '~',
|
|
111
|
+
]);
|
|
112
|
+
/**
|
|
113
|
+
* The longest expression a manifest may declare. An expression is copied verbatim into the schema
|
|
114
|
+
* hash, the DDL, and every diagnostic that quotes it, so it is bounded for the same reason a field
|
|
115
|
+
* name is.
|
|
116
|
+
*/
|
|
117
|
+
export const SQL_EXPRESSION_MAX_LENGTH = 1_000;
|
|
118
|
+
/**
|
|
119
|
+
* The most interior nodes an expression may build.
|
|
120
|
+
*
|
|
121
|
+
* Cloudflare's SQLite refuses an expression tree deeper than 100. Measured in workerd by bisection:
|
|
122
|
+
* a chain of 100 bare terms — that is, 99 `AND` operators — parses, and 101 terms (100 operators)
|
|
123
|
+
* fails with "Expression tree is too large (maximum depth 100)". The limit counts operators rather
|
|
124
|
+
* than terms. The character cap does not stand in for it: 1000 characters buys well over 100 terms,
|
|
125
|
+
* so without a node budget a manifest could pass `xeer check` and then fail to apply.
|
|
126
|
+
*
|
|
127
|
+
* Every interior node of the tree is built by exactly one operator, function call, or grammar
|
|
128
|
+
* keyword, and a root-to-leaf path visits each of its nodes once — so depth never exceeds the count
|
|
129
|
+
* of those tokens. Counting them is therefore a sound upper bound, and one that stays cheap: it
|
|
130
|
+
* needs no parser, only the tokens the lexer already produced.
|
|
131
|
+
*
|
|
132
|
+
* What is deliberately *not* counted is anything that costs no depth. An `IN` list is a single node
|
|
133
|
+
* however many values it holds — 2000 of them parse, and 1000 of them still reject a non-member —
|
|
134
|
+
* so a long enum is free.
|
|
135
|
+
* Parentheses are free too: workerd parsed 200 nested ones without complaint.
|
|
136
|
+
*/
|
|
137
|
+
export const SQL_EXPRESSION_MAX_NODES = 90;
|
|
138
|
+
export class SqlExpressionError extends Error {
|
|
139
|
+
constructor(message) {
|
|
140
|
+
super(message);
|
|
141
|
+
this.name = 'SqlExpressionError';
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
/** SQLite stores text as NUL-terminated internally, so a NUL truncates the value it appears in. */
|
|
145
|
+
const NUL = String.fromCharCode(0);
|
|
146
|
+
function refuseNul(value, what) {
|
|
147
|
+
if (value.includes(NUL))
|
|
148
|
+
throw new SqlExpressionError(`A SQL ${what} may not contain a NUL character.`);
|
|
149
|
+
}
|
|
150
|
+
/** `title` → `"title"`. Doubling the quote is the whole of SQLite's identifier escaping. */
|
|
151
|
+
export function sqlIdentifier(name) {
|
|
152
|
+
refuseNul(name, 'identifier');
|
|
153
|
+
return `"${name.replaceAll('"', '""')}"`;
|
|
154
|
+
}
|
|
155
|
+
/** `O'Brien` → `'O''Brien'`. Doubling the apostrophe is the whole of SQLite's string escaping. */
|
|
156
|
+
export function sqlStringLiteral(value) {
|
|
157
|
+
refuseNul(value, 'string literal');
|
|
158
|
+
return `'${value.replaceAll("'", "''")}'`;
|
|
159
|
+
}
|
|
160
|
+
/** `00ff` → `X'00ff'`. Lowercase hex only, so one byte sequence has exactly one spelling. */
|
|
161
|
+
export function sqlBlobLiteral(hex) {
|
|
162
|
+
if (!/^(?:[0-9a-f]{2})*$/.test(hex)) {
|
|
163
|
+
throw new SqlExpressionError(`A SQL blob literal must be an even number of lowercase hex digits; received ${JSON.stringify(hex)}.`);
|
|
164
|
+
}
|
|
165
|
+
return `X'${hex}'`;
|
|
166
|
+
}
|
|
167
|
+
/** SQL has no spelling for NaN or an infinity, so a number that is not finite has no literal. */
|
|
168
|
+
export function sqlNumberLiteral(value) {
|
|
169
|
+
if (!Number.isFinite(value)) {
|
|
170
|
+
throw new SqlExpressionError(`A SQL number literal must be finite; received ${String(value)}.`);
|
|
171
|
+
}
|
|
172
|
+
return Object.is(value, -0) ? '0' : String(value);
|
|
173
|
+
}
|
|
174
|
+
const IDENTIFIER_START = /[A-Za-z_]/;
|
|
175
|
+
const IDENTIFIER_BODY = /[A-Za-z0-9_]/;
|
|
176
|
+
const NUMBER = /^(?:0[xX][0-9a-fA-F]+|(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?)/;
|
|
177
|
+
function lex(source) {
|
|
178
|
+
const tokens = [];
|
|
179
|
+
let depth = 0;
|
|
180
|
+
let at = 0;
|
|
181
|
+
const fail = (message) => {
|
|
182
|
+
throw new SqlExpressionError(`${message} in expression ${JSON.stringify(source)}.`);
|
|
183
|
+
};
|
|
184
|
+
/** The next character that is not whitespace, used to tell `lower(x)` from a column named `lower`. */
|
|
185
|
+
const peekNonSpace = (from) => {
|
|
186
|
+
let index = from;
|
|
187
|
+
while (index < source.length && /\s/.test(source[index]))
|
|
188
|
+
index += 1;
|
|
189
|
+
return source[index] ?? '';
|
|
190
|
+
};
|
|
191
|
+
while (at < source.length) {
|
|
192
|
+
const char = source[at];
|
|
193
|
+
if (/\s/.test(char)) {
|
|
194
|
+
at += 1;
|
|
195
|
+
continue;
|
|
196
|
+
}
|
|
197
|
+
if (char === '-' && source[at + 1] === '-')
|
|
198
|
+
fail('A comment is not allowed');
|
|
199
|
+
if (char === '/' && source[at + 1] === '*')
|
|
200
|
+
fail('A comment is not allowed');
|
|
201
|
+
if (char === ';')
|
|
202
|
+
fail('A statement separator is not allowed');
|
|
203
|
+
if (char === '?' || char === ':' || char === '@' || char === '$')
|
|
204
|
+
fail('A bind parameter is not allowed');
|
|
205
|
+
if (char === '[' || char === ']' || char === '`')
|
|
206
|
+
fail('Only double quotes may quote an identifier');
|
|
207
|
+
// A dot is a qualified name unless a digit follows it, in which case it opens `.5`.
|
|
208
|
+
if (char === '.' && !/\d/.test(source[at + 1] ?? '')) {
|
|
209
|
+
fail('A qualified name is not allowed; an expression may only read its own table');
|
|
210
|
+
}
|
|
211
|
+
if (char === "'") {
|
|
212
|
+
let value = '';
|
|
213
|
+
let index = at + 1;
|
|
214
|
+
for (;;) {
|
|
215
|
+
if (index >= source.length)
|
|
216
|
+
fail('A string literal is unterminated');
|
|
217
|
+
if (source[index] === "'") {
|
|
218
|
+
if (source[index + 1] !== "'")
|
|
219
|
+
break;
|
|
220
|
+
value += "'";
|
|
221
|
+
index += 2;
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
value += source[index];
|
|
225
|
+
index += 1;
|
|
226
|
+
}
|
|
227
|
+
tokens.push({ kind: 'literal', text: sqlStringLiteral(value) });
|
|
228
|
+
at = index + 1;
|
|
229
|
+
continue;
|
|
230
|
+
}
|
|
231
|
+
if (char === '"') {
|
|
232
|
+
let name = '';
|
|
233
|
+
let index = at + 1;
|
|
234
|
+
for (;;) {
|
|
235
|
+
if (index >= source.length)
|
|
236
|
+
fail('A quoted identifier is unterminated');
|
|
237
|
+
if (source[index] === '"') {
|
|
238
|
+
if (source[index + 1] !== '"')
|
|
239
|
+
break;
|
|
240
|
+
name += '"';
|
|
241
|
+
index += 2;
|
|
242
|
+
continue;
|
|
243
|
+
}
|
|
244
|
+
name += source[index];
|
|
245
|
+
index += 1;
|
|
246
|
+
}
|
|
247
|
+
if (name === '')
|
|
248
|
+
fail('A quoted identifier is empty');
|
|
249
|
+
tokens.push({ kind: 'column', text: name });
|
|
250
|
+
at = index + 1;
|
|
251
|
+
continue;
|
|
252
|
+
}
|
|
253
|
+
// `x'00ff'` is a blob literal, not the column `x` beside a string.
|
|
254
|
+
if ((char === 'x' || char === 'X') && source[at + 1] === "'") {
|
|
255
|
+
const end = source.indexOf("'", at + 2);
|
|
256
|
+
if (end < 0)
|
|
257
|
+
fail('A blob literal is unterminated');
|
|
258
|
+
tokens.push({ kind: 'literal', text: sqlBlobLiteral(source.slice(at + 2, end)) });
|
|
259
|
+
at = end + 1;
|
|
260
|
+
continue;
|
|
261
|
+
}
|
|
262
|
+
if (IDENTIFIER_START.test(char)) {
|
|
263
|
+
let index = at;
|
|
264
|
+
while (index < source.length && IDENTIFIER_BODY.test(source[index]))
|
|
265
|
+
index += 1;
|
|
266
|
+
const word = source.slice(at, index);
|
|
267
|
+
const upper = word.toUpperCase();
|
|
268
|
+
at = index;
|
|
269
|
+
if (peekNonSpace(index) === '(') {
|
|
270
|
+
// A parenthesis after a word usually means a call, but not always: `IN (…)` and `NOT (…)` are
|
|
271
|
+
// grammar, and `CAST(x AS INTEGER)` only wears a call's shape. The allowlist is consulted
|
|
272
|
+
// first so `rtrim(x)` stays a function while a bare `COLLATE RTRIM` stays a collation name.
|
|
273
|
+
if (BANNED_WORDS.has(upper))
|
|
274
|
+
fail(`A subquery or statement keyword (${upper}) is not allowed`);
|
|
275
|
+
if (upper !== 'CAST' && SQL_EXPRESSION_FUNCTIONS.includes(word.toLowerCase())) {
|
|
276
|
+
tokens.push({ kind: 'function', text: word.toLowerCase() });
|
|
277
|
+
continue;
|
|
278
|
+
}
|
|
279
|
+
if (KEYWORDS.has(upper)) {
|
|
280
|
+
tokens.push({ kind: 'keyword', text: upper });
|
|
281
|
+
continue;
|
|
282
|
+
}
|
|
283
|
+
fail(`Function ${JSON.stringify(word)} is not one an expression may call`);
|
|
284
|
+
}
|
|
285
|
+
if (BANNED_WORDS.has(upper))
|
|
286
|
+
fail(`A subquery or statement keyword (${upper}) is not allowed`);
|
|
287
|
+
if (KEYWORDS.has(upper)) {
|
|
288
|
+
tokens.push({ kind: 'keyword', text: upper });
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
tokens.push({ kind: 'column', text: word });
|
|
292
|
+
continue;
|
|
293
|
+
}
|
|
294
|
+
const number = NUMBER.exec(source.slice(at));
|
|
295
|
+
if (number) {
|
|
296
|
+
tokens.push({ kind: 'literal', text: number[0] });
|
|
297
|
+
at += number[0].length;
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
if (char === '(') {
|
|
301
|
+
depth += 1;
|
|
302
|
+
tokens.push({ kind: 'open', text: '(' });
|
|
303
|
+
at += 1;
|
|
304
|
+
continue;
|
|
305
|
+
}
|
|
306
|
+
if (char === ')') {
|
|
307
|
+
depth -= 1;
|
|
308
|
+
if (depth < 0)
|
|
309
|
+
fail('Parentheses are unbalanced');
|
|
310
|
+
tokens.push({ kind: 'close', text: ')' });
|
|
311
|
+
at += 1;
|
|
312
|
+
continue;
|
|
313
|
+
}
|
|
314
|
+
if (char === ',') {
|
|
315
|
+
tokens.push({ kind: 'comma', text: ',' });
|
|
316
|
+
at += 1;
|
|
317
|
+
continue;
|
|
318
|
+
}
|
|
319
|
+
const operator = OPERATORS.find((candidate) => source.startsWith(candidate, at));
|
|
320
|
+
if (operator) {
|
|
321
|
+
tokens.push({ kind: 'operator', text: operator });
|
|
322
|
+
at += operator.length;
|
|
323
|
+
continue;
|
|
324
|
+
}
|
|
325
|
+
fail(`Character ${JSON.stringify(char)} is not allowed`);
|
|
326
|
+
}
|
|
327
|
+
if (depth !== 0)
|
|
328
|
+
throw new SqlExpressionError(`Parentheses are unbalanced in expression ${JSON.stringify(source)}.`);
|
|
329
|
+
if (tokens.length === 0)
|
|
330
|
+
throw new SqlExpressionError('An expression may not be empty.');
|
|
331
|
+
return tokens;
|
|
332
|
+
}
|
|
333
|
+
/** Collation names, which are the only words that may follow COLLATE. */
|
|
334
|
+
const COLLATION_NAMES = new Set(['BINARY', 'NOCASE', 'RTRIM']);
|
|
335
|
+
/** Type names, which are the only words that may follow `CAST(x AS`. */
|
|
336
|
+
const TYPE_NAMES = new Set(['BLOB', 'INTEGER', 'NUMERIC', 'REAL', 'TEXT']);
|
|
337
|
+
/** Keywords that are values in their own right rather than grammar joining two others. */
|
|
338
|
+
const VALUE_KEYWORDS = new Set(['FALSE', 'NULL', 'TRUE']);
|
|
339
|
+
const PREFIX_OPERATORS = new Set(['+', '-', '~']);
|
|
340
|
+
/** The comparisons `NOT` may negate, as in `x NOT LIKE y` and `x NOT BETWEEN a AND b`. */
|
|
341
|
+
const NEGATABLE = new Set(['BETWEEN', 'GLOB', 'IN', 'LIKE', 'MATCH']);
|
|
342
|
+
/**
|
|
343
|
+
* Binding power per infix operator. Only the *relative* order matters here: this validates a token
|
|
344
|
+
* sequence rather than building a tree, and every precedence assignment that is internally
|
|
345
|
+
* consistent accepts the same language. It follows SQLite's own order anyway, so that a reader
|
|
346
|
+
* comparing the two is not asked to hold a second one in their head.
|
|
347
|
+
*/
|
|
348
|
+
const BINARY_PRECEDENCE = Object.freeze({
|
|
349
|
+
OR: 1,
|
|
350
|
+
AND: 2,
|
|
351
|
+
// `NOT` as a prefix sits between AND and comparison; see NOT_PRECEDENCE.
|
|
352
|
+
'=': 4, '==': 4, '!=': 4, '<>': 4, IS: 4, IN: 4, LIKE: 4, GLOB: 4, MATCH: 4, BETWEEN: 4,
|
|
353
|
+
'<': 5, '<=': 5, '>': 5, '>=': 5,
|
|
354
|
+
'&': 6, '|': 6, '<<': 6, '>>': 6,
|
|
355
|
+
'+': 7, '-': 7,
|
|
356
|
+
'*': 8, '/': 8, '%': 8,
|
|
357
|
+
'||': 9,
|
|
358
|
+
});
|
|
359
|
+
const NOT_PRECEDENCE = 3;
|
|
360
|
+
/**
|
|
361
|
+
* Refuse a token sequence that is not a well-formed expression.
|
|
362
|
+
*
|
|
363
|
+
* The lexer decides which tokens may appear; it says nothing about their order, so it accepted
|
|
364
|
+
* `"title" AND`, `lower()` and `title, slug` as readily as a real predicate. Those generate DDL that
|
|
365
|
+
* SQLite then refuses — `near ")": syntax error`, `wrong number of arguments to function lower()` —
|
|
366
|
+
* which turns a repairable diagnostic into a failure when the schema is applied. This is the pass
|
|
367
|
+
* that makes the module's promise true rather than nearly true.
|
|
368
|
+
*
|
|
369
|
+
* It is a validator, not a parser: nothing is built and nothing is evaluated. It walks the tokens by
|
|
370
|
+
* precedence climbing and throws at the first sequence no expression could have produced.
|
|
371
|
+
*/
|
|
372
|
+
function validateGrammar(tokens, source, columns) {
|
|
373
|
+
let at = 0;
|
|
374
|
+
// A declaration rather than an arrow, so its `never` return narrows the code after every call.
|
|
375
|
+
function fail(message) {
|
|
376
|
+
throw new SqlExpressionError(`${message} in expression ${JSON.stringify(source)}.`);
|
|
377
|
+
}
|
|
378
|
+
const peek = (offset = 0) => tokens[at + offset];
|
|
379
|
+
const describe = (token) => (token === undefined ? 'the end of the expression' : JSON.stringify(token.text));
|
|
380
|
+
const at_ = (word, offset = 0) => {
|
|
381
|
+
const token = peek(offset);
|
|
382
|
+
return token?.kind === 'keyword' && token.text === word;
|
|
383
|
+
};
|
|
384
|
+
const expect = (word, what) => {
|
|
385
|
+
if (!at_(word))
|
|
386
|
+
fail(`${what} needs ${word} but found ${describe(peek())}`);
|
|
387
|
+
at += 1;
|
|
388
|
+
};
|
|
389
|
+
const arguments_ = (close) => {
|
|
390
|
+
let count = 0;
|
|
391
|
+
if (peek()?.kind !== 'close') {
|
|
392
|
+
for (;;) {
|
|
393
|
+
parseExpression(0);
|
|
394
|
+
count += 1;
|
|
395
|
+
if (peek()?.kind === 'comma') {
|
|
396
|
+
at += 1;
|
|
397
|
+
continue;
|
|
398
|
+
}
|
|
399
|
+
break;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
if (peek()?.kind !== 'close')
|
|
403
|
+
fail(`${close} is not closed`);
|
|
404
|
+
at += 1;
|
|
405
|
+
return count;
|
|
406
|
+
};
|
|
407
|
+
function parseCase() {
|
|
408
|
+
if (at_('END'))
|
|
409
|
+
fail('CASE needs at least one WHEN branch');
|
|
410
|
+
if (!at_('WHEN'))
|
|
411
|
+
parseExpression(0);
|
|
412
|
+
if (!at_('WHEN'))
|
|
413
|
+
fail(`CASE needs at least one WHEN branch but found ${describe(peek())}`);
|
|
414
|
+
while (at_('WHEN')) {
|
|
415
|
+
at += 1;
|
|
416
|
+
parseExpression(0);
|
|
417
|
+
expect('THEN', 'A CASE branch');
|
|
418
|
+
parseExpression(0);
|
|
419
|
+
}
|
|
420
|
+
if (at_('ELSE')) {
|
|
421
|
+
at += 1;
|
|
422
|
+
parseExpression(0);
|
|
423
|
+
}
|
|
424
|
+
expect('END', 'CASE');
|
|
425
|
+
}
|
|
426
|
+
function parseCast() {
|
|
427
|
+
if (peek()?.kind !== 'open')
|
|
428
|
+
fail(`CAST needs a parenthesized expression but found ${describe(peek())}`);
|
|
429
|
+
at += 1;
|
|
430
|
+
parseExpression(0);
|
|
431
|
+
expect('AS', 'CAST');
|
|
432
|
+
const type = peek();
|
|
433
|
+
if (type?.kind !== 'keyword' || !TYPE_NAMES.has(type.text)) {
|
|
434
|
+
fail(`CAST needs a type name (${[...TYPE_NAMES].join(', ')}) but found ${describe(type)}`);
|
|
435
|
+
}
|
|
436
|
+
at += 1;
|
|
437
|
+
if (peek()?.kind !== 'close')
|
|
438
|
+
fail('A CAST is not closed');
|
|
439
|
+
at += 1;
|
|
440
|
+
}
|
|
441
|
+
function parsePrimary() {
|
|
442
|
+
const token = peek();
|
|
443
|
+
if (token === undefined)
|
|
444
|
+
fail('An expression is incomplete');
|
|
445
|
+
if (token.kind === 'literal' || token.kind === 'column') {
|
|
446
|
+
at += 1;
|
|
447
|
+
return;
|
|
448
|
+
}
|
|
449
|
+
if (token.kind === 'function') {
|
|
450
|
+
at += 1;
|
|
451
|
+
if (peek()?.kind !== 'open')
|
|
452
|
+
fail(`Function ${JSON.stringify(token.text)} needs an argument list`);
|
|
453
|
+
at += 1;
|
|
454
|
+
checkArity(token.text, arguments_(`The argument list of ${JSON.stringify(token.text)}`), source);
|
|
455
|
+
return;
|
|
456
|
+
}
|
|
457
|
+
if (token.kind === 'open') {
|
|
458
|
+
at += 1;
|
|
459
|
+
parseExpression(0);
|
|
460
|
+
if (peek()?.kind === 'comma') {
|
|
461
|
+
fail('A comma may only separate the arguments of a function call or the values of an IN list');
|
|
462
|
+
}
|
|
463
|
+
if (peek()?.kind !== 'close')
|
|
464
|
+
fail(`A parenthesized expression is not closed; found ${describe(peek())}`);
|
|
465
|
+
at += 1;
|
|
466
|
+
return;
|
|
467
|
+
}
|
|
468
|
+
if (token.kind === 'keyword') {
|
|
469
|
+
if (VALUE_KEYWORDS.has(token.text)) {
|
|
470
|
+
at += 1;
|
|
471
|
+
return;
|
|
472
|
+
}
|
|
473
|
+
if (token.text === 'CASE') {
|
|
474
|
+
at += 1;
|
|
475
|
+
parseCase();
|
|
476
|
+
return;
|
|
477
|
+
}
|
|
478
|
+
if (token.text === 'CAST') {
|
|
479
|
+
at += 1;
|
|
480
|
+
parseCast();
|
|
481
|
+
return;
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
if (token.kind === 'operator')
|
|
485
|
+
fail(`Operator ${JSON.stringify(token.text)} is missing an operand`);
|
|
486
|
+
// A field may be named `text` or `end`, which the lexer reads as grammar rather than as a column.
|
|
487
|
+
// Saying only that it cannot start an expression describes the symptom; the author needs the fix.
|
|
488
|
+
if (token.kind === 'keyword') {
|
|
489
|
+
const shadowed = columns.find((column) => column.toLowerCase() === token.text.toLowerCase());
|
|
490
|
+
if (shadowed !== undefined) {
|
|
491
|
+
fail(`${JSON.stringify(shadowed)} is a SQL keyword here, so it reads as grammar rather than as the `
|
|
492
|
+
+ `field of that name; write it quoted as ${JSON.stringify(`"${shadowed}"`)}`);
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
fail(`${describe(token)} cannot start an expression`);
|
|
496
|
+
}
|
|
497
|
+
/** `x ISNULL`, `x NOTNULL`, `x NOT NULL` and `x COLLATE NOCASE` all bind to the value before them. */
|
|
498
|
+
function parsePostfix() {
|
|
499
|
+
for (;;) {
|
|
500
|
+
if (at_('ISNULL') || at_('NOTNULL')) {
|
|
501
|
+
at += 1;
|
|
502
|
+
continue;
|
|
503
|
+
}
|
|
504
|
+
if (at_('NOT') && at_('NULL', 1)) {
|
|
505
|
+
at += 2;
|
|
506
|
+
continue;
|
|
507
|
+
}
|
|
508
|
+
if (at_('COLLATE')) {
|
|
509
|
+
at += 1;
|
|
510
|
+
const name = peek();
|
|
511
|
+
if (name?.kind !== 'keyword' || !COLLATION_NAMES.has(name.text)) {
|
|
512
|
+
fail(`COLLATE needs a collation name (${[...COLLATION_NAMES].join(', ')}) but found ${describe(name)}`);
|
|
513
|
+
}
|
|
514
|
+
at += 1;
|
|
515
|
+
continue;
|
|
516
|
+
}
|
|
517
|
+
return;
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
function parseUnary() {
|
|
521
|
+
const token = peek();
|
|
522
|
+
if (token?.kind === 'operator' && PREFIX_OPERATORS.has(token.text)) {
|
|
523
|
+
at += 1;
|
|
524
|
+
parseUnary();
|
|
525
|
+
return;
|
|
526
|
+
}
|
|
527
|
+
if (at_('NOT')) {
|
|
528
|
+
at += 1;
|
|
529
|
+
parseExpression(NOT_PRECEDENCE);
|
|
530
|
+
return;
|
|
531
|
+
}
|
|
532
|
+
parsePrimary();
|
|
533
|
+
parsePostfix();
|
|
534
|
+
}
|
|
535
|
+
function parseExpression(minimum) {
|
|
536
|
+
parseUnary();
|
|
537
|
+
for (;;) {
|
|
538
|
+
let token = peek();
|
|
539
|
+
let negated = false;
|
|
540
|
+
if (token?.kind === 'keyword' && token.text === 'NOT') {
|
|
541
|
+
const following = peek(1);
|
|
542
|
+
if (following?.kind !== 'keyword' || !NEGATABLE.has(following.text))
|
|
543
|
+
return;
|
|
544
|
+
negated = true;
|
|
545
|
+
token = following;
|
|
546
|
+
}
|
|
547
|
+
if (token === undefined || (token.kind !== 'operator' && token.kind !== 'keyword'))
|
|
548
|
+
return;
|
|
549
|
+
const precedence = BINARY_PRECEDENCE[token.text];
|
|
550
|
+
if (precedence === undefined || precedence < minimum)
|
|
551
|
+
return;
|
|
552
|
+
const word = token.text;
|
|
553
|
+
at += negated ? 2 : 1;
|
|
554
|
+
if (word === 'IS') {
|
|
555
|
+
if (at_('NOT'))
|
|
556
|
+
at += 1;
|
|
557
|
+
parseExpression(precedence + 1);
|
|
558
|
+
continue;
|
|
559
|
+
}
|
|
560
|
+
if (word === 'IN') {
|
|
561
|
+
if (peek()?.kind !== 'open')
|
|
562
|
+
fail(`IN needs a parenthesized list of values but found ${describe(peek())}`);
|
|
563
|
+
at += 1;
|
|
564
|
+
arguments_('An IN list');
|
|
565
|
+
continue;
|
|
566
|
+
}
|
|
567
|
+
if (word === 'BETWEEN') {
|
|
568
|
+
// The bounds are parsed above AND's precedence so the `AND` that separates them is not
|
|
569
|
+
// swallowed as an ordinary conjunction — which is what makes a missing one detectable.
|
|
570
|
+
parseExpression(BINARY_PRECEDENCE.AND + 1);
|
|
571
|
+
if (!at_('AND'))
|
|
572
|
+
fail(`BETWEEN needs AND between its two bounds but found ${describe(peek())}`);
|
|
573
|
+
at += 1;
|
|
574
|
+
parseExpression(BINARY_PRECEDENCE.AND + 1);
|
|
575
|
+
continue;
|
|
576
|
+
}
|
|
577
|
+
parseExpression(precedence + 1);
|
|
578
|
+
if ((word === 'LIKE' || word === 'GLOB' || word === 'MATCH') && at_('ESCAPE')) {
|
|
579
|
+
at += 1;
|
|
580
|
+
parseExpression(precedence + 1);
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
parseExpression(0);
|
|
585
|
+
const trailing = peek();
|
|
586
|
+
if (trailing !== undefined) {
|
|
587
|
+
if (trailing.kind === 'comma') {
|
|
588
|
+
fail('A comma may only separate the arguments of a function call or the values of an IN list');
|
|
589
|
+
}
|
|
590
|
+
fail(`${describe(trailing)} needs an operator before it`);
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
function checkArity(name, count, source) {
|
|
594
|
+
const arity = SQL_FUNCTION_ARITY[name];
|
|
595
|
+
// Unreachable while the allowlist and the arity table agree, which a test enforces.
|
|
596
|
+
if (arity === undefined) {
|
|
597
|
+
throw new SqlExpressionError(`Function ${JSON.stringify(name)} has no declared argument count.`);
|
|
598
|
+
}
|
|
599
|
+
const [minimum, maximum] = arity;
|
|
600
|
+
const plural = (count_) => (count_ === 1 ? '1 argument' : `${count_} arguments`);
|
|
601
|
+
if (count < minimum) {
|
|
602
|
+
throw new SqlExpressionError(`Function ${JSON.stringify(name)} takes at least ${plural(minimum)} but was given ${count} `
|
|
603
|
+
+ `in expression ${JSON.stringify(source)}.`);
|
|
604
|
+
}
|
|
605
|
+
if (count > maximum) {
|
|
606
|
+
throw new SqlExpressionError(`Function ${JSON.stringify(name)} takes at most ${plural(maximum)} but was given ${count} `
|
|
607
|
+
+ `in expression ${JSON.stringify(source)}.`);
|
|
608
|
+
}
|
|
609
|
+
}
|
|
610
|
+
/** Single spaces everywhere except where SQL reads better — and always the same way. */
|
|
611
|
+
function joinTokens(tokens) {
|
|
612
|
+
return tokens.reduce((sql, token, index) => {
|
|
613
|
+
if (index === 0)
|
|
614
|
+
return token.text;
|
|
615
|
+
const previous = tokens[index - 1];
|
|
616
|
+
const tight = previous.kind === 'open'
|
|
617
|
+
|| token.kind === 'close' || token.kind === 'comma'
|
|
618
|
+
|| (token.kind === 'open' && (previous.kind === 'function'
|
|
619
|
+
|| (previous.kind === 'keyword' && previous.text === 'CAST')));
|
|
620
|
+
return tight ? sql + token.text : `${sql} ${token.text}`;
|
|
621
|
+
}, '');
|
|
622
|
+
}
|
|
623
|
+
/**
|
|
624
|
+
* Validate an author-written expression against the columns it is allowed to read, and return the
|
|
625
|
+
* canonical SQL for it. Throws {@link SqlExpressionError} with a message naming the rule it broke.
|
|
626
|
+
*/
|
|
627
|
+
export function compileSqlExpression(source, columns) {
|
|
628
|
+
if (source.length > SQL_EXPRESSION_MAX_LENGTH) {
|
|
629
|
+
throw new SqlExpressionError(`An expression may be at most ${SQL_EXPRESSION_MAX_LENGTH} characters; received ${source.length}, which is too long.`);
|
|
630
|
+
}
|
|
631
|
+
const tokens = lex(source);
|
|
632
|
+
const nodes = tokens.filter((token) => token.kind === 'operator' || token.kind === 'function' || token.kind === 'keyword').length;
|
|
633
|
+
if (nodes > SQL_EXPRESSION_MAX_NODES) {
|
|
634
|
+
throw new SqlExpressionError(`Expression ${JSON.stringify(source)} is too deeply nested: it builds ${nodes} operations and the limit is `
|
|
635
|
+
+ `${SQL_EXPRESSION_MAX_NODES}, above which SQLite refuses to parse it. Split it into separate constraints.`);
|
|
636
|
+
}
|
|
637
|
+
validateGrammar(tokens, source, columns);
|
|
638
|
+
const declared = new Set(columns);
|
|
639
|
+
const referenced = new Set();
|
|
640
|
+
const emitted = tokens.map((token) => {
|
|
641
|
+
if (token.kind !== 'column')
|
|
642
|
+
return token;
|
|
643
|
+
if (!declared.has(token.text)) {
|
|
644
|
+
throw new SqlExpressionError(`Expression ${JSON.stringify(source)} references unknown column ${JSON.stringify(token.text)}.`);
|
|
645
|
+
}
|
|
646
|
+
referenced.add(token.text);
|
|
647
|
+
return { kind: 'column', text: sqlIdentifier(token.text) };
|
|
648
|
+
});
|
|
649
|
+
return { sql: joinTokens(emitted), columns: [...referenced].sort() };
|
|
650
|
+
}
|