@memberjunction/sql-dialect 6.1.0-edge.1 → 6.1.0-edge.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,676 @@
1
+ import { PostgreSQLDialect } from './postgresqlDialect.js';
2
+ /**
3
+ * The single PostgreSQL identifier auto-quoting tokenizer.
4
+ *
5
+ * This module exists because there used to be TWO hand-synced copies of it —
6
+ * `PostgreSQLCodeGenProvider.quoteSQLForExecution` (codegen-time SQL) and
7
+ * `PostgreSQLDataProvider.autoQuoteIdentifiers` (every runtime raw-SQL statement) —
8
+ * which drifted apart in both their keyword sets and their quoting predicate.
9
+ * Both now delegate here. Do not reintroduce a private copy in either provider.
10
+ *
11
+ * @module @memberjunction/sql-dialect/postgresqlAutoQuote
12
+ * @see MJ issues #3604, #3590, #3691
13
+ */
14
+ const pgDialect = new PostgreSQLDialect();
15
+ /**
16
+ * SQL keywords, type names and built-in function names that must NOT be quoted.
17
+ *
18
+ * Matched **case-SENSITIVELY against the ALL-CAPS form only** — see
19
+ * {@link AutoQuotePostgreSQLIdentifiers} for why. This set is the union of the two
20
+ * former per-provider sets plus `TYPE`/`DATA` (previously a separate case-sensitive
21
+ * tier, now subsumed by the all-caps-only rule).
22
+ *
23
+ * Adding a word here only suppresses quoting of its ALL-CAPS spelling, so a word that
24
+ * is also an MJ column name (`NAME`, `TEXT`, `VALUES`, `LENGTH`, …) is safe to include:
25
+ * the mixed-case column form still quotes.
26
+ */
27
+ /**
28
+ * Known tokenization limitations (comments containing an apostrophe, and `E'...'` escape
29
+ * strings) are tracked as **MJ #3775** — pre-existing, unchanged by the consolidation, and now
30
+ * fixable in one place rather than two.
31
+ */
32
+ /**
33
+ * NOT the only identifier quoter in MJ, and deliberately so.
34
+ *
35
+ * `PostgreSQLDataProvider` carries a separate, METADATA-DRIVEN quoter
36
+ * (`quoteIdentifiersInSQL` / `quoteFieldNamesInToken`, used by `TransformExternalSQLClause`) which
37
+ * knows the entity's actual field list and can therefore quote precisely, without any keyword
38
+ * heuristic. This module is the fallback for SQL where no entity context exists — hand-authored
39
+ * and stored SQL reaching `ExecuteSQL`.
40
+ *
41
+ * They are not duplicates and this is not an unfinished consolidation: a metadata-driven quoter
42
+ * cannot serve arbitrary SQL, and a heuristic one should not be used where the field list is
43
+ * known. Do not merge them.
44
+ */
45
+ export const PostgreSQLQuotingKeywords = new Set([
46
+ // DML/DDL keywords
47
+ 'SELECT', 'INSERT', 'INTO', 'UPDATE', 'DELETE', 'FROM', 'WHERE', 'AND', 'OR', 'NOT',
48
+ 'JOIN', 'LEFT', 'RIGHT', 'INNER', 'OUTER', 'CROSS', 'FULL', 'ON', 'AS', 'SET',
49
+ 'VALUES', 'NULL', 'LIKE', 'IN', 'EXISTS', 'BETWEEN', 'CASE', 'WHEN', 'THEN',
50
+ 'ELSE', 'END', 'ORDER', 'BY', 'GROUP', 'HAVING', 'LIMIT', 'OFFSET', 'UNION',
51
+ 'ALL', 'CREATE', 'ALTER', 'DROP', 'TABLE', 'INDEX', 'VIEW', 'EXEC', 'DECLARE',
52
+ 'BEGIN', 'COMMIT', 'ROLLBACK', 'TRANSACTION', 'TRUE', 'FALSE', 'IS', 'ASC', 'DESC',
53
+ 'DISTINCT', 'PRIMARY', 'KEY', 'FOREIGN', 'REFERENCES', 'CONSTRAINT', 'DEFAULT',
54
+ 'IF', 'OBJECT', 'TOP', 'WITH', 'OVER', 'PARTITION', 'ROW_NUMBER', 'RANK',
55
+ 'DENSE_RANK', 'LAG', 'LEAD', 'FIRST_VALUE', 'LAST_VALUE', 'ROWS', 'RANGE',
56
+ 'PRECEDING', 'FOLLOWING', 'UNBOUNDED', 'CURRENT', 'ROW', 'FETCH', 'NEXT', 'ONLY',
57
+ 'SCHEMA', 'CASCADE', 'RESTRICT', 'NO', 'ACTION', 'TRIGGER', 'FUNCTION', 'PROCEDURE',
58
+ 'RETURNS', 'RETURN', 'RETURNING', 'EXECUTE', 'CALL', 'RAISE', 'NOTICE', 'EXCEPTION', 'PERFORM',
59
+ 'GRANT', 'REVOKE', 'TO', 'USAGE', 'PRIVILEGES', 'OWNER',
60
+ 'WINDOW', 'FILTER', 'EXCEPT', 'INTERSECT', 'COLLATE', 'TABLESAMPLE',
61
+ // DDL sub-keywords
62
+ 'ADD', 'COLUMN', 'DO', 'RENAME', 'COMMENT', 'UNIQUE', 'CHECK',
63
+ 'CONFLICT', 'NOTHING', 'EXCLUDED', 'ZONE', 'AT', 'FOR', 'EACH', 'OF',
64
+ 'BEFORE', 'AFTER', 'INSTEAD', 'USING', 'ANY', 'SOME',
65
+ 'ENABLE', 'DISABLE', 'GENERATED', 'ALWAYS', 'IDENTITY',
66
+ 'SECURITY', 'DEFINER', 'INVOKER', 'FORCE', 'COPY',
67
+ 'TEMPORARY', 'TEMP', 'RECURSIVE', 'MATERIALIZED', 'CONCURRENTLY',
68
+ // Formerly the separate case-sensitive `_SQL_KEYWORDS_UPPERCASE_ONLY` tier. `TYPE` appears in
69
+ // `ALTER COLUMN <c> TYPE <t>` / `CREATE TYPE`, `DATA` in `ALTER COLUMN <c> SET DATA TYPE <t>` —
70
+ // and both are also common MJ column names. They needed case-sensitive matching before the
71
+ // whole set became all-caps-only; now they are ordinary members of it.
72
+ 'TYPE', 'DATA',
73
+ // PL/pgSQL control flow
74
+ 'NEW', 'OLD', 'FOUND', 'LOOP', 'WHILE', 'EXIT', 'CONTINUE',
75
+ 'ELSIF', 'ELSEIF', 'STRICT',
76
+ // Transaction / constraint control (used by SET CONSTRAINTS ALL IMMEDIATE
77
+ // emitted before ALTER TABLE so deferred trigger events flush). Without
78
+ // CONSTRAINTS / IMMEDIATE / DEFERRED in the keyword set, the tokenizer
79
+ // double-quotes them as identifiers and PG rejects the resulting SQL.
80
+ 'CONSTRAINTS', 'IMMEDIATE', 'DEFERRED', 'SAVEPOINT', 'RELEASE',
81
+ // SQL Server types (still appear in raw SQL fragments at runtime)
82
+ 'NVARCHAR', 'VARCHAR', 'UNIQUEIDENTIFIER', 'DATETIMEOFFSET', 'DATETIME', 'DATETIME2',
83
+ 'BIGINT', 'SMALLINT', 'TINYINT', 'FLOAT', 'REAL', 'DECIMAL', 'NUMERIC', 'MONEY',
84
+ 'BIT', 'INT', 'TEXT', 'NTEXT', 'IMAGE', 'BINARY', 'VARBINARY', 'CHAR', 'NCHAR',
85
+ 'XML', 'GEOGRAPHY', 'GEOMETRY', 'HIERARCHYID', 'SQL_VARIANT', 'SYSNAME',
86
+ 'NEWSEQUENTIALID', 'NEWID', 'GETUTCDATE', 'GETDATE', 'SYSDATETIMEOFFSET',
87
+ 'OBJECT_ID', 'SCOPE_IDENTITY',
88
+ // Aggregate / scalar functions
89
+ 'COUNT', 'MAX', 'MIN', 'SUM', 'AVG', 'ROUND', 'NULLIF', 'ABS', 'CEIL', 'CEILING', 'FLOOR',
90
+ 'SIGN', 'MOD', 'POWER', 'SQRT', 'LOG', 'EXP', 'RANDOM',
91
+ 'COALESCE', 'CAST', 'CONVERT', 'ISNULL',
92
+ 'LEN', 'LENGTH', 'DATALENGTH', 'LOWER', 'UPPER', 'LTRIM', 'RTRIM', 'TRIM', 'REPLACE',
93
+ 'SUBSTRING', 'CHARINDEX', 'PATINDEX', 'STUFF', 'CONCAT', 'FORMAT',
94
+ 'POSITION', 'OVERLAY', 'EXTRACT', 'GREATEST', 'LEAST',
95
+ 'DATEADD', 'DATEDIFF', 'DATEPART', 'YEAR', 'MONTH', 'DAY', 'HOUR', 'MINUTE',
96
+ 'SECOND', 'NOW', 'CURRENT_TIMESTAMP',
97
+ // PostgreSQL specific
98
+ 'BOOLEAN', 'SERIAL', 'BIGSERIAL', 'UUID', 'JSONB', 'JSON', 'ARRAY', 'TIMESTAMPTZ',
99
+ 'TIMESTAMP', 'DATE', 'TIME', 'INTERVAL', 'CITEXT', 'INET', 'MACADDR',
100
+ // PG type names that show up in CAST(... AS T) and ::T expressions in
101
+ // hand-written SQL across the codebase. Without these in the keyword
102
+ // set the tokenizer emits "INTEGER" / "DOUBLE" / "BYTEA" as quoted
103
+ // identifiers and PG rejects them as unknown user-defined types.
104
+ 'INTEGER', 'DOUBLE', 'PRECISION', 'BYTEA', 'OID', 'REGCLASS', 'REGPROC', 'NAME',
105
+ 'GEN_RANDOM_UUID', 'TO_CHAR', 'TO_DATE', 'TO_TIMESTAMP', 'TO_NUMBER',
106
+ 'STRING_AGG', 'ARRAY_AGG', 'UNNEST', 'LATERAL', 'ILIKE',
107
+ 'LANGUAGE', 'PLPGSQL', 'VOLATILE', 'STABLE', 'IMMUTABLE', 'SETOF', 'RECORD',
108
+ 'INOUT', 'OUT', 'VARIADIC', 'PARALLEL', 'SAFE', 'UNSAFE',
109
+ // information_schema column names
110
+ 'TABLE_SCHEMA', 'TABLE_NAME', 'TABLE_CATALOG', 'COLUMN_NAME', 'DATA_TYPE',
111
+ 'IS_NULLABLE', 'COLUMN_DEFAULT', 'CHARACTER_MAXIMUM_LENGTH', 'NUMERIC_PRECISION',
112
+ 'NUMERIC_SCALE', 'ORDINAL_POSITION', 'COLUMN_COMMENT',
113
+ // MJ SQL constructs
114
+ 'INFORMATION_SCHEMA', 'COLUMNS', 'TABLES', 'ROUTINES',
115
+ // PostgreSQL reserved words the shipped baseline emits that were absent from this set.
116
+ // None is an MJ column name, so adding them costs nothing and closes the gap the reverse
117
+ // guard below measures. `SYSTEM` is the `TABLESAMPLE SYSTEM` sampling method; `VALID` and
118
+ // `DEFERRABLE`/`INITIALLY` are constraint attributes; `CURRENT_USER`/`SESSION_USER` are
119
+ // niladic functions written without parentheses, so rule 3 (word before `(`) never sees them.
120
+ 'BOTH', 'CURRENT_USER', 'SESSION_USER', 'DEFERRABLE', 'INITIALLY', 'EXTENSION', 'VALID', 'SYSTEM',
121
+ ]);
122
+ /**
123
+ * Structural words matched **case-INsensitively**, unlike {@link PostgreSQLQuotingKeywords}.
124
+ *
125
+ * These are the only words exempt from the all-caps-only rule, and the exemption is
126
+ * deliberately tiny: it covers SQL fragments authored OUTSIDE this repo that reach
127
+ * `ExecuteSQL` — a saved `UserView.OrderBy` of `Name Desc`, a GraphQL `ExtraFilter` of
128
+ * `A=1 And B=2`. Those work today (keywords used to match case-insensitively) and would
129
+ * otherwise become `"Name" "Desc"` — a syntax error in stored user data this change
130
+ * cannot reach and fix.
131
+ *
132
+ * Every word here is one that can never legally be an MJ column name. That invariant is
133
+ * not a judgement call — `postgresqlAutoQuote.baseline.test.ts` derives every column name
134
+ * from the shipped PostgreSQL baseline DDL and fails the build if any of them ever matches
135
+ * this set case-insensitively. Do not add a word without keeping that guard green.
136
+ */
137
+ export const PostgreSQLStructuralKeywords = new Set([
138
+ // Predicate vocabulary — a saved `OrderBy` of `Name Desc`, an `ExtraFilter` of `A=1 And B=2`.
139
+ 'AND', 'OR', 'NOT', 'IS', 'NULL', 'LIKE', 'ILIKE', 'IN', 'BETWEEN', 'EXISTS',
140
+ 'ASC', 'DESC', 'NULLS', 'FIRST', 'LAST',
141
+ ]);
142
+ // A NOTE ON WHAT IS DELIBERATELY ABSENT FROM THE SET ABOVE.
143
+ //
144
+ // Widening this tier to the whole clause skeleton (`SELECT FROM WHERE JOIN AS ON BY DISTINCT
145
+ // HAVING UNION INTERSECT EXCEPT LIMIT OFFSET CASE WHEN THEN ELSE END`) is tempting, because it
146
+ // would let a stored `MJ: Queries` body written as `Select … From … Where …` parse. It was tried
147
+ // and reverted, for two reasons that only show up when you look at the whole predicate:
148
+ //
149
+ // 1. This tier is matched case-INsensitively and is evaluated BEFORE the dot-qualification
150
+ // rule, so adding a word makes it unquotable *even as `alias.Column`* — the one form the
151
+ // rest of this module treats as an unambiguous identifier. A customer column named `Case`,
152
+ // `End`, `Limit` or `Offset` would fold, which is precisely the defect class this whole
153
+ // change exists to eliminate, reintroduced for 20 words.
154
+ // 2. It does not actually deliver. `Cast(Amount As Decimal)` still fails (the type name quotes),
155
+ // `Insert Into Target (Name)` still fails, `Select Top 10` still fails. Mixed-case SQL needs
156
+ // a real parser, not a bigger denylist — so the widening paid the full price for a fraction
157
+ // of the benefit.
158
+ //
159
+ // Mixed-case SQL keywords beyond the predicate vocabulary are therefore a KNOWN LIMITATION: a
160
+ // stored query body written `Select … From …` does not survive on PostgreSQL. Rewriting it in
161
+ // upper case fixes it, and the error is a loud syntax error rather than silently wrong rows.
162
+ /**
163
+ * Words that are structural **only when followed by a specific next word**, and ordinary
164
+ * quotable identifiers otherwise.
165
+ *
166
+ * This is how the two-word clause forms are covered without giving up column names. `Order` and
167
+ * `Group` are believable columns; `Order By` and `Group By` are not columns at all.
168
+ * `Left`/`Right`/`Full` are believable columns AND scalar functions; `Left Join` is neither.
169
+ * Gating on the following word separates the cases exactly, rather than trading one breakage for
170
+ * another — which is why this tier is safe to extend and {@link PostgreSQLStructuralKeywords}
171
+ * is not.
172
+ *
173
+ * Matched case-insensitively on both sides.
174
+ */
175
+ /**
176
+ * Words that are structural **only when followed by a specific next word**, and ordinary
177
+ * quotable identifiers otherwise.
178
+ *
179
+ * This exists so the clause skeleton can be covered without giving up column names that
180
+ * genuinely occur. `Order` and `Group` are believable columns; `Order By` and `Group By` are
181
+ * not columns at all. `Left`/`Right`/`Full` are believable columns AND scalar functions;
182
+ * `Left Join` is neither. Gating on the following word separates the two cases exactly,
183
+ * rather than trading one breakage for another.
184
+ *
185
+ * Matched case-insensitively on both sides, like {@link PostgreSQLStructuralKeywords}.
186
+ */
187
+ export const PostgreSQLContextualStructuralKeywords = new Map([
188
+ ['ORDER', new Set(['BY'])],
189
+ ['GROUP', new Set(['BY'])],
190
+ ['LEFT', new Set(['JOIN', 'OUTER'])],
191
+ ['RIGHT', new Set(['JOIN', 'OUTER'])],
192
+ ['FULL', new Set(['JOIN', 'OUTER'])],
193
+ ['INNER', new Set(['JOIN'])],
194
+ ['CROSS', new Set(['JOIN'])],
195
+ ['OUTER', new Set(['JOIN'])],
196
+ ]);
197
+ /**
198
+ * Every word that appears on the RIGHT of a pair above — the only words for which the reverse
199
+ * lookup can possibly succeed. Gating on this makes the backwards scan run for three words
200
+ * instead of every word in the statement (measured: it is the whole of the tokenizer's ~2x
201
+ * worst-case slowdown on whitespace-heavy input).
202
+ */
203
+ const PostgreSQLContextualFollowers = new Set([...PostgreSQLContextualStructuralKeywords.values()].flatMap((s) => [...s]));
204
+ /**
205
+ * Quotes mixed-case identifiers in a raw SQL string so PostgreSQL preserves their case.
206
+ *
207
+ * MJ has a great deal of hand-written SQL — in resolvers, engines, dashboard components,
208
+ * and codegen templates — that references PascalCase columns and views unquoted
209
+ * (`FROM __mj.vwAIAgentRuns`, `WHERE TestRun IS NULL`). SQL Server resolves those
210
+ * case-insensitively; PostgreSQL folds an unquoted identifier to lowercase and then fails
211
+ * to find the mixed-case column codegen actually created. This function bridges that gap.
212
+ *
213
+ * ## Keywords are recognized ONLY in ALL-CAPS
214
+ *
215
+ * The keyword set is matched case-SENSITIVELY. This is the crux of the design, and it
216
+ * replaces a case-insensitive denylist that was wrong by construction: the set of SQL
217
+ * keywords and the set of MJ column names overlap (`Name`, `Values`, `Length`, `Precision`,
218
+ * `Log`, `Rank`, `Action`, `Columns`, `Language`, `Month`, `Text` are all real columns AND
219
+ * all keywords/type names/functions). Under case-insensitive matching every name in that
220
+ * intersection was emitted unquoted, folded to lowercase on PG, and failed with
221
+ * `column "..." does not exist` — while SQL Server, being case-insensitive, hid the defect
222
+ * from T-SQL-first authoring entirely.
223
+ *
224
+ * Case-sensitive matching resolves the overlap cleanly because SQL dialects always emit
225
+ * keywords in upper case, so the keyword form and the column form are textually distinct:
226
+ * `TEXT` is the type, `Text` is the column.
227
+ *
228
+ * **An ALL-CAPS word that is NOT a keyword is still an identifier.** `ID` and `URL` are
229
+ * all-caps by nature, so the predicate is `!(isAllUpper && isKeyword)` rather than a pure
230
+ * case rule — a pure case rule would fold them to `id`/`url`.
231
+ *
232
+ * ## The rule
233
+ *
234
+ * Applied to each bare word, in this order (order is load-bearing — the keyword branch
235
+ * runs before the function-call branch so `VALUES(` stays a keyword):
236
+ *
237
+ * 1. ALL-CAPS and in {@link PostgreSQLQuotingKeywords} → keyword/type → do not quote
238
+ * 2. In {@link PostgreSQLStructuralKeywords} (any case) → do not quote
239
+ * 2a. In {@link PostgreSQLContextualStructuralKeywords} AND followed by one of its permitted
240
+ * next words (`Order By`, `Left Join`) → do not quote. Anywhere else, an identifier.
241
+ * 3. Immediately followed by `(` and not preceded by `.` → function call → do not quote
242
+ * 4. All-lowercase, or `__mj_`-prefixed → unchanged → do not quote
243
+ * 5. Starts uppercase, or is preceded by `.` → identifier → QUOTE
244
+ *
245
+ * Rule 3 keeps mixed-case function spellings (`Coalesce(`, `IsNull(`) working now that
246
+ * keyword matching is case-sensitive, and additionally fixes ALL-CAPS functions that were
247
+ * simply missing from the set (`JSONB_BUILD_OBJECT(` used to be quoted, and broke). The
248
+ * `.`-guard exists because MJ creates its stored procedures with quoted mixed-case names,
249
+ * so hand-written `__mj.spCreateFoo(...)` must still be quoted — a dot-qualified callable
250
+ * is far more likely to be an MJ object than a built-in. Rule 5's `.` clause is what makes
251
+ * `__mj.vwAIAgentRuns` work (MJ's `vwXxx` view convention starts lowercase).
252
+ *
253
+ * Known caveat of rule 3: `INSERT INTO Target(Name)` with no space leaves `Target` bare,
254
+ * because a bare word before `(` is indistinguishable from a call. There are no such
255
+ * occurrences in this repo, and the spaced form `INSERT INTO Target (Name)` quotes
256
+ * correctly. Related: `x::Text` now yields `x::"Text"`; write the cast as `::text` or
257
+ * `::TEXT`.
258
+ *
259
+ * ## What is skipped
260
+ *
261
+ * String literals (with `''` escapes and `E`/`N`/`U&` prefixes), `--` line comments,
262
+ * `/* *\/` block comments (nested, as PostgreSQL specifies), dollar-quoted blocks
263
+ * (`$$`/`$tag$`), already-quoted identifiers (with `""` escapes), square-bracketed
264
+ * SQL-Server-style identifiers, `@`-prefixed parameters, and PG positional parameters (`$1`).
265
+ * Skipping already-quoted identifiers is what makes this function **idempotent** —
266
+ * `f(f(x)) === f(x)`.
267
+ *
268
+ * ## Why comments are skipped rather than tolerated
269
+ *
270
+ * Comment handling is not cosmetic. The scanner is a parity machine: an apostrophe inside an
271
+ * unrecognized comment opens a string-literal scan that runs to the next `'`, which is the
272
+ * OPENING quote of a real literal. From there every literal and every code region swaps roles.
273
+ * Against this repository's own shipped query SQL that rewrote literal VALUES —
274
+ * `WHERE "StepType" = 'Prompt'` became `= '"Prompt"'`, and the `jsonb_build_object` keys in
275
+ * `get-conversation-complete.pg.sql` became `'"ID"'` — because line 10 of
276
+ * `calculate-ai-agent-run-cost.pg.sql` contains the word `doesn't` in a comment. Nothing
277
+ * throws; the query simply returns the wrong rows. `postgresqlAutoQuote.shippedQueries.test.ts`
278
+ * pins that whole file set as a no-op so it cannot come back.
279
+ *
280
+ * @param sql Raw SQL text.
281
+ * @returns The same SQL with mixed-case identifiers double-quoted.
282
+ */
283
+ export function AutoQuotePostgreSQLIdentifiers(sql) {
284
+ const result = [];
285
+ let i = 0;
286
+ const len = sql.length;
287
+ while (i < len) {
288
+ const ch = sql[i];
289
+ // Comments come FIRST. Every branch below this one is parity-sensitive, and a comment
290
+ // is the one region whose contents are guaranteed not to be SQL — see the module doc.
291
+ if (ch === '-' && sql[i + 1] === '-') {
292
+ i = skipLineComment(sql, i, len, result);
293
+ continue;
294
+ }
295
+ if (ch === '/' && sql[i + 1] === '*') {
296
+ i = skipBlockComment(sql, i, len, result);
297
+ continue;
298
+ }
299
+ if (ch === '{' && (sql[i + 1] === '{' || sql[i + 1] === '%' || sql[i + 1] === '#')) {
300
+ i = skipTemplatePlaceholder(sql, i, len, result);
301
+ continue;
302
+ }
303
+ if (ch === "'") {
304
+ i = skipSingleQuotedString(sql, i, len, result);
305
+ continue;
306
+ }
307
+ if (ch === '$') {
308
+ i = skipDollarQuotedBlock(sql, i, len, result);
309
+ continue;
310
+ }
311
+ if (ch === '"') {
312
+ i = skipDoubleQuotedIdentifier(sql, i, len, result);
313
+ continue;
314
+ }
315
+ if (ch === '[') {
316
+ i = skipBracketedIdentifier(sql, i, len, result);
317
+ continue;
318
+ }
319
+ if (ch === '@') {
320
+ i = skipAtParameter(sql, i, len, result);
321
+ continue;
322
+ }
323
+ if (/[a-zA-Z_]/.test(ch)) {
324
+ // `E'…'` / `N'…'` / `U&'…'` are one literal, not a word followed by a literal.
325
+ // Tokenizing the prefix as a word emitted `"E"'…'`, which PG rejects.
326
+ const prefix = literalPrefixLength(sql, i, len);
327
+ if (prefix > 0) {
328
+ result.push(sql.substring(i, i + prefix));
329
+ // Only the E-form honours backslash escapes, so only it may treat `\'` as
330
+ // part of the literal rather than its terminator.
331
+ const backslashEscapes = sql[i] === 'E' || sql[i] === 'e';
332
+ i = skipSingleQuotedString(sql, i + prefix, len, result, backslashEscapes);
333
+ continue;
334
+ }
335
+ i = processWord(sql, i, len, result);
336
+ continue;
337
+ }
338
+ result.push(ch);
339
+ i++;
340
+ }
341
+ return result.join('');
342
+ }
343
+ /**
344
+ * Length of a string-literal prefix starting at `start`, or 0 when this is an ordinary word.
345
+ *
346
+ * Recognizes PostgreSQL's `E'…'` (C-style escapes) and `U&'…'` (Unicode escapes), plus T-SQL's
347
+ * `N'…'` which reaches this tokenizer in raw SQL fragments carried over from SQL Server.
348
+ */
349
+ function literalPrefixLength(sql, start, len) {
350
+ const ch = sql[start];
351
+ if ((ch === 'E' || ch === 'e' || ch === 'N' || ch === 'n') && sql[start + 1] === "'")
352
+ return 1;
353
+ if ((ch === 'U' || ch === 'u') && sql[start + 1] === '&' && start + 2 < len && sql[start + 2] === "'")
354
+ return 3;
355
+ return 0;
356
+ }
357
+ /** Closing delimiter for each Nunjucks opening delimiter. */
358
+ const TEMPLATE_DELIMITERS = new Map([
359
+ ['{', '}}'],
360
+ ['%', '%}'],
361
+ ['#', '#}'],
362
+ ]);
363
+ /**
364
+ * Skips a Nunjucks template tag — `{{ … }}`, `{% … %}`, `{# … #}`.
365
+ *
366
+ * `MJ: Queries` bodies are Nunjucks templates. `WHERE cd."ConversationID" = {{ ConversationID |
367
+ * sqlString }}` and `{% if AgentID %}` are both shipped shapes. The names inside the delimiters
368
+ * are PARAMETER names, matched exactly at render time; quoting one to `{{ "ConversationID" |
369
+ * sqlString }}` makes the lookup miss and the parameter never substitutes, so the query loses
370
+ * its filter and returns the unfiltered set. Rendering normally happens before `ExecuteSQL`, so
371
+ * this is defence in depth — but the cost is a dozen lines and the failure it prevents is silent.
372
+ */
373
+ function skipTemplatePlaceholder(sql, start, len, result) {
374
+ const closing = TEMPLATE_DELIMITERS.get(sql[start + 1]);
375
+ const close = closing ? sql.indexOf(closing, start + 2) : -1;
376
+ if (close === -1) {
377
+ // No closing delimiter. Consuming to end-of-input would silently emit every identifier
378
+ // after a stray `{{` unquoted — total, and invisible. Emit the two delimiter characters
379
+ // and resume normal scanning instead, which is the posture the dollar-quote branch
380
+ // already takes for a missing close tag.
381
+ result.push(sql.substring(start, start + 2));
382
+ return start + 2;
383
+ }
384
+ const end = close + 2;
385
+ result.push(sql.substring(start, end));
386
+ return end;
387
+ }
388
+ /** Skips a `--` line comment through to (but not including) its terminating newline. */
389
+ function skipLineComment(sql, start, len, result) {
390
+ let j = start + 2;
391
+ while (j < len && sql[j] !== '\n')
392
+ j++;
393
+ result.push(sql.substring(start, j));
394
+ return j;
395
+ }
396
+ /**
397
+ * Skips a block comment. PostgreSQL block comments NEST, unlike the SQL standard's: an inner
398
+ * open marker must be matched by its own close marker before the outer comment ends, which is
399
+ * why this tracks depth rather than searching for the first close. An unterminated comment
400
+ * consumes the remainder of the input, which is what PostgreSQL itself does.
401
+ */
402
+ function skipBlockComment(sql, start, len, result) {
403
+ let j = start + 2;
404
+ let depth = 1;
405
+ while (j < len && depth > 0) {
406
+ if (sql[j] === '/' && sql[j + 1] === '*') {
407
+ depth++;
408
+ j += 2;
409
+ }
410
+ else if (sql[j] === '*' && sql[j + 1] === '/') {
411
+ depth--;
412
+ j += 2;
413
+ }
414
+ else
415
+ j++;
416
+ }
417
+ result.push(sql.substring(start, j));
418
+ return j;
419
+ }
420
+ /**
421
+ * Skips a single-quoted string literal, handling `''` escapes.
422
+ *
423
+ * @param backslashEscapes true for an `E'…'` literal, where `\'` does not terminate the string.
424
+ */
425
+ function skipSingleQuotedString(sql, start, len, result, backslashEscapes = false) {
426
+ let j = start + 1;
427
+ while (j < len) {
428
+ if (backslashEscapes && sql[j] === '\\' && j + 1 < len) {
429
+ j += 2;
430
+ }
431
+ else if (sql[j] === "'" && j + 1 < len && sql[j + 1] === "'") {
432
+ j += 2;
433
+ }
434
+ else if (sql[j] === "'") {
435
+ j++;
436
+ break;
437
+ }
438
+ else {
439
+ j++;
440
+ }
441
+ }
442
+ result.push(sql.substring(start, j));
443
+ return j;
444
+ }
445
+ /**
446
+ * Skips a dollar-quoted block ($$ ... $$ or $tag$ ... $tag$).
447
+ * Falls through to literal `$` for PG positional params ($1, $2, etc.):
448
+ * those start with `$` followed by a digit then a non-`$` character, so
449
+ * the tag-detection scan finds no closing `$` and we push the lone `$`.
450
+ */
451
+ function skipDollarQuotedBlock(sql, start, len, result) {
452
+ let tagEnd = start + 1;
453
+ if (tagEnd < len && sql[tagEnd] === '$') {
454
+ // Simple $$ tag
455
+ tagEnd = start + 2;
456
+ }
457
+ else {
458
+ // Look for $identifier$ pattern
459
+ while (tagEnd < len && /[a-zA-Z0-9_]/.test(sql[tagEnd]))
460
+ tagEnd++;
461
+ if (tagEnd < len && sql[tagEnd] === '$') {
462
+ tagEnd++;
463
+ }
464
+ else {
465
+ // Not a dollar-quote, just a $ character (e.g. PG positional param $1)
466
+ result.push(sql[start]);
467
+ return start + 1;
468
+ }
469
+ }
470
+ const tag = sql.substring(start, tagEnd);
471
+ const closePos = sql.indexOf(tag, tagEnd);
472
+ if (closePos !== -1) {
473
+ const blockEnd = closePos + tag.length;
474
+ result.push(sql.substring(start, blockEnd));
475
+ return blockEnd;
476
+ }
477
+ // No closing tag found, pass through rest of string
478
+ result.push(sql.substring(start));
479
+ return len;
480
+ }
481
+ /**
482
+ * Skips an already double-quoted identifier — this is what makes the tokenizer idempotent.
483
+ * A `""` pair inside the identifier is an escaped quote, not the close, so stopping at the
484
+ * first `"` would resume mid-identifier and quote the remainder as if it were code.
485
+ */
486
+ function skipDoubleQuotedIdentifier(sql, start, len, result) {
487
+ let j = start + 1;
488
+ while (j < len) {
489
+ if (sql[j] === '"' && sql[j + 1] === '"') {
490
+ j += 2;
491
+ continue;
492
+ }
493
+ if (sql[j] === '"') {
494
+ j++;
495
+ break;
496
+ }
497
+ j++;
498
+ }
499
+ result.push(sql.substring(start, j));
500
+ return j;
501
+ }
502
+ /** Skips a square-bracketed identifier (SQL Server style; passed through verbatim) */
503
+ function skipBracketedIdentifier(sql, start, len, result) {
504
+ let j = start + 1;
505
+ while (j < len && sql[j] !== ']')
506
+ j++;
507
+ if (j < len)
508
+ j++;
509
+ result.push(sql.substring(start, j));
510
+ return j;
511
+ }
512
+ /** Skips an @-prefixed parameter (e.g. @userId for legacy SQL Server-style params) */
513
+ function skipAtParameter(sql, start, len, result) {
514
+ let j = start + 1;
515
+ while (j < len && /[a-zA-Z0-9_]/.test(sql[j]))
516
+ j++;
517
+ result.push(sql.substring(start, j));
518
+ return j;
519
+ }
520
+ /**
521
+ * Processes a word token, quoting it when it is an identifier rather than a keyword
522
+ * or a function name. See {@link AutoQuotePostgreSQLIdentifiers} for the full rule and
523
+ * the reasoning behind each branch; the branch order here matches it exactly.
524
+ */
525
+ function processWord(sql, start, len, result) {
526
+ let j = start + 1;
527
+ while (j < len && /[a-zA-Z0-9_]/.test(sql[j]))
528
+ j++;
529
+ const word = sql.substring(start, j);
530
+ if (isBareWord(word, sql, start, j)) {
531
+ result.push(word);
532
+ }
533
+ else {
534
+ result.push(pgDialect.QuoteIdentifier(word));
535
+ }
536
+ return j;
537
+ }
538
+ /**
539
+ * The next bare word after `from`, skipping whitespace only — an empty string when the next
540
+ * non-space character is not a word character. Deliberately does NOT skip comments: a word
541
+ * separated from its partner by a comment is not the two-word construct being matched.
542
+ */
543
+ function nextWord(sql, from) {
544
+ let i = from;
545
+ while (i < sql.length && /\s/.test(sql[i]))
546
+ i++;
547
+ const start = i;
548
+ while (i < sql.length && /[a-zA-Z0-9_]/.test(sql[i]))
549
+ i++;
550
+ return sql.substring(start, i);
551
+ }
552
+ /**
553
+ * The bare word immediately before `to`, skipping whitespace only — the mirror of
554
+ * {@link nextWord}. Empty when the preceding non-space character is not a word character, which
555
+ * is what keeps `t.Order` and `(Order` from pairing with whatever came before them.
556
+ *
557
+ * Empty in two further cases, both of which exist to make this the true mirror of the forward
558
+ * lookup rather than an approximation of it:
559
+ *
560
+ * - **The word found is itself dot-qualified or already quoted** (`t.Order By`, `"Order" By`).
561
+ * Those keys do not match forward — a dot-qualified key never reaches the contextual tier,
562
+ * and a quoted one is not a word at all — so pairing backwards against them makes the two
563
+ * directions disagree and breaks `f(f(x)) === f(x)`: pass 1 emits `t."Order" By`, pass 2 then
564
+ * sees a quoted key, declines to pair, and emits `t."Order" "By"`.
565
+ * - **The scan would cross into a `--` comment.** `nextWord` gets this for free (it starts on
566
+ * the `-` and returns empty); backwards there is no such guard, so a line comment ending in
567
+ * the word `order` would leave a real column named `By` on the next line unquoted — a fresh
568
+ * instance of exactly the case-folding failure this module exists to prevent.
569
+ */
570
+ function previousWord(sql, to) {
571
+ let i = to - 1;
572
+ while (i >= 0 && /\s/.test(sql[i]))
573
+ i--;
574
+ const end = i + 1;
575
+ while (i >= 0 && /[a-zA-Z0-9_]/.test(sql[i]))
576
+ i--;
577
+ const before = i >= 0 ? sql[i] : '';
578
+ if (before === '.' || before === '"')
579
+ return '';
580
+ // A newline between the found word and `to` means the word sits on an earlier line; if that
581
+ // line has an unclosed `--`, the word is comment text, not SQL.
582
+ const gap = sql.substring(end, to);
583
+ if (gap.includes('\n')) {
584
+ const lineStart = sql.lastIndexOf('\n', i) + 1;
585
+ if (sql.substring(lineStart, end).includes('--'))
586
+ return '';
587
+ }
588
+ return sql.substring(i + 1, end);
589
+ }
590
+ /** True when a word must be emitted verbatim rather than quoted as an identifier. */
591
+ function isBareWord(word, sql, start, end) {
592
+ const precededByDot = start > 0 && sql[start - 1] === '.';
593
+ // 1. Keywords, recognized ONLY in their ALL-CAPS form. An ALL-CAPS word that is not a
594
+ // keyword (`ID`, `URL`) falls through and is quoted as the identifier it is.
595
+ //
596
+ // This tier runs BEFORE the dot rule below, and the ordering is load-bearing in one
597
+ // direction only. Some entries in the keyword set exist *specifically* for their
598
+ // dot-qualified form — `INFORMATION_SCHEMA.COLUMNS`, `.TABLES`, `.ROUTINES` — and the
599
+ // catalog's real relation names are lower case, so quoting the right-hand half emits
600
+ // `INFORMATION_SCHEMA."COLUMNS"`, which does not resolve. CodeGen executes that exact
601
+ // SQL through `qsql()` on every PostgreSQL run (`manage-metadata.ts`, three call sites,
602
+ // two of them unconditional), so quoting it turns a working run into a hard failure.
603
+ // Because this tier is case-SENSITIVE it cannot swallow a mixed-case column: `Case` is
604
+ // not `CASE`, so `e.Case` still reaches the dot rule and quotes.
605
+ if (word === word.toUpperCase() && PostgreSQLQuotingKeywords.has(word)) {
606
+ return true;
607
+ }
608
+ // 2. A dot-qualified word is a MEMBER REFERENCE — `alias.Column`, `schema.object`. No tier
609
+ // BELOW this one may override that, because no SQL dialect has a structural keyword in
610
+ // that position. Checking it ahead of the remaining tiers is what stops any word added to
611
+ // the structural or contextual sets from making a legitimate column unquotable, which is
612
+ // the failure mode this module exists to prevent and the one a widened structural tier
613
+ // reintroduced.
614
+ if (precededByDot) {
615
+ return false;
616
+ }
617
+ // 3. Structural words, any case — the compatibility tier for externally authored SQL.
618
+ if (PostgreSQLStructuralKeywords.has(word.toUpperCase())) {
619
+ return true;
620
+ }
621
+ // 2a. Structural only in front of a specific next word (`Order By`, `Left Join`). Anywhere
622
+ // else these are ordinary identifiers and fall through to be quoted.
623
+ const followers = PostgreSQLContextualStructuralKeywords.get(word.toUpperCase());
624
+ if (followers && followers.has(nextWord(sql, end).toUpperCase())) {
625
+ return true;
626
+ }
627
+ // 2b. …and the FOLLOWER of such a pair is structural too. Leaving `Order` bare while quoting
628
+ // `By` produces `Order "By" "Name"`, which is no more valid than the form it replaced —
629
+ // the pair is one construct and both halves have to be recognized. This is the reverse
630
+ // lookup: is the immediately preceding word a contextual key that permits this one?
631
+ // It chains correctly through `Full Outer Join`, where `Outer` is both a follower of
632
+ // `Full` and a key whose follower is `Join`.
633
+ const upper = word.toUpperCase();
634
+ if (PostgreSQLContextualFollowers.has(upper)) {
635
+ const precedingFollowers = PostgreSQLContextualStructuralKeywords.get(previousWord(sql, start).toUpperCase());
636
+ if (precedingFollowers && precedingFollowers.has(upper)) {
637
+ return true;
638
+ }
639
+ }
640
+ // 3. Function call: immediately followed by `(`, and not dot-qualified (MJ's own stored
641
+ // procedures are dot-qualified with quoted mixed-case names and must stay quoted).
642
+ if (sql[end] === '(' && !precededByDot) {
643
+ return true;
644
+ }
645
+ // 4. Anything all-lowercase is left alone; otherwise a word is an identifier if it starts
646
+ // uppercase or is a member reference after a `.`.
647
+ //
648
+ // There is NO `__mj_` carve-out here, and its removal is the point. Both prior tokenizer
649
+ // copies returned bare for any word starting `__mj_`, evaluated ahead of the dot rule, so
650
+ // even the qualified form escaped:
651
+ //
652
+ // SELECT t.__mj_UpdatedAt FROM __mj.Entity t => ... t.__mj_UpdatedAt ...
653
+ //
654
+ // which folds to `__mj_updatedat` and fails with `column "__mj_updatedat" does not exist`
655
+ // — verbatim the defect this module exists to fix. The clause was also redundant for the
656
+ // purpose it was written for: all-lowercase `__mj_*` names are already covered by
657
+ // `isAllLower` on the line below, so the ONLY words it ever affected were the mixed-case
658
+ // real columns — `__mj_CreatedAt`, `__mj_UpdatedAt`, `__mj_Latitude`, `__mj_Longitude`,
659
+ // `__mj_UDT` — i.e. exactly the five it broke. Do not reintroduce it.
660
+ // The framework columns need one positive rule rather than merely the absence of the old
661
+ // carve-out. Dropping the exemption alone fixes only the dot-qualified form, because
662
+ // `__mj_UpdatedAt` does not START with an uppercase letter — so a bare
663
+ // `SELECT MAX(__mj_UpdatedAt) AS MaxUpdatedAt` (the shape the Query entity's own
664
+ // CacheValidationSQL field description documents) would still fold and fail. A `__mj_` word
665
+ // carrying any uppercase is unambiguously one of MJ's five framework columns: the prefix is
666
+ // MJ's namespace and no SQL keyword lives there, so there is nothing for this to collide
667
+ // with. All-lowercase `__mj_*` names remain bare via `isAllLower` below, unchanged.
668
+ const isFrameworkColumn = word.startsWith('__mj_') && word !== word.toLowerCase();
669
+ if (isFrameworkColumn) {
670
+ return false;
671
+ }
672
+ const isAllLower = word === word.toLowerCase();
673
+ const startsUpper = /^[A-Z]/.test(word);
674
+ return isAllLower || !(startsUpper || precededByDot);
675
+ }
676
+ //# sourceMappingURL=postgresqlAutoQuote.js.map