@xdbml/parse 0.1.0-poc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/parser.js ADDED
@@ -0,0 +1,2083 @@
1
+ /**
2
+ * xDBML parser.
3
+ *
4
+ * Hand-written recursive-descent. Reads the token stream produced by the
5
+ * Lexer and emits the AST defined in ./ast.ts. Pragmatic and intentionally
6
+ * permissive at the parse level: several spec constraints (tuple position
7
+ * contiguity, named-type vs. builtin shadowing, ref-path array-crossing,
8
+ * polymorphic alternative selectors in paths) are deferred to a future
9
+ * semantic-analysis pass. The grammar test cases the parser passes are
10
+ * the official xDBML example files in /examples.
11
+ */
12
+ import { TokenKind, tokenize, } from "./lexer.js";
13
+ import { resolveImport, classifyModuleSource, ModuleSourceError } from "./module-resolver.js";
14
+ export class ParseError extends Error {
15
+ position;
16
+ constructor(message, position) {
17
+ super(`${message} (line ${position.line}, column ${position.column})`);
18
+ this.position = position;
19
+ }
20
+ }
21
+ /* -------------------------------------------------------------------------
22
+ * Keyword recognition.
23
+ *
24
+ * Per spec §3.8, language keywords are case-insensitive. The lexer emits
25
+ * raw Identifier tokens; the parser decides whether each one is a keyword
26
+ * via `kw()` (lowercase comparison).
27
+ * ----------------------------------------------------------------------- */
28
+ const CONTAINER_KEYWORDS = new Set([
29
+ 'container', 'schema', 'database', 'keyspace', 'namespace', 'dataset', 'bucket',
30
+ ]);
31
+ const ENTITY_KEYWORDS = new Set(['table', 'entity', 'collection', 'record']);
32
+ /**
33
+ * Element-type keywords accepted in module-system import items
34
+ * (spec §26.3). Stored lowercased; matching is case-insensitive.
35
+ *
36
+ * `field` is recognized but explicitly rejected by parseImportItem in P4
37
+ * (field-level imports have special declaration-vs-placement semantics
38
+ * that will land in a later batch).
39
+ *
40
+ * `project` is intentionally excluded -- spec §26.1 forbids importing
41
+ * Project declarations.
42
+ */
43
+ const IMPORT_ELEMENT_TYPES = new Set([
44
+ 'table', 'entity', 'collection', 'record',
45
+ 'enum', 'tablepartial', 'note',
46
+ 'schema', 'container', 'tablegroup',
47
+ 'type', 'edge', 'view', 'diagramview',
48
+ 'field',
49
+ ]);
50
+ const STRUCTURAL_TYPE_KEYWORDS = new Set([
51
+ 'object', 'struct', 'record', 'array', 'list', 'map', 'dict', 'dictionary',
52
+ 'set', 'union', 'oneof', 'anyof', 'allof', 'json', 'jsonb', 'variant',
53
+ ]);
54
+ function kw(token) {
55
+ if (!token || token.kind !== TokenKind.Identifier)
56
+ return null;
57
+ return token.text.toLowerCase();
58
+ }
59
+ function isKw(token, expected) {
60
+ return kw(token) === expected;
61
+ }
62
+ /** Canonical capitalization for container keywords */
63
+ function canonContainerKw(raw) {
64
+ const lower = raw.toLowerCase();
65
+ switch (lower) {
66
+ case 'container': return 'Container';
67
+ case 'schema': return 'Schema';
68
+ case 'database': return 'Database';
69
+ case 'keyspace': return 'Keyspace';
70
+ case 'namespace': return 'Namespace';
71
+ case 'dataset': return 'Dataset';
72
+ case 'bucket': return 'Bucket';
73
+ default: throw new Error(`Not a container keyword: ${raw}`);
74
+ }
75
+ }
76
+ function canonEntityKw(raw) {
77
+ const lower = raw.toLowerCase();
78
+ switch (lower) {
79
+ case 'table': return 'Table';
80
+ case 'entity': return 'Entity';
81
+ case 'collection': return 'Collection';
82
+ case 'record': return 'Record';
83
+ default: throw new Error(`Not an entity keyword: ${raw}`);
84
+ }
85
+ }
86
+ export class Parser {
87
+ tokens;
88
+ idx = 0;
89
+ /**
90
+ * Parse-time options (v0.2 / P5+). Carries the importer's filePath, the
91
+ * optional readFile resolver, and the maxDepth bound. Used by
92
+ * parseModuleDirective to resolve reference-only directives. May be an
93
+ * empty object when no options were supplied (the public `parse(source)`
94
+ * 1-arg form).
95
+ */
96
+ options;
97
+ /**
98
+ * The set of file paths currently being parsed in the resolution chain.
99
+ * Used for cycle detection: when resolving a directive whose `from` path
100
+ * is already in this set, the parser produces an empty clone for that
101
+ * directive rather than recursing (matching spec §26.15: cycles are
102
+ * allowed; name resolution handles them). The set is passed by reference
103
+ * across recursive parse() calls so all transitive levels see it.
104
+ *
105
+ * The set contains the resolved ABSOLUTE paths (post-readFile-key path
106
+ * computation), not the source-text `from` strings, so two directives
107
+ * that name the same file via different relative paths still collide.
108
+ */
109
+ resolutionStack;
110
+ /**
111
+ * Current recursion depth. Incremented before each recursive parse(),
112
+ * compared against options.maxDepth. Reaching the limit throws.
113
+ */
114
+ depth;
115
+ constructor(tokens, options = {}, resolutionStack = new Set(), depth = 0) {
116
+ this.tokens = tokens;
117
+ this.options = options;
118
+ this.resolutionStack = resolutionStack;
119
+ this.depth = depth;
120
+ }
121
+ /* ----- low-level token helpers ----- */
122
+ peek(lookahead = 0) {
123
+ return this.tokens[this.idx + lookahead];
124
+ }
125
+ advance() {
126
+ const t = this.tokens[this.idx];
127
+ if (this.idx < this.tokens.length - 1)
128
+ this.idx += 1;
129
+ return t;
130
+ }
131
+ check(kind) {
132
+ return this.peek().kind === kind;
133
+ }
134
+ match(kind) {
135
+ if (this.check(kind))
136
+ return this.advance();
137
+ return null;
138
+ }
139
+ expect(kind, msg) {
140
+ if (this.check(kind))
141
+ return this.advance();
142
+ const t = this.peek();
143
+ throw new ParseError(`${msg} (got ${t.kind} ${JSON.stringify(t.text)})`, t.start);
144
+ }
145
+ spanFrom(start) {
146
+ // span end = end-position of the previously consumed token if any
147
+ const prev = this.idx > 0 ? this.tokens[this.idx - 1] : this.tokens[0];
148
+ return {
149
+ start,
150
+ end: prev.end,
151
+ };
152
+ }
153
+ /* ----- entry point ----- */
154
+ parseDocument() {
155
+ const start = this.peek().start;
156
+ let version;
157
+ let experimental;
158
+ if (isKw(this.peek(), 'xdbml') && this.peek(1).kind === TokenKind.Colon) {
159
+ version = this.parseVersionDeclaration();
160
+ }
161
+ if (isKw(this.peek(), 'experimental') && this.peek(1).kind === TokenKind.Colon) {
162
+ experimental = this.parseExperimentalDeclaration();
163
+ }
164
+ const statements = [];
165
+ while (!this.check(TokenKind.EOF)) {
166
+ statements.push(this.parseTopLevelStatement());
167
+ }
168
+ return {
169
+ kind: 'XDbmlDocument',
170
+ version,
171
+ experimental,
172
+ statements,
173
+ span: this.spanFrom(start),
174
+ };
175
+ }
176
+ /* ----- version & experimental ----- */
177
+ parseVersionDeclaration() {
178
+ const start = this.peek().start;
179
+ this.advance(); // xdbml
180
+ this.expect(TokenKind.Colon, "Expected ':' after 'xdbml'");
181
+ const numTok = this.expect(TokenKind.NumberLiteral, 'Expected version number');
182
+ return {
183
+ kind: 'VersionDeclaration',
184
+ version: numTok.text,
185
+ span: this.spanFrom(start),
186
+ };
187
+ }
188
+ parseExperimentalDeclaration() {
189
+ const start = this.peek().start;
190
+ this.advance(); // experimental
191
+ this.expect(TokenKind.Colon, "Expected ':' after 'experimental'");
192
+ this.expect(TokenKind.LBracket, "Expected '[' for feature list");
193
+ const features = [];
194
+ if (!this.check(TokenKind.RBracket)) {
195
+ features.push(this.expect(TokenKind.Identifier, 'Expected feature name').text);
196
+ while (this.match(TokenKind.Comma)) {
197
+ features.push(this.expect(TokenKind.Identifier, 'Expected feature name').text);
198
+ }
199
+ }
200
+ this.expect(TokenKind.RBracket, "Expected ']'");
201
+ return {
202
+ kind: 'ExperimentalDeclaration',
203
+ features,
204
+ span: this.spanFrom(start),
205
+ };
206
+ }
207
+ /* ----- top-level dispatch ----- */
208
+ parseTopLevelStatement() {
209
+ const t = this.peek();
210
+ const k = kw(t);
211
+ if (k === null) {
212
+ throw new ParseError(`Unexpected token ${t.kind} ${JSON.stringify(t.text)} at top level`, t.start);
213
+ }
214
+ if (k === 'project')
215
+ return this.parseProject();
216
+ if (CONTAINER_KEYWORDS.has(k))
217
+ return this.parseContainer();
218
+ if (ENTITY_KEYWORDS.has(k))
219
+ return this.parseEntity();
220
+ if (k === 'type')
221
+ return this.parseTypeDecl();
222
+ if (k === 'edge')
223
+ return this.parseEdge();
224
+ if (k === 'view')
225
+ return this.parseView();
226
+ if (k === 'enum')
227
+ return this.parseEnum();
228
+ if (k === 'ref')
229
+ return this.parseRef();
230
+ if (k === 'tablepartial')
231
+ return this.parseTablePartial();
232
+ if (k === 'tablegroup')
233
+ return this.parseTableGroup();
234
+ if (k === 'note')
235
+ return this.parseNoteDeclaration();
236
+ if (k === 'records')
237
+ return this.parseTopLevelRecords();
238
+ if (k === 'use' || k === 'reuse')
239
+ return this.parseModuleDirective('file-scope');
240
+ throw new ParseError(`Unknown top-level construct: ${t.text}`, t.start);
241
+ }
242
+ /* ----- Project ----- */
243
+ parseProject() {
244
+ const start = this.peek().start;
245
+ this.advance(); // Project
246
+ const name = this.parseIdentLikeName('project name');
247
+ this.expect(TokenKind.LBrace, "Expected '{' after Project name");
248
+ const body = [];
249
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
250
+ // Project body is either an inline Note block or a setting line.
251
+ if (isKw(this.peek(), 'note')) {
252
+ body.push(this.parseNoteBlockOrSetting());
253
+ }
254
+ else {
255
+ body.push(this.parseLineSetting());
256
+ }
257
+ }
258
+ this.expect(TokenKind.RBrace, "Expected '}' closing Project");
259
+ return {
260
+ kind: 'ProjectDeclaration',
261
+ name,
262
+ body,
263
+ span: this.spanFrom(start),
264
+ };
265
+ }
266
+ /**
267
+ * A Note inside a Project/Container/Entity body. May appear as:
268
+ * Note: 'short text'
269
+ * Note: '''long text'''
270
+ * Note { '''long text''' }
271
+ */
272
+ parseNoteBlockOrSetting() {
273
+ const start = this.peek().start;
274
+ this.advance(); // Note
275
+ if (this.match(TokenKind.Colon)) {
276
+ const s = this.parseSettingValueExpectingString('Expected string after Note:');
277
+ return {
278
+ kind: 'NoteBlock',
279
+ body: s,
280
+ span: this.spanFrom(start),
281
+ };
282
+ }
283
+ this.expect(TokenKind.LBrace, "Expected ':' or '{' after Note");
284
+ const body = this.parseSettingValueExpectingString('Expected string inside Note { ... }');
285
+ this.expect(TokenKind.RBrace, "Expected '}' closing Note block");
286
+ return {
287
+ kind: 'NoteBlock',
288
+ body,
289
+ span: this.spanFrom(start),
290
+ };
291
+ }
292
+ parseSettingValueExpectingString(msg) {
293
+ const t = this.peek();
294
+ if (t.kind === TokenKind.StringLiteral || t.kind === TokenKind.MultilineString) {
295
+ this.advance();
296
+ return t.value ?? '';
297
+ }
298
+ throw new ParseError(msg, t.start);
299
+ }
300
+ /**
301
+ * Top-level `Note name { '''...''' }` standalone declaration.
302
+ */
303
+ parseNoteDeclaration() {
304
+ const start = this.peek().start;
305
+ this.advance(); // Note
306
+ // Could be: `Note: '...'`, `Note name { '''...''' }`, or `Note { ... }`
307
+ let name;
308
+ if (this.check(TokenKind.Identifier) || this.check(TokenKind.QuotedIdentifier)) {
309
+ const tok = this.advance();
310
+ name = tok.kind === TokenKind.QuotedIdentifier ? (tok.value ?? '') : tok.text;
311
+ }
312
+ if (this.match(TokenKind.Colon)) {
313
+ const body = this.parseSettingValueExpectingString('Expected string after Note:');
314
+ return {
315
+ kind: 'NoteDeclaration',
316
+ name,
317
+ body,
318
+ span: this.spanFrom(start),
319
+ };
320
+ }
321
+ this.expect(TokenKind.LBrace, "Expected '{' or ':' after Note");
322
+ const body = this.parseSettingValueExpectingString('Expected string inside Note block');
323
+ this.expect(TokenKind.RBrace, "Expected '}' closing Note");
324
+ return {
325
+ kind: 'NoteDeclaration',
326
+ name,
327
+ body,
328
+ span: this.spanFrom(start),
329
+ };
330
+ }
331
+ /**
332
+ * Parse a `name: value` line inside a Project body. Used for project
333
+ * settings like `targets: PostgreSQL` or `database_type: 'MySQL'`.
334
+ */
335
+ parseLineSetting() {
336
+ const start = this.peek().start;
337
+ const nameTok = this.peek();
338
+ if (nameTok.kind !== TokenKind.Identifier && nameTok.kind !== TokenKind.QuotedIdentifier) {
339
+ throw new ParseError(`Expected setting name, got ${nameTok.kind}`, nameTok.start);
340
+ }
341
+ const nameSource = nameTok.kind === TokenKind.QuotedIdentifier ? (nameTok.value ?? '') : nameTok.text;
342
+ this.advance();
343
+ this.expect(TokenKind.Colon, "Expected ':' after setting name");
344
+ const value = this.parseSettingValue();
345
+ return {
346
+ kind: 'Setting',
347
+ name: nameSource.toLowerCase(),
348
+ nameSource,
349
+ value,
350
+ span: this.spanFrom(start),
351
+ };
352
+ }
353
+ /* ----- Container ----- */
354
+ parseContainer() {
355
+ const start = this.peek().start;
356
+ const kwTok = this.advance();
357
+ const name = this.parseIdentLikeName('container name');
358
+ const settings = this.maybeSettingsBlock();
359
+ this.expect(TokenKind.LBrace, "Expected '{' after Container name");
360
+ const body = [];
361
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
362
+ const t = this.peek();
363
+ const k = kw(t);
364
+ if (k === 'note') {
365
+ body.push(this.parseNoteBlockOrSetting());
366
+ }
367
+ else if (k && ENTITY_KEYWORDS.has(k)) {
368
+ body.push(this.parseEntity());
369
+ }
370
+ else if (k === 'edge') {
371
+ body.push(this.parseEdge());
372
+ }
373
+ else if (k === 'view') {
374
+ body.push(this.parseView());
375
+ }
376
+ else if (k === 'enum') {
377
+ body.push(this.parseEnum());
378
+ }
379
+ else if (k === 'use' || k === 'reuse') {
380
+ body.push(this.parseModuleDirective('container-body'));
381
+ }
382
+ else {
383
+ // Unknown line; tolerate as no-op rather than fail the whole parse.
384
+ throw new ParseError(`Unexpected token in Container body: ${t.kind} ${JSON.stringify(t.text)}`, t.start);
385
+ }
386
+ }
387
+ this.expect(TokenKind.RBrace, "Expected '}' closing Container");
388
+ return {
389
+ kind: 'ContainerDeclaration',
390
+ keyword: canonContainerKw(kwTok.text),
391
+ name,
392
+ settings,
393
+ body,
394
+ span: this.spanFrom(start),
395
+ };
396
+ }
397
+ /* ----- Entity ----- */
398
+ parseEntity() {
399
+ const start = this.peek().start;
400
+ const kwTok = this.advance();
401
+ const name = this.parseEntityName();
402
+ let alias;
403
+ if (isKw(this.peek(), 'as')) {
404
+ this.advance();
405
+ alias = this.parseIdentLikeName('alias');
406
+ }
407
+ const settings = this.maybeSettingsBlock();
408
+ this.expect(TokenKind.LBrace, "Expected '{' after entity name");
409
+ const body = this.parseEntityBody();
410
+ this.expect(TokenKind.RBrace, "Expected '}' closing entity");
411
+ return {
412
+ kind: 'EntityDeclaration',
413
+ keyword: canonEntityKw(kwTok.text),
414
+ name,
415
+ alias,
416
+ settings,
417
+ body,
418
+ span: this.spanFrom(start),
419
+ };
420
+ }
421
+ /**
422
+ * Entity names may be bare (`users`), dotted (`core.users` — implicit
423
+ * container), or quoted (`"my-table"`).
424
+ */
425
+ parseEntityName() {
426
+ const t = this.peek();
427
+ if (t.kind === TokenKind.QuotedIdentifier) {
428
+ this.advance();
429
+ return t.value ?? '';
430
+ }
431
+ if (t.kind !== TokenKind.Identifier) {
432
+ throw new ParseError(`Expected entity name, got ${t.kind}`, t.start);
433
+ }
434
+ this.advance();
435
+ let name = t.text;
436
+ while (this.check(TokenKind.Dot)) {
437
+ this.advance();
438
+ const next = this.expect(TokenKind.Identifier, 'Expected identifier after dot in entity name');
439
+ name += `.${next.text}`;
440
+ }
441
+ return name;
442
+ }
443
+ parseEntityBody() {
444
+ const body = [];
445
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
446
+ const t = this.peek();
447
+ const k = kw(t);
448
+ if (k === 'note') {
449
+ body.push(this.parseNoteBlockOrSetting());
450
+ }
451
+ else if (k === 'indexes') {
452
+ body.push(this.parseIndexes());
453
+ }
454
+ else if (k === 'checks') {
455
+ body.push(this.parseChecks());
456
+ }
457
+ else if (k === 'records') {
458
+ body.push(this.parseRecordsBlock());
459
+ }
460
+ else if (t.kind === TokenKind.Tilde) {
461
+ body.push(this.parsePartialInjection());
462
+ }
463
+ else {
464
+ body.push(this.parseFieldDeclaration());
465
+ }
466
+ }
467
+ return body;
468
+ }
469
+ parsePartialInjection() {
470
+ const start = this.peek().start;
471
+ this.expect(TokenKind.Tilde, "Expected '~'");
472
+ const nameTok = this.expect(TokenKind.Identifier, "Expected partial name after '~'");
473
+ return {
474
+ kind: 'PartialInjection',
475
+ partialName: nameTok.text,
476
+ span: this.spanFrom(start),
477
+ };
478
+ }
479
+ /**
480
+ * Parse a `records { ... }` block inside an entity body (§25.1, implicit
481
+ * column list). Values are stored as SettingValue cells; row boundaries
482
+ * are determined by source line (see `parseRecordRow`).
483
+ */
484
+ parseRecordsBlock() {
485
+ const start = this.peek().start;
486
+ this.advance(); // records
487
+ this.expect(TokenKind.LBrace, "Expected '{' after 'records'");
488
+ const rows = [];
489
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
490
+ rows.push(this.parseRecordRow());
491
+ }
492
+ this.expect(TokenKind.RBrace, "Expected '}' closing records");
493
+ return {
494
+ kind: 'RecordsBlock',
495
+ rows,
496
+ span: this.spanFrom(start),
497
+ };
498
+ }
499
+ /**
500
+ * Top-level records declaration (§25.2, new in v0.2):
501
+ *
502
+ * records users (id, name, email) { ... }
503
+ * records core.users (id, name, email) { ... }
504
+ *
505
+ * The entity reference can be a bare name or a dotted path for cross-
506
+ * container references. The column list is required; it tells the
507
+ * generator which columns each row's values are populating.
508
+ */
509
+ parseTopLevelRecords() {
510
+ const start = this.peek().start;
511
+ this.advance(); // records
512
+ // Entity reference: bare identifier or dotted path (`core.users`).
513
+ const refStart = this.peek().start;
514
+ const head = this.expect(TokenKind.Identifier, "Expected entity name after 'records'");
515
+ let entityRef = head.text;
516
+ while (this.check(TokenKind.Dot)) {
517
+ this.advance();
518
+ const next = this.expect(TokenKind.Identifier, "Expected identifier after '.' in entity reference");
519
+ entityRef += `.${next.text}`;
520
+ }
521
+ // Explicit column list -- required for top-level form.
522
+ this.expect(TokenKind.LParen, "Expected '(' starting column list after entity reference");
523
+ const columns = [];
524
+ if (!this.check(TokenKind.RParen)) {
525
+ const first = this.expect(TokenKind.Identifier, 'Expected column name');
526
+ columns.push(first.text);
527
+ while (this.match(TokenKind.Comma)) {
528
+ if (this.check(TokenKind.RParen))
529
+ break; // tolerate trailing comma
530
+ const next = this.expect(TokenKind.Identifier, 'Expected column name after comma');
531
+ columns.push(next.text);
532
+ }
533
+ }
534
+ this.expect(TokenKind.RParen, "Expected ')' closing column list");
535
+ // Row body.
536
+ this.expect(TokenKind.LBrace, "Expected '{' starting records body");
537
+ const rows = [];
538
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
539
+ rows.push(this.parseRecordRow());
540
+ }
541
+ this.expect(TokenKind.RBrace, "Expected '}' closing records body");
542
+ void refStart; // currently unused but reserved for future improved error reporting
543
+ return {
544
+ kind: 'TopLevelRecordsDeclaration',
545
+ entityRef,
546
+ columns,
547
+ rows,
548
+ span: this.spanFrom(start),
549
+ };
550
+ }
551
+ /**
552
+ * Parse a single row of comma-separated values.
553
+ *
554
+ * Row delimiter rule: a comma continues the row only when the next value
555
+ * is on the same source line as the comma. If the comma is followed by
556
+ * a token on a later line (or the closing `}`), the comma is treated as
557
+ * a trailing comma and the row ends. This rule:
558
+ *
559
+ * - Tolerates trailing commas at end of row
560
+ * - Supports triple-quoted multi-line string VALUES (the comma after
561
+ * the closing `'''` is on the line of the closing triple, and the
562
+ * next value sits on that same line)
563
+ * - Does NOT support multi-line rows where a row's values are spread
564
+ * across multiple source lines connected by commas
565
+ */
566
+ parseRecordRow() {
567
+ const start = this.peek().start;
568
+ const values = [this.parseSettingValue()];
569
+ while (this.check(TokenKind.Comma)) {
570
+ const commaLine = this.peek().start.line;
571
+ this.advance(); // consume comma
572
+ // Check what follows the comma. If it's on a later line, treat as trailing.
573
+ const nextTok = this.peek();
574
+ if (nextTok.kind === TokenKind.RBrace || nextTok.kind === TokenKind.EOF) {
575
+ // trailing comma at end of block
576
+ break;
577
+ }
578
+ if (nextTok.start.line > commaLine) {
579
+ // trailing comma at end of row (next value is on a later line)
580
+ break;
581
+ }
582
+ values.push(this.parseSettingValue());
583
+ }
584
+ return {
585
+ kind: 'RecordRow',
586
+ values,
587
+ span: this.spanFrom(start),
588
+ };
589
+ }
590
+ /* ----- Module-system directives (spec §26, new in v0.2) ----- */
591
+ /**
592
+ * Parse a `use` or `reuse` directive. Called from both the top-level
593
+ * dispatcher and the Container body dispatcher; the caller indicates
594
+ * which context via the `context` argument. The context affects which
595
+ * placements are legal (e.g., field imports must be at file scope) but
596
+ * does NOT affect the directive's syntactic shape.
597
+ *
598
+ * Grammar:
599
+ *
600
+ * ('use' | 'reuse') importSpec 'from' StringLiteral metadataSettings? cloneBlock?
601
+ *
602
+ * importSpec ::= '*' | '{' importItem (',' importItem)* '}'
603
+ * importItem ::= elementType path ('as' Identifier)?
604
+ * elementType ::= 'table' | 'entity' | 'collection' | 'record' |
605
+ * 'enum' | 'tablepartial' | 'note' | 'schema' |
606
+ * 'container' | 'tablegroup' | 'type' | 'edge' |
607
+ * 'view' | 'diagramview' | 'field'
608
+ * metadataSettings ::= '[' setting (',' setting)* ']'
609
+ * cloneBlock ::= '{' topLevelStatement* '}'
610
+ */
611
+ parseModuleDirective(context) {
612
+ const start = this.peek().start;
613
+ const modeTok = this.advance(); // 'use' or 'reuse'
614
+ const mode = modeTok.text.toLowerCase();
615
+ // Import spec: '*' or '{ ... }'
616
+ let spec;
617
+ if (this.check(TokenKind.Star)) {
618
+ this.advance();
619
+ spec = { kind: 'ImportAll' };
620
+ }
621
+ else if (this.check(TokenKind.LBrace)) {
622
+ this.advance();
623
+ const items = [];
624
+ // Skip leading whitespace/newlines (already handled by lexer).
625
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
626
+ items.push(this.parseImportItem(context));
627
+ if (this.match(TokenKind.Comma)) {
628
+ // Tolerate trailing comma before the closing brace.
629
+ continue;
630
+ }
631
+ else {
632
+ break;
633
+ }
634
+ }
635
+ this.expect(TokenKind.RBrace, "Expected '}' closing import item list");
636
+ if (items.length === 0) {
637
+ throw new ParseError(`Expected at least one import item between '{' and '}'`, start);
638
+ }
639
+ spec = { kind: 'ImportList', items };
640
+ }
641
+ else {
642
+ const t = this.peek();
643
+ throw new ParseError(`Expected '*' or '{' after '${mode}', got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
644
+ }
645
+ // 'from' keyword
646
+ if (!isKw(this.peek(), 'from')) {
647
+ const t = this.peek();
648
+ throw new ParseError(`Expected 'from' after import spec, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
649
+ }
650
+ this.advance(); // from
651
+ // The path: a single string literal.
652
+ const pathTok = this.peek();
653
+ if (pathTok.kind !== TokenKind.StringLiteral) {
654
+ throw new ParseError(`Expected string literal path after 'from', got ${pathTok.kind} ${JSON.stringify(pathTok.text)}`, pathTok.start);
655
+ }
656
+ this.advance();
657
+ const from = pathTok.value ?? '';
658
+ // v0.3 §26.14: classify the source string up front so a disallowed form
659
+ // (non-https scheme, protocol-relative, embedded credentials, bare host)
660
+ // surfaces as a located error pointing at the string itself, regardless
661
+ // of whether a resolver is present.
662
+ try {
663
+ classifyModuleSource(from);
664
+ }
665
+ catch (e) {
666
+ if (e instanceof ModuleSourceError) {
667
+ throw new ParseError(e.message, pathTok.start);
668
+ }
669
+ throw e;
670
+ }
671
+ // Optional metadata settings: '[cloned_at: ...]'
672
+ const settings = this.maybeSettingsBlock();
673
+ // Optional clone block: '{ ...top-level statements... }'
674
+ let clone;
675
+ if (this.check(TokenKind.LBrace)) {
676
+ clone = this.parseCloneBlock();
677
+ }
678
+ // P5: if no inline clone block, attempt to resolve the referenced file
679
+ // using the supplied readFile callback. If no callback was supplied
680
+ // (the bare `parse(source)` 1-arg form), fall back to the P4 rejection.
681
+ let resolvedPath;
682
+ let resolutionCycle = false;
683
+ if (!clone) {
684
+ if (this.options.readFile) {
685
+ // Build the directive shape we need to pass to the resolver. We
686
+ // haven't finalized the AST node yet (we need its `clone` field),
687
+ // so we pass a partial directive that has all the fields resolveImport
688
+ // reads (from, span, mode).
689
+ const partial = {
690
+ kind: 'ModuleImportDirective',
691
+ mode,
692
+ spec,
693
+ from,
694
+ settings,
695
+ span: this.spanFrom(start),
696
+ };
697
+ const result = resolveImport(partial, this.options, this.resolutionStack, this.depth, recursiveParse);
698
+ switch (result.kind) {
699
+ case 'resolved':
700
+ clone = result.clone;
701
+ resolvedPath = result.resolvedPath;
702
+ break;
703
+ case 'cycle':
704
+ // Per spec §26.15, cycles are allowed; the parser produces a
705
+ // directive with no clone, and name resolution (P6+) is
706
+ // expected to bridge the cycle. We leave clone undefined.
707
+ resolvedPath = result.resolvedPath;
708
+ resolutionCycle = true;
709
+ break;
710
+ case 'no-resolver':
711
+ // Shouldn't reach this branch because we already checked
712
+ // readFile above, but treat it as the P4 rejection
713
+ // defensively rather than silently producing an unresolved
714
+ // directive.
715
+ throw new ParseError(`Reference-only '${mode}' directive (no clone block) could not be resolved: ` +
716
+ `no readFile resolver was supplied in ParseOptions.`, start);
717
+ }
718
+ }
719
+ else {
720
+ // P4 fallback: no clone, no resolver. Reject with the original
721
+ // message pointing to the clone-block escape hatch.
722
+ throw new ParseError(`Reference-only '${mode}' directive (no clone block) cannot be resolved: ` +
723
+ `no readFile resolver was supplied in ParseOptions. ` +
724
+ `Either provide a ParseOptions.readFile callback when calling parse(), ` +
725
+ `or add an inline clone block to the directive to make the file self-contained.`, start);
726
+ }
727
+ }
728
+ void resolvedPath;
729
+ void resolutionCycle; // currently unused; reserved for future provenance metadata
730
+ return {
731
+ kind: 'ModuleImportDirective',
732
+ mode,
733
+ spec,
734
+ from,
735
+ settings,
736
+ clone,
737
+ span: this.spanFrom(start),
738
+ };
739
+ }
740
+ /**
741
+ * Parse one import item: an element-type keyword, a dotted source path,
742
+ * and an optional `as <alias>`.
743
+ *
744
+ * entity core.dim_customer
745
+ * type Email
746
+ * type Email as PII_Email
747
+ * field core.dim_customer.email (rejected in P4)
748
+ */
749
+ parseImportItem(context) {
750
+ const start = this.peek().start;
751
+ // Element type keyword.
752
+ const elemTok = this.peek();
753
+ if (elemTok.kind !== TokenKind.Identifier) {
754
+ throw new ParseError(`Expected import element type keyword, got ${elemTok.kind} ${JSON.stringify(elemTok.text)}`, elemTok.start);
755
+ }
756
+ const elementType = elemTok.text.toLowerCase();
757
+ if (!IMPORT_ELEMENT_TYPES.has(elementType)) {
758
+ throw new ParseError(`Unknown import element type '${elemTok.text}'. ` +
759
+ `Expected one of: ${Array.from(IMPORT_ELEMENT_TYPES).join(', ')}.`, elemTok.start);
760
+ }
761
+ if (elementType === 'field' && context !== 'file-scope') {
762
+ // Spec §26.8: field imports must appear at file scope. Inside a
763
+ // Container body, the field's eventual placement (as a Named Type)
764
+ // would have no meaningful container scope -- field imports are
765
+ // always lifted to file scope by flatten(), regardless of where
766
+ // the directive sits.
767
+ throw new ParseError(`Field-level imports must appear at file scope, not inside a Container body (spec §26.8).`, elemTok.start);
768
+ }
769
+ this.advance(); // consume element type keyword
770
+ // Dotted source path.
771
+ const pathHead = this.expect(TokenKind.Identifier, `Expected source path after '${elementType}'`);
772
+ let sourcePath = pathHead.text;
773
+ while (this.check(TokenKind.Dot)) {
774
+ this.advance();
775
+ const next = this.expect(TokenKind.Identifier, `Expected identifier after '.' in source path`);
776
+ sourcePath += `.${next.text}`;
777
+ }
778
+ // Optional 'as <alias>'
779
+ let alias;
780
+ if (isKw(this.peek(), 'as')) {
781
+ this.advance(); // as
782
+ const aliasTok = this.expect(TokenKind.Identifier, `Expected identifier after 'as'`);
783
+ alias = aliasTok.text;
784
+ }
785
+ return {
786
+ kind: 'ImportItem',
787
+ elementType,
788
+ sourcePath,
789
+ alias,
790
+ span: this.spanFrom(start),
791
+ };
792
+ }
793
+ /**
794
+ * Parse a clone block. The block contains zero or more declarations
795
+ * that match the import items by name and element type (matching is
796
+ * downstream-consumer's job; the parser is permissive).
797
+ *
798
+ * Per spec §26.6, clone content uses the importing file's vocabulary
799
+ * (aliases already applied) and is parsed under the importing file's
800
+ * xdbml version directive.
801
+ *
802
+ * Most clone-block content uses TopLevelStatement shapes (Entity, Type,
803
+ * Container, etc.). The exception is field imports (§26.8): when the
804
+ * directive imports one or more fields via `field <path>` items, the
805
+ * clone block holds each field as a bare FieldDeclaration with no entity
806
+ * wrapper. The dispatch below checks whether the next token starts a
807
+ * known top-level keyword and falls through to FieldDeclaration when
808
+ * it doesn't.
809
+ */
810
+ parseCloneBlock() {
811
+ const start = this.peek().start;
812
+ this.expect(TokenKind.LBrace, "Expected '{' starting clone block");
813
+ const statements = [];
814
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
815
+ if (this.isCloneTopLevelStart()) {
816
+ statements.push(this.parseTopLevelStatement());
817
+ }
818
+ else {
819
+ // Bare field declaration -- the field-import case. Per spec §26.6
820
+ // the field appears without an entity wrapper.
821
+ statements.push(this.parseFieldDeclaration());
822
+ }
823
+ }
824
+ this.expect(TokenKind.RBrace, "Expected '}' closing clone block");
825
+ return {
826
+ kind: 'CloneBlock',
827
+ statements,
828
+ span: this.spanFrom(start),
829
+ };
830
+ }
831
+ /**
832
+ * Lookahead helper: does the current token start a top-level statement?
833
+ *
834
+ * Used by parseCloneBlock to dispatch between "this is a top-level
835
+ * declaration" (Entity, Type, Container, etc.) and "this is a bare
836
+ * field declaration" (for field imports). A field declaration starts
837
+ * with an identifier followed by a type expression; a top-level
838
+ * statement starts with one of the known top-level keywords.
839
+ *
840
+ * Mirrors the dispatch in parseTopLevelStatement(). If we add new
841
+ * top-level constructs there, this set should grow in parallel.
842
+ */
843
+ isCloneTopLevelStart() {
844
+ const k = kw(this.peek());
845
+ if (k === null)
846
+ return false;
847
+ if (k === 'project')
848
+ return true;
849
+ if (CONTAINER_KEYWORDS.has(k))
850
+ return true;
851
+ if (ENTITY_KEYWORDS.has(k))
852
+ return true;
853
+ if (k === 'type')
854
+ return true;
855
+ if (k === 'edge')
856
+ return true;
857
+ if (k === 'view')
858
+ return true;
859
+ if (k === 'enum')
860
+ return true;
861
+ if (k === 'ref')
862
+ return true;
863
+ if (k === 'tablepartial')
864
+ return true;
865
+ if (k === 'tablegroup')
866
+ return true;
867
+ if (k === 'note')
868
+ return true;
869
+ if (k === 'records')
870
+ return true;
871
+ if (k === 'use' || k === 'reuse')
872
+ return true;
873
+ return false;
874
+ }
875
+ /**
876
+ * `field_name typeExpression [settings]` or `"quoted name" typeExpression [settings]`.
877
+ *
878
+ * Critical lookahead point: we're invoked from a context where the next
879
+ * token MUST be a field name (Identifier or QuotedIdentifier), and the
880
+ * token after it is a type expression. If the next thing is a Note block
881
+ * or a partial injection or `indexes`, those should have been handled by
882
+ * the caller already.
883
+ */
884
+ parseFieldDeclaration() {
885
+ const start = this.peek().start;
886
+ const nameTok = this.peek();
887
+ let name;
888
+ let nameQuoted = false;
889
+ if (nameTok.kind === TokenKind.QuotedIdentifier) {
890
+ this.advance();
891
+ name = nameTok.value ?? '';
892
+ nameQuoted = true;
893
+ }
894
+ else if (nameTok.kind === TokenKind.Identifier) {
895
+ this.advance();
896
+ name = nameTok.text;
897
+ }
898
+ else {
899
+ throw new ParseError(`Expected field name, got ${nameTok.kind} ${JSON.stringify(nameTok.text)}`, nameTok.start);
900
+ }
901
+ const type = this.parseTypeExpression();
902
+ const settings = this.maybeSettingsBlock();
903
+ return {
904
+ kind: 'FieldDeclaration',
905
+ name,
906
+ nameQuoted,
907
+ type,
908
+ settings,
909
+ span: this.spanFrom(start),
910
+ };
911
+ }
912
+ /* ----- Type expressions ----- */
913
+ /**
914
+ * Parse a type expression. Dispatch on the leading keyword/identifier:
915
+ *
916
+ * - `object { ... }` (and synonyms struct/record)
917
+ * - `array [ ... ]` (and synonym list)
918
+ * - `map [k, v]` (and synonyms dict/dictionary)
919
+ * - `set [t]`
920
+ * - `union [ ... ]`
921
+ * - `oneOf { ... }` / `anyOf { ... }` / `allOf { ... }`
922
+ * - `json { ... }` (and synonyms jsonb/variant; block optional)
923
+ * - Otherwise: scalar / named-type reference. With optional `(p, s)`.
924
+ */
925
+ parseTypeExpression() {
926
+ const t = this.peek();
927
+ const k = kw(t);
928
+ if (k === 'object' || k === 'struct' || k === 'record')
929
+ return this.parseObjectType();
930
+ if (k === 'array' || k === 'list')
931
+ return this.parseArrayType();
932
+ if (k === 'map' || k === 'dict' || k === 'dictionary')
933
+ return this.parseMapType();
934
+ if (k === 'set')
935
+ return this.parseSetType();
936
+ if (k === 'union')
937
+ return this.parseUnionType();
938
+ if (k === 'oneof')
939
+ return this.parsePolymorphicType('oneOf');
940
+ if (k === 'anyof')
941
+ return this.parsePolymorphicType('anyOf');
942
+ if (k === 'allof')
943
+ return this.parsePolymorphicType('allOf');
944
+ if (k === 'json' || k === 'jsonb' || k === 'variant')
945
+ return this.parseJsonType();
946
+ return this.parseScalarOrNamedType();
947
+ }
948
+ parseObjectType() {
949
+ const start = this.peek().start;
950
+ const kwTok = this.advance();
951
+ this.expect(TokenKind.LBrace, "Expected '{' after object keyword");
952
+ const fields = [];
953
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
954
+ const t = this.peek();
955
+ const k = kw(t);
956
+ if (k === 'note') {
957
+ fields.push(this.parseNoteBlockOrSetting());
958
+ }
959
+ else if (t.kind === TokenKind.Tilde) {
960
+ fields.push(this.parsePartialInjection());
961
+ }
962
+ else {
963
+ fields.push(this.parseFieldDeclaration());
964
+ }
965
+ }
966
+ this.expect(TokenKind.RBrace, "Expected '}' closing object");
967
+ const keyword = kwTok.text.toLowerCase();
968
+ return {
969
+ kind: 'ObjectType',
970
+ keyword,
971
+ fields,
972
+ span: this.spanFrom(start),
973
+ };
974
+ }
975
+ /**
976
+ * `array [ ... ]`. The bracket body has several forms:
977
+ *
978
+ * 1. `[varchar]` -- bare element type
979
+ * 2. `[varchar [not null]]` -- element type with settings
980
+ * 3. `[line_item object { ... }]` -- named element type (common with object)
981
+ * 4. `[[0] x object {...}, [1] y object {...}]` -- tuple type
982
+ *
983
+ * Disambiguation: if the first token inside the bracket is `[`, it's a
984
+ * tuple (each tuple element starts with `[N]`). Otherwise we look at the
985
+ * shape: if the first thing is an identifier and the second is also an
986
+ * identifier or a structural-type keyword, it's `name type` form;
987
+ * otherwise the first thing is the bare type.
988
+ */
989
+ parseArrayType() {
990
+ const start = this.peek().start;
991
+ const kwTok = this.advance();
992
+ this.expect(TokenKind.LBracket, "Expected '[' after array keyword");
993
+ // Tuple form?
994
+ if (this.check(TokenKind.LBracket)) {
995
+ const elements = this.parseTupleElements();
996
+ this.expect(TokenKind.RBracket, "Expected ']' closing tuple");
997
+ const tuple = {
998
+ kind: 'TupleType',
999
+ elements,
1000
+ span: this.spanFrom(start),
1001
+ };
1002
+ // Return the tuple wrapped in an ArrayType so the caller knows it's an array.
1003
+ // For PoC simplicity, we encode tuple type by returning it directly and
1004
+ // letting downstream tooling recognize TupleType. But the ArrayType wrapper
1005
+ // is what fields use. Convention: when array body is a tuple, we use the
1006
+ // TupleType kind directly. Cast to satisfy TypeScript.
1007
+ return tuple;
1008
+ }
1009
+ // `name type` form: Identifier followed by something that starts a type.
1010
+ const first = this.peek();
1011
+ const second = this.peek(1);
1012
+ let elementName;
1013
+ if (first.kind === TokenKind.Identifier
1014
+ && (second.kind === TokenKind.Identifier
1015
+ || second.kind === TokenKind.LBrace // `name object { ... }` etc.
1016
+ )
1017
+ && kw(first) !== null
1018
+ && !STRUCTURAL_TYPE_KEYWORDS.has(kw(first))
1019
+ // and the second token must look like the start of a type
1020
+ && this.tokenStartsType(second)) {
1021
+ elementName = this.advance().text;
1022
+ }
1023
+ const elementType = this.parseTypeExpression();
1024
+ const elementSettings = this.maybeSettingsBlock();
1025
+ this.expect(TokenKind.RBracket, "Expected ']' closing array");
1026
+ return {
1027
+ kind: 'ArrayType',
1028
+ keyword: kwTok.text.toLowerCase(),
1029
+ elementType,
1030
+ elementName,
1031
+ elementSettings: elementSettings.length > 0 ? elementSettings : undefined,
1032
+ span: this.spanFrom(start),
1033
+ };
1034
+ }
1035
+ /** True if the token looks like the start of a TypeExpression. */
1036
+ tokenStartsType(t) {
1037
+ if (t.kind === TokenKind.LBrace)
1038
+ return true; // object {...} with implicit `object` keyword? no -- but the parser test should accept structural keywords primarily
1039
+ if (t.kind === TokenKind.Identifier)
1040
+ return true; // could be a scalar like 'int' or a structural keyword like 'object'
1041
+ return false;
1042
+ }
1043
+ parseTupleElements() {
1044
+ const out = [];
1045
+ while (this.check(TokenKind.LBracket) && !this.check(TokenKind.EOF)) {
1046
+ const start = this.peek().start;
1047
+ this.advance(); // [
1048
+ const numTok = this.expect(TokenKind.NumberLiteral, 'Expected position number in tuple');
1049
+ this.expect(TokenKind.RBracket, "Expected ']' after position");
1050
+ const nameTok = this.expect(TokenKind.Identifier, 'Expected tuple element name');
1051
+ const type = this.parseTypeExpression();
1052
+ const settings = this.maybeSettingsBlock();
1053
+ out.push({
1054
+ kind: 'TupleElement',
1055
+ position: parseInt(numTok.text, 10),
1056
+ name: nameTok.text,
1057
+ type,
1058
+ settings,
1059
+ span: this.spanFrom(start),
1060
+ });
1061
+ if (!this.match(TokenKind.Comma))
1062
+ break;
1063
+ }
1064
+ return out;
1065
+ }
1066
+ parseMapType() {
1067
+ const start = this.peek().start;
1068
+ const kwTok = this.advance();
1069
+ this.expect(TokenKind.LBracket, "Expected '[' after map");
1070
+ const keyType = this.parseTypeExpression();
1071
+ this.expect(TokenKind.Comma, "Expected ',' between map key and value");
1072
+ const valueType = this.parseTypeExpression();
1073
+ this.expect(TokenKind.RBracket, "Expected ']' closing map");
1074
+ return {
1075
+ kind: 'MapType',
1076
+ keyword: kwTok.text.toLowerCase(),
1077
+ keyType,
1078
+ valueType,
1079
+ span: this.spanFrom(start),
1080
+ };
1081
+ }
1082
+ parseSetType() {
1083
+ const start = this.peek().start;
1084
+ this.advance(); // set
1085
+ this.expect(TokenKind.LBracket, "Expected '[' after set");
1086
+ const elementType = this.parseTypeExpression();
1087
+ this.expect(TokenKind.RBracket, "Expected ']' closing set");
1088
+ return {
1089
+ kind: 'SetType',
1090
+ elementType,
1091
+ span: this.spanFrom(start),
1092
+ };
1093
+ }
1094
+ /**
1095
+ * `union [t1, t2, null]`. Members are scalars or null.
1096
+ */
1097
+ parseUnionType() {
1098
+ const start = this.peek().start;
1099
+ this.advance(); // union
1100
+ this.expect(TokenKind.LBracket, "Expected '[' after union");
1101
+ const members = [];
1102
+ members.push(this.parseUnionMember());
1103
+ while (this.match(TokenKind.Comma)) {
1104
+ members.push(this.parseUnionMember());
1105
+ }
1106
+ this.expect(TokenKind.RBracket, "Expected ']' closing union");
1107
+ return {
1108
+ kind: 'UnionType',
1109
+ members,
1110
+ span: this.spanFrom(start),
1111
+ };
1112
+ }
1113
+ parseUnionMember() {
1114
+ const t = this.peek();
1115
+ if (isKw(t, 'null')) {
1116
+ const start = t.start;
1117
+ this.advance();
1118
+ return {
1119
+ kind: 'NullTypeLiteral',
1120
+ span: this.spanFrom(start),
1121
+ };
1122
+ }
1123
+ // Reuse scalar parser; named types and scalars look the same syntactically.
1124
+ return this.parseScalarOrNamedType();
1125
+ }
1126
+ parsePolymorphicType(flavor) {
1127
+ const start = this.peek().start;
1128
+ this.advance(); // oneOf | anyOf | allOf
1129
+ this.expect(TokenKind.LBrace, `Expected '{' after ${flavor}`);
1130
+ const alternatives = [];
1131
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1132
+ alternatives.push(this.parsePolymorphicAlternative());
1133
+ }
1134
+ this.expect(TokenKind.RBrace, `Expected '}' closing ${flavor}`);
1135
+ const settings = this.maybeSettingsBlock();
1136
+ return {
1137
+ kind: flavor === 'oneOf' ? 'OneOfType' : flavor === 'anyOf' ? 'AnyOfType' : 'AllOfType',
1138
+ alternatives,
1139
+ settings,
1140
+ span: this.spanFrom(start),
1141
+ };
1142
+ }
1143
+ /** `alternative_name typeExpression [settings]` -- shape is the same as a field declaration, context disambiguates */
1144
+ parsePolymorphicAlternative() {
1145
+ const start = this.peek().start;
1146
+ const nameTok = this.expect(TokenKind.Identifier, 'Expected polymorphic alternative name');
1147
+ const type = this.parseTypeExpression();
1148
+ const settings = this.maybeSettingsBlock();
1149
+ return {
1150
+ kind: 'PolymorphicAlternative',
1151
+ name: nameTok.text,
1152
+ type,
1153
+ settings,
1154
+ span: this.spanFrom(start),
1155
+ };
1156
+ }
1157
+ parseJsonType() {
1158
+ const start = this.peek().start;
1159
+ const kwTok = this.advance();
1160
+ let fields;
1161
+ if (this.check(TokenKind.LBrace)) {
1162
+ this.advance();
1163
+ fields = [];
1164
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1165
+ const t = this.peek();
1166
+ const k = kw(t);
1167
+ if (k === 'note') {
1168
+ fields.push(this.parseNoteBlockOrSetting());
1169
+ }
1170
+ else if (t.kind === TokenKind.Tilde) {
1171
+ fields.push(this.parsePartialInjection());
1172
+ }
1173
+ else {
1174
+ fields.push(this.parseFieldDeclaration());
1175
+ }
1176
+ }
1177
+ this.expect(TokenKind.RBrace, "Expected '}' closing json block");
1178
+ }
1179
+ return {
1180
+ kind: 'JsonType',
1181
+ keyword: kwTok.text.toLowerCase(),
1182
+ fields,
1183
+ span: this.spanFrom(start),
1184
+ };
1185
+ }
1186
+ /**
1187
+ * Scalar type or named-type reference. Both look like an Identifier with
1188
+ * optional `(p, s)` parameter list. The distinction is made later at the
1189
+ * semantic-analysis stage (named types are user-declared identifiers that
1190
+ * resolve to a TypeDeclaration; scalars are the open set of built-ins).
1191
+ *
1192
+ * Resolution heuristic for the PoC: if the identifier's lowercase form is
1193
+ * a known SQL/BSON scalar name, we tag ScalarType; otherwise we'd ideally
1194
+ * defer to the semantic pass. For the PoC we always emit ScalarType for
1195
+ * common scalar names and ScalarType for everything else too; callers
1196
+ * that need to distinguish can post-process.
1197
+ *
1198
+ * Actually a cleaner choice: emit ScalarType when there are parameters
1199
+ * (no named type takes `(p,s)`), and otherwise emit NamedTypeReference
1200
+ * iff the name's first character is uppercase (heuristic) -- but that
1201
+ * conflicts with Decimal128 etc. So: always emit ScalarType; the
1202
+ * semantic-analysis pass walks Type declarations and rewrites scalars
1203
+ * whose names resolve to user types as NamedTypeReference. The PoC keeps
1204
+ * the AST shape consistent regardless.
1205
+ */
1206
+ parseScalarOrNamedType() {
1207
+ const start = this.peek().start;
1208
+ const t = this.peek();
1209
+ if (t.kind !== TokenKind.Identifier && t.kind !== TokenKind.QuotedIdentifier) {
1210
+ throw new ParseError(`Expected type name, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
1211
+ }
1212
+ this.advance();
1213
+ const name = t.kind === TokenKind.QuotedIdentifier ? (t.value ?? '') : t.text;
1214
+ let params;
1215
+ if (this.check(TokenKind.LParen)) {
1216
+ this.advance();
1217
+ params = [];
1218
+ if (!this.check(TokenKind.RParen)) {
1219
+ params.push(this.parseTypeParam());
1220
+ while (this.match(TokenKind.Comma)) {
1221
+ params.push(this.parseTypeParam());
1222
+ }
1223
+ }
1224
+ this.expect(TokenKind.RParen, "Expected ')' closing type parameters");
1225
+ }
1226
+ return {
1227
+ kind: 'ScalarType',
1228
+ name,
1229
+ params,
1230
+ span: this.spanFrom(start),
1231
+ };
1232
+ }
1233
+ parseTypeParam() {
1234
+ const t = this.peek();
1235
+ if (t.kind === TokenKind.NumberLiteral || t.kind === TokenKind.StringLiteral || t.kind === TokenKind.Identifier) {
1236
+ this.advance();
1237
+ return t.value ?? t.text;
1238
+ }
1239
+ throw new ParseError(`Expected type parameter, got ${t.kind}`, t.start);
1240
+ }
1241
+ /* ----- Type declaration (§13) ----- */
1242
+ parseTypeDecl() {
1243
+ const start = this.peek().start;
1244
+ this.advance(); // Type
1245
+ const name = this.parseIdentLikeName('type name');
1246
+ // After `Type <Name>`, the next token disambiguates the form:
1247
+ //
1248
+ // { ... } v0.1 object form, no pre-body settings
1249
+ // [ settings ] { ... } v0.1 object form, pre-body settings (permissive)
1250
+ // typeExpression v0.2 scalar form (spec §14.7)
1251
+ // typeExpression [ settings ] v0.2 scalar form with field-level settings
1252
+ //
1253
+ // Note that LBrace and LBracket are distinct from any start-of-type-expression
1254
+ // token (Identifier, scalar/bson type keywords, structural type keywords like
1255
+ // `object`, `array`, `oneOf`, etc.), so the dispatch is unambiguous from
1256
+ // peek(0) alone.
1257
+ if (this.check(TokenKind.LBrace)) {
1258
+ // v0.1 object form, no pre-body settings.
1259
+ return this.finishObjectTypeDecl(start, name, /* settings */ []);
1260
+ }
1261
+ if (this.check(TokenKind.LBracket)) {
1262
+ // v0.1 object form with pre-body settings (permissive shape; not used in
1263
+ // any current example or spec text but historically accepted).
1264
+ const settings = this.maybeSettingsBlock();
1265
+ return this.finishObjectTypeDecl(start, name, settings);
1266
+ }
1267
+ // Anything else is the v0.2 scalar form. parseTypeExpression handles
1268
+ // scalars, BSON types, named-type references, and the parameterized
1269
+ // forms like `decimal(10, 2)`. It also handles structural type
1270
+ // expressions like `array(int)` -- the spec calls this "scalar" because
1271
+ // that's the typical use case, but the syntactic form supports any
1272
+ // type expression as the base.
1273
+ const scalarBase = this.parseTypeExpression();
1274
+ const settings = this.maybeSettingsBlock();
1275
+ return {
1276
+ kind: 'TypeDeclaration',
1277
+ name,
1278
+ scalarBase,
1279
+ settings,
1280
+ body: [],
1281
+ span: this.spanFrom(start),
1282
+ };
1283
+ }
1284
+ /**
1285
+ * Finish parsing a v0.1 object-form Type after the name (and optional
1286
+ * pre-body settings) have been consumed. Handles the `{ ...body }` part.
1287
+ */
1288
+ finishObjectTypeDecl(start, name, settings) {
1289
+ this.expect(TokenKind.LBrace, "Expected '{' after Type name");
1290
+ const body = [];
1291
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1292
+ const t = this.peek();
1293
+ const k = kw(t);
1294
+ if (k === 'note') {
1295
+ body.push(this.parseNoteBlockOrSetting());
1296
+ }
1297
+ else if (t.kind === TokenKind.Tilde) {
1298
+ body.push(this.parsePartialInjection());
1299
+ }
1300
+ else {
1301
+ body.push(this.parseFieldDeclaration());
1302
+ }
1303
+ }
1304
+ this.expect(TokenKind.RBrace, "Expected '}' closing Type");
1305
+ return {
1306
+ kind: 'TypeDeclaration',
1307
+ name,
1308
+ settings,
1309
+ body,
1310
+ span: this.spanFrom(start),
1311
+ };
1312
+ }
1313
+ /* ----- Edge ----- */
1314
+ parseEdge() {
1315
+ const start = this.peek().start;
1316
+ this.advance(); // Edge
1317
+ const name = this.parseIdentLikeName('edge name');
1318
+ const settings = this.maybeSettingsBlock();
1319
+ this.expect(TokenKind.LBrace, "Expected '{' after Edge settings");
1320
+ const body = this.parseEntityBody();
1321
+ this.expect(TokenKind.RBrace, "Expected '}' closing Edge");
1322
+ return {
1323
+ kind: 'EdgeDeclaration',
1324
+ name,
1325
+ settings,
1326
+ body,
1327
+ span: this.spanFrom(start),
1328
+ };
1329
+ }
1330
+ /* ----- View ----- */
1331
+ parseView() {
1332
+ const start = this.peek().start;
1333
+ this.advance(); // View
1334
+ const name = this.parseIdentLikeName('view name');
1335
+ const settings = this.maybeSettingsBlock();
1336
+ this.expect(TokenKind.LBrace, "Expected '{' after View name");
1337
+ const body = [];
1338
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1339
+ const t = this.peek();
1340
+ const k = kw(t);
1341
+ if (k === 'note') {
1342
+ body.push(this.parseNoteBlockOrSetting());
1343
+ }
1344
+ else if (k === 'source_query') {
1345
+ body.push(this.parseSourceQueryItem());
1346
+ }
1347
+ else {
1348
+ body.push(this.parseFieldDeclaration());
1349
+ }
1350
+ }
1351
+ this.expect(TokenKind.RBrace, "Expected '}' closing View");
1352
+ return {
1353
+ kind: 'ViewDeclaration',
1354
+ name,
1355
+ settings,
1356
+ body,
1357
+ span: this.spanFrom(start),
1358
+ };
1359
+ }
1360
+ parseSourceQueryItem() {
1361
+ const start = this.peek().start;
1362
+ this.advance(); // source_query
1363
+ this.expect(TokenKind.Colon, "Expected ':' after source_query");
1364
+ const t = this.peek();
1365
+ if (t.kind !== TokenKind.StringLiteral && t.kind !== TokenKind.MultilineString) {
1366
+ throw new ParseError('Expected string after source_query:', t.start);
1367
+ }
1368
+ this.advance();
1369
+ return {
1370
+ kind: 'SourceQueryItem',
1371
+ query: t.value ?? '',
1372
+ span: this.spanFrom(start),
1373
+ };
1374
+ }
1375
+ /* ----- Enum ----- */
1376
+ parseEnum() {
1377
+ const start = this.peek().start;
1378
+ const kwTok = this.advance(); // enum
1379
+ const name = this.parseIdentLikeName('enum name');
1380
+ this.expect(TokenKind.LBrace, "Expected '{' after enum name");
1381
+ const values = [];
1382
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1383
+ const vStart = this.peek().start;
1384
+ const vt = this.peek();
1385
+ let vname;
1386
+ let vquoted = false;
1387
+ if (vt.kind === TokenKind.StringLiteral) {
1388
+ // legacy form: 'A+'
1389
+ this.advance();
1390
+ vname = vt.value ?? '';
1391
+ vquoted = true;
1392
+ }
1393
+ else if (vt.kind === TokenKind.QuotedIdentifier) {
1394
+ this.advance();
1395
+ vname = vt.value ?? '';
1396
+ vquoted = true;
1397
+ }
1398
+ else if (vt.kind === TokenKind.Identifier) {
1399
+ this.advance();
1400
+ vname = vt.text;
1401
+ }
1402
+ else {
1403
+ throw new ParseError(`Expected enum value, got ${vt.kind}`, vt.start);
1404
+ }
1405
+ const settings = this.maybeSettingsBlock();
1406
+ values.push({
1407
+ kind: 'EnumValue',
1408
+ name: vname,
1409
+ nameQuoted: vquoted,
1410
+ settings,
1411
+ span: this.spanFrom(vStart),
1412
+ });
1413
+ }
1414
+ this.expect(TokenKind.RBrace, "Expected '}' closing enum");
1415
+ return {
1416
+ kind: 'EnumDeclaration',
1417
+ keywordCasing: kwTok.text,
1418
+ name,
1419
+ values,
1420
+ span: this.spanFrom(start),
1421
+ };
1422
+ }
1423
+ /* ----- Ref ----- */
1424
+ parseRef() {
1425
+ const start = this.peek().start;
1426
+ this.advance(); // Ref
1427
+ let name;
1428
+ if (this.check(TokenKind.Identifier)) {
1429
+ // Could be the optional name, OR it could be the start of a refSpec.
1430
+ // The discriminator: if the next token after the identifier is ':' or '{',
1431
+ // it's a named Ref. Otherwise the identifier is the first path of a
1432
+ // long-form refSpec inside braces -- but the grammar always requires
1433
+ // braces for the body in long form, so a Ref starting `Ref word ...`
1434
+ // where word is not followed by `:` or `{` is malformed.
1435
+ // For Ref: ... and Ref name: ..., we handle both by looking ahead.
1436
+ const id = this.peek();
1437
+ const next = this.peek(1);
1438
+ if (next.kind === TokenKind.Colon || next.kind === TokenKind.LBrace) {
1439
+ this.advance();
1440
+ name = id.text;
1441
+ }
1442
+ }
1443
+ let spec;
1444
+ let settings = [];
1445
+ if (this.match(TokenKind.Colon)) {
1446
+ // short form: `Ref: a > b [settings]`
1447
+ spec = this.parseRefSpec();
1448
+ settings = this.maybeSettingsBlock();
1449
+ }
1450
+ else if (this.match(TokenKind.LBrace)) {
1451
+ // long form: `Ref name { a > b }`
1452
+ spec = this.parseRefSpec();
1453
+ this.expect(TokenKind.RBrace, "Expected '}' closing Ref body");
1454
+ }
1455
+ else {
1456
+ const t = this.peek();
1457
+ throw new ParseError("Expected ':' or '{' after Ref", t.start);
1458
+ }
1459
+ return {
1460
+ kind: 'RefDeclaration',
1461
+ name,
1462
+ spec,
1463
+ settings,
1464
+ span: this.spanFrom(start),
1465
+ };
1466
+ }
1467
+ parseRefSpec() {
1468
+ const start = this.peek().start;
1469
+ const source = this.parseRefEndpoint();
1470
+ const op = this.parseCardinalityOperator();
1471
+ const target = this.parseRefEndpoint();
1472
+ return {
1473
+ kind: 'RefSpec',
1474
+ source,
1475
+ operator: op,
1476
+ target,
1477
+ span: this.spanFrom(start),
1478
+ };
1479
+ }
1480
+ parseCardinalityOperator() {
1481
+ const t = this.peek();
1482
+ if (t.kind === TokenKind.LAngle) {
1483
+ this.advance();
1484
+ return '<';
1485
+ }
1486
+ if (t.kind === TokenKind.RAngle) {
1487
+ this.advance();
1488
+ return '>';
1489
+ }
1490
+ if (t.kind === TokenKind.Minus) {
1491
+ this.advance();
1492
+ return '-';
1493
+ }
1494
+ if (t.kind === TokenKind.ManyToMany) {
1495
+ this.advance();
1496
+ return '<>';
1497
+ }
1498
+ throw new ParseError(`Expected cardinality operator, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
1499
+ }
1500
+ parseRefEndpoint() {
1501
+ const start = this.peek().start;
1502
+ // Composite FK form: `entity.(field1, field2)` -- detect by looking ahead
1503
+ // for a `.(` after the initial identifier path.
1504
+ const segments = this.parsePathSegments();
1505
+ // After segments, if the next token is `.(`, parse composite list
1506
+ let compositeFields;
1507
+ if (this.check(TokenKind.Dot) && this.peek(1).kind === TokenKind.LParen) {
1508
+ this.advance(); // .
1509
+ this.advance(); // (
1510
+ compositeFields = [];
1511
+ compositeFields.push(this.expect(TokenKind.Identifier, 'Expected field name').text);
1512
+ while (this.match(TokenKind.Comma)) {
1513
+ compositeFields.push(this.expect(TokenKind.Identifier, 'Expected field name').text);
1514
+ }
1515
+ this.expect(TokenKind.RParen, "Expected ')' closing composite field list");
1516
+ }
1517
+ return {
1518
+ kind: 'RefEndpoint',
1519
+ path: segments,
1520
+ compositeFields,
1521
+ span: this.spanFrom(start),
1522
+ };
1523
+ }
1524
+ /**
1525
+ * Parse a dotted path with the §18 segment vocabulary:
1526
+ *
1527
+ * IDENTIFIER -- a field segment
1528
+ * .IDENTIFIER -- field
1529
+ * .[N] -- array index (positional)
1530
+ * .[*] -- array wildcard (via ArrayWildcard token)
1531
+ * ."quoted name" -- quoted-identifier field
1532
+ * .["literal key"] -- map literal key
1533
+ *
1534
+ * We start by consuming an identifier/qualified head, then walk pathTail.
1535
+ * The JSONPath-alias forms `[N]`, `[*]` without a leading dot are
1536
+ * recognized as well; they normalize to the dot-prefixed form.
1537
+ *
1538
+ * For the PoC we stop at the first token that doesn't continue a path
1539
+ * (e.g., a cardinality operator, a comma, a settings bracket).
1540
+ */
1541
+ parsePathSegments() {
1542
+ const segments = [];
1543
+ const headStart = this.peek().start;
1544
+ const headTok = this.expect(TokenKind.Identifier, 'Expected path start identifier');
1545
+ segments.push({
1546
+ kind: 'PathField',
1547
+ name: headTok.text,
1548
+ span: {
1549
+ start: headStart,
1550
+ end: headTok.end,
1551
+ },
1552
+ });
1553
+ // Walk tail
1554
+ while (true) {
1555
+ const t = this.peek();
1556
+ // Stop before a composite `.(`
1557
+ if (t.kind === TokenKind.Dot && this.peek(1).kind === TokenKind.LParen) {
1558
+ break;
1559
+ }
1560
+ if (t.kind === TokenKind.Dot) {
1561
+ this.advance();
1562
+ const next = this.peek();
1563
+ const segStart = next.start;
1564
+ if (next.kind === TokenKind.Identifier) {
1565
+ this.advance();
1566
+ segments.push({
1567
+ kind: 'PathField',
1568
+ name: next.text,
1569
+ span: this.spanFrom(segStart),
1570
+ });
1571
+ }
1572
+ else if (next.kind === TokenKind.QuotedIdentifier) {
1573
+ this.advance();
1574
+ segments.push({
1575
+ kind: 'PathField',
1576
+ name: next.value ?? '',
1577
+ span: this.spanFrom(segStart),
1578
+ });
1579
+ }
1580
+ else if (next.kind === TokenKind.LBracket) {
1581
+ // .[N] or .["literal key"]
1582
+ this.advance();
1583
+ const inner = this.peek();
1584
+ if (inner.kind === TokenKind.NumberLiteral) {
1585
+ this.advance();
1586
+ this.expect(TokenKind.RBracket, "Expected ']' after array index");
1587
+ segments.push({
1588
+ kind: 'PathArrayIndex',
1589
+ index: parseInt(inner.text, 10),
1590
+ span: this.spanFrom(segStart),
1591
+ });
1592
+ }
1593
+ else if (inner.kind === TokenKind.StringLiteral) {
1594
+ this.advance();
1595
+ this.expect(TokenKind.RBracket, "Expected ']' after map key");
1596
+ segments.push({
1597
+ kind: 'PathMapKey',
1598
+ key: inner.value ?? '',
1599
+ span: this.spanFrom(segStart),
1600
+ });
1601
+ }
1602
+ else {
1603
+ throw new ParseError(`Unexpected token inside path bracket: ${inner.kind}`, inner.start);
1604
+ }
1605
+ }
1606
+ else if (next.kind === TokenKind.ArrayWildcard) {
1607
+ // .[*]
1608
+ this.advance();
1609
+ segments.push({
1610
+ kind: 'PathArrayWildcard',
1611
+ span: this.spanFrom(segStart),
1612
+ });
1613
+ }
1614
+ else {
1615
+ throw new ParseError(`Unexpected token after '.' in path: ${next.kind} ${JSON.stringify(next.text)}`, next.start);
1616
+ }
1617
+ }
1618
+ else if (t.kind === TokenKind.ArrayWildcard) {
1619
+ // JSONPath-alias: `[*]` immediately after a segment
1620
+ const segStart = t.start;
1621
+ this.advance();
1622
+ segments.push({
1623
+ kind: 'PathArrayWildcard',
1624
+ span: this.spanFrom(segStart),
1625
+ });
1626
+ }
1627
+ else if (t.kind === TokenKind.LBracket && this.peek(1).kind === TokenKind.NumberLiteral && this.peek(2).kind === TokenKind.RBracket) {
1628
+ // JSONPath-alias: `[N]` immediately after a segment
1629
+ const segStart = t.start;
1630
+ this.advance(); // [
1631
+ const numTok = this.advance();
1632
+ this.advance(); // ]
1633
+ segments.push({
1634
+ kind: 'PathArrayIndex',
1635
+ index: parseInt(numTok.text, 10),
1636
+ span: this.spanFrom(segStart),
1637
+ });
1638
+ }
1639
+ else {
1640
+ break;
1641
+ }
1642
+ }
1643
+ return segments;
1644
+ }
1645
+ /* ----- TablePartial / TableGroup ----- */
1646
+ parseTablePartial() {
1647
+ const start = this.peek().start;
1648
+ this.advance(); // TablePartial
1649
+ const name = this.parseIdentLikeName('TablePartial name');
1650
+ const settings = this.maybeSettingsBlock();
1651
+ this.expect(TokenKind.LBrace, "Expected '{' after TablePartial name");
1652
+ const body = this.parseEntityBody();
1653
+ this.expect(TokenKind.RBrace, "Expected '}' closing TablePartial");
1654
+ return {
1655
+ kind: 'TablePartialDeclaration',
1656
+ name,
1657
+ settings,
1658
+ body,
1659
+ span: this.spanFrom(start),
1660
+ };
1661
+ }
1662
+ parseTableGroup() {
1663
+ const start = this.peek().start;
1664
+ this.advance(); // TableGroup
1665
+ const name = this.parseIdentLikeName('TableGroup name');
1666
+ const settings = this.maybeSettingsBlock();
1667
+ this.expect(TokenKind.LBrace, "Expected '{' after TableGroup name");
1668
+ const members = [];
1669
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1670
+ const t = this.peek();
1671
+ if (t.kind === TokenKind.Identifier) {
1672
+ this.advance();
1673
+ let n = t.text;
1674
+ while (this.check(TokenKind.Dot)) {
1675
+ this.advance();
1676
+ const next = this.expect(TokenKind.Identifier, 'Expected identifier after dot');
1677
+ n += `.${next.text}`;
1678
+ }
1679
+ members.push(n);
1680
+ // optional trailing semicolons in some sources
1681
+ this.match(TokenKind.Semicolon);
1682
+ }
1683
+ else if (t.kind === TokenKind.Semicolon || t.kind === TokenKind.Comma) {
1684
+ this.advance();
1685
+ }
1686
+ else {
1687
+ throw new ParseError(`Unexpected token in TableGroup: ${t.kind}`, t.start);
1688
+ }
1689
+ }
1690
+ this.expect(TokenKind.RBrace, "Expected '}' closing TableGroup");
1691
+ return {
1692
+ kind: 'TableGroupDeclaration',
1693
+ name,
1694
+ settings,
1695
+ members,
1696
+ span: this.spanFrom(start),
1697
+ };
1698
+ }
1699
+ /* ----- Indexes ----- */
1700
+ parseIndexes() {
1701
+ const start = this.peek().start;
1702
+ this.advance(); // indexes
1703
+ this.expect(TokenKind.LBrace, "Expected '{' after indexes");
1704
+ const entries = [];
1705
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1706
+ entries.push(this.parseIndexEntry());
1707
+ }
1708
+ this.expect(TokenKind.RBrace, "Expected '}' closing indexes");
1709
+ return {
1710
+ kind: 'IndexesBlock',
1711
+ entries,
1712
+ span: this.spanFrom(start),
1713
+ };
1714
+ }
1715
+ parseIndexEntry() {
1716
+ const start = this.peek().start;
1717
+ let components;
1718
+ if (this.check(TokenKind.LParen)) {
1719
+ this.advance(); // (
1720
+ components = [];
1721
+ components.push(this.parseIndexComponent());
1722
+ while (this.match(TokenKind.Comma)) {
1723
+ components.push(this.parseIndexComponent());
1724
+ }
1725
+ this.expect(TokenKind.RParen, "Expected ')' closing composite index");
1726
+ }
1727
+ else {
1728
+ components = [this.parseIndexComponent()];
1729
+ }
1730
+ const settings = this.maybeSettingsBlock();
1731
+ return {
1732
+ kind: 'IndexEntry',
1733
+ components,
1734
+ settings,
1735
+ span: this.spanFrom(start),
1736
+ };
1737
+ }
1738
+ parseIndexComponent() {
1739
+ const t = this.peek();
1740
+ if (t.kind === TokenKind.ExpressionLiteral) {
1741
+ const start = t.start;
1742
+ this.advance();
1743
+ const c = {
1744
+ kind: 'IndexExpressionComponent',
1745
+ expression: t.value ?? '',
1746
+ span: this.spanFrom(start),
1747
+ };
1748
+ return c;
1749
+ }
1750
+ const start = this.peek().start;
1751
+ const path = this.parsePathSegments();
1752
+ const c = {
1753
+ kind: 'IndexPathComponent',
1754
+ path,
1755
+ span: this.spanFrom(start),
1756
+ };
1757
+ return c;
1758
+ }
1759
+ /* ----- Checks (spec §10, new in v0.2) ----- */
1760
+ parseChecks() {
1761
+ const start = this.peek().start;
1762
+ this.advance(); // checks
1763
+ this.expect(TokenKind.LBrace, "Expected '{' after checks");
1764
+ const entries = [];
1765
+ while (!this.check(TokenKind.RBrace) && !this.check(TokenKind.EOF)) {
1766
+ entries.push(this.parseCheckEntry());
1767
+ }
1768
+ this.expect(TokenKind.RBrace, "Expected '}' closing checks");
1769
+ return {
1770
+ kind: 'ChecksBlock',
1771
+ entries,
1772
+ span: this.spanFrom(start),
1773
+ };
1774
+ }
1775
+ parseCheckEntry() {
1776
+ const start = this.peek().start;
1777
+ const exprTok = this.peek();
1778
+ if (exprTok.kind !== TokenKind.ExpressionLiteral) {
1779
+ throw new ParseError(`Expected backtick-wrapped check expression, got ${exprTok.kind} ${JSON.stringify(exprTok.text)}`, exprTok.start);
1780
+ }
1781
+ this.advance();
1782
+ // The expression is opaque to xDBML per spec §10.3. The value field of
1783
+ // an ExpressionLiteral token already has the surrounding backticks stripped.
1784
+ const expression = exprTok.value ?? '';
1785
+ const settings = this.maybeSettingsBlock();
1786
+ return {
1787
+ kind: 'CheckEntry',
1788
+ expression,
1789
+ settings,
1790
+ span: this.spanFrom(start),
1791
+ };
1792
+ }
1793
+ /* ----- Settings block ----- */
1794
+ maybeSettingsBlock() {
1795
+ if (!this.check(TokenKind.LBracket))
1796
+ return [];
1797
+ this.advance(); // [
1798
+ const settings = [];
1799
+ if (this.check(TokenKind.RBracket)) {
1800
+ this.advance();
1801
+ return settings;
1802
+ }
1803
+ settings.push(this.parseSetting());
1804
+ while (this.match(TokenKind.Comma)) {
1805
+ // tolerate trailing comma
1806
+ if (this.check(TokenKind.RBracket))
1807
+ break;
1808
+ settings.push(this.parseSetting());
1809
+ }
1810
+ this.expect(TokenKind.RBracket, "Expected ']' closing settings");
1811
+ return settings;
1812
+ }
1813
+ /**
1814
+ * A single setting. Forms:
1815
+ * flag -- bare identifier(s), e.g. `pk`, `not null`
1816
+ * name: value -- key/value, e.g. `default: 'x'`, `synonyms: [...]`
1817
+ * ref: > target -- inline ref
1818
+ *
1819
+ * The grammar for "flag" is annoying because `not null` is two words but is
1820
+ * still one flag. We handle the multi-word flags by greedy lowercase prefix
1821
+ * match: `not` followed by `null` becomes `not null`; `primary` followed by
1822
+ * `key` becomes `primary key`.
1823
+ */
1824
+ parseSetting() {
1825
+ const start = this.peek().start;
1826
+ const t = this.peek();
1827
+ if (t.kind !== TokenKind.Identifier && t.kind !== TokenKind.QuotedIdentifier) {
1828
+ throw new ParseError(`Expected setting, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
1829
+ }
1830
+ // Two-word flag prefixes
1831
+ if (t.kind === TokenKind.Identifier) {
1832
+ const lower = t.text.toLowerCase();
1833
+ if (lower === 'not' && kw(this.peek(1)) === 'null') {
1834
+ this.advance();
1835
+ this.advance();
1836
+ return {
1837
+ kind: 'Setting',
1838
+ name: 'not null',
1839
+ nameSource: 'not null',
1840
+ value: null,
1841
+ span: this.spanFrom(start),
1842
+ };
1843
+ }
1844
+ if (lower === 'primary' && kw(this.peek(1)) === 'key') {
1845
+ this.advance();
1846
+ this.advance();
1847
+ return {
1848
+ kind: 'Setting',
1849
+ name: 'primary key',
1850
+ nameSource: 'primary key',
1851
+ value: null,
1852
+ span: this.spanFrom(start),
1853
+ };
1854
+ }
1855
+ }
1856
+ // Read the name token (single-word for now)
1857
+ const nameTok = this.advance();
1858
+ const nameSource = nameTok.kind === TokenKind.QuotedIdentifier ? (nameTok.value ?? '') : nameTok.text;
1859
+ const lower = nameSource.toLowerCase();
1860
+ // If next token is not a colon, this is a pure flag setting.
1861
+ if (!this.match(TokenKind.Colon)) {
1862
+ // Spec §8: `required` is accepted as a synonym for `not null` and
1863
+ // parsers MUST normalize it. `name` becomes the canonical form so
1864
+ // every downstream consumer (inspector REQUIRED badge, layout's
1865
+ // required-flag detection, generators) checks one value.
1866
+ // `nameSource` keeps the user's original spelling so the settings
1867
+ // table renders what was typed and round-tripping the AST back to
1868
+ // source preserves the author's wording.
1869
+ const canonicalName = lower === 'required' ? 'not null' : lower;
1870
+ return {
1871
+ kind: 'Setting',
1872
+ name: canonicalName,
1873
+ nameSource,
1874
+ value: null,
1875
+ span: this.spanFrom(start),
1876
+ };
1877
+ }
1878
+ // `ref: > target` form
1879
+ if (lower === 'ref') {
1880
+ const op = this.parseCardinalityOperator();
1881
+ const target = this.parseRefEndpoint();
1882
+ const v = {
1883
+ kind: 'RefValue',
1884
+ operator: op,
1885
+ target,
1886
+ span: this.spanFrom(start),
1887
+ };
1888
+ return {
1889
+ kind: 'Setting',
1890
+ name: 'ref',
1891
+ nameSource,
1892
+ value: v,
1893
+ span: this.spanFrom(start),
1894
+ };
1895
+ }
1896
+ const value = this.parseSettingValue();
1897
+ return {
1898
+ kind: 'Setting',
1899
+ name: lower,
1900
+ nameSource,
1901
+ value,
1902
+ span: this.spanFrom(start),
1903
+ };
1904
+ }
1905
+ /**
1906
+ * A setting value. Open-vocabulary:
1907
+ * string literal, multi-line string, number, boolean, null, identifier
1908
+ * (or dotted identifier path), expression literal, list `[...]`
1909
+ */
1910
+ parseSettingValue() {
1911
+ const start = this.peek().start;
1912
+ const t = this.peek();
1913
+ if (t.kind === TokenKind.StringLiteral || t.kind === TokenKind.MultilineString) {
1914
+ this.advance();
1915
+ const v = {
1916
+ kind: 'StringValue',
1917
+ value: t.value ?? '',
1918
+ multiline: t.kind === TokenKind.MultilineString,
1919
+ span: this.spanFrom(start),
1920
+ };
1921
+ return v;
1922
+ }
1923
+ if (t.kind === TokenKind.NumberLiteral) {
1924
+ this.advance();
1925
+ const v = {
1926
+ kind: 'NumberValue',
1927
+ value: t.text,
1928
+ span: this.spanFrom(start),
1929
+ };
1930
+ return v;
1931
+ }
1932
+ // negative number: `-3`
1933
+ if (t.kind === TokenKind.Minus && this.peek(1).kind === TokenKind.NumberLiteral) {
1934
+ this.advance();
1935
+ const numTok = this.advance();
1936
+ const v = {
1937
+ kind: 'NumberValue',
1938
+ value: `-${numTok.text}`,
1939
+ span: this.spanFrom(start),
1940
+ };
1941
+ return v;
1942
+ }
1943
+ if (t.kind === TokenKind.ExpressionLiteral) {
1944
+ this.advance();
1945
+ const v = {
1946
+ kind: 'ExpressionValue',
1947
+ expression: t.value ?? '',
1948
+ span: this.spanFrom(start),
1949
+ };
1950
+ return v;
1951
+ }
1952
+ if (t.kind === TokenKind.LBracket) {
1953
+ this.advance();
1954
+ const items = [];
1955
+ if (!this.check(TokenKind.RBracket)) {
1956
+ items.push(this.parseSettingValue());
1957
+ while (this.match(TokenKind.Comma)) {
1958
+ if (this.check(TokenKind.RBracket))
1959
+ break;
1960
+ items.push(this.parseSettingValue());
1961
+ }
1962
+ }
1963
+ this.expect(TokenKind.RBracket, "Expected ']' closing list value");
1964
+ const v = {
1965
+ kind: 'ListValue',
1966
+ items,
1967
+ span: this.spanFrom(start),
1968
+ };
1969
+ return v;
1970
+ }
1971
+ if (t.kind === TokenKind.Identifier || t.kind === TokenKind.QuotedIdentifier) {
1972
+ const lower = t.kind === TokenKind.Identifier ? t.text.toLowerCase() : '';
1973
+ if (lower === 'true' || lower === 'false') {
1974
+ this.advance();
1975
+ return {
1976
+ kind: 'BooleanValue',
1977
+ value: lower === 'true',
1978
+ span: this.spanFrom(start),
1979
+ };
1980
+ }
1981
+ if (lower === 'null') {
1982
+ this.advance();
1983
+ return {
1984
+ kind: 'NullValue',
1985
+ span: this.spanFrom(start),
1986
+ };
1987
+ }
1988
+ // Multi-word identifier values: `set null`, `no action`, `set default`
1989
+ // (referential-action values used in delete/update settings).
1990
+ this.advance();
1991
+ let value = t.kind === TokenKind.QuotedIdentifier ? (t.value ?? '') : t.text;
1992
+ // Greedy continuation: identifier followed by identifier(s) without
1993
+ // intervening punctuation are joined with a space. We stop at any
1994
+ // delimiter, including `,` `]` `>` etc.
1995
+ while (this.peek().kind === TokenKind.Identifier) {
1996
+ const next = this.peek();
1997
+ const nextLower = next.text.toLowerCase();
1998
+ // Don't pull in `null` if it's a separate value; but in
1999
+ // `default: null` the lone identifier `null` was already handled above.
2000
+ // For value contexts, we allow `set null`, `set default`, `no action`.
2001
+ if ((value.toLowerCase() === 'set' && (nextLower === 'null' || nextLower === 'default'))
2002
+ || (value.toLowerCase() === 'no' && nextLower === 'action')) {
2003
+ this.advance();
2004
+ value += ` ${next.text}`;
2005
+ }
2006
+ else {
2007
+ // Continue dotted identifiers: `core.users`
2008
+ break;
2009
+ }
2010
+ }
2011
+ // Dotted identifier continuation: `Oracle`, `core.users`
2012
+ while (this.check(TokenKind.Dot)) {
2013
+ this.advance();
2014
+ const next = this.expect(TokenKind.Identifier, 'Expected identifier after dot');
2015
+ value += `.${next.text}`;
2016
+ }
2017
+ const v = {
2018
+ kind: 'IdentifierValue',
2019
+ value,
2020
+ span: this.spanFrom(start),
2021
+ };
2022
+ return v;
2023
+ }
2024
+ throw new ParseError(`Expected setting value, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
2025
+ }
2026
+ /* ----- Generic ident name parser ----- */
2027
+ parseIdentLikeName(what) {
2028
+ const t = this.peek();
2029
+ if (t.kind === TokenKind.Identifier) {
2030
+ this.advance();
2031
+ return t.text;
2032
+ }
2033
+ if (t.kind === TokenKind.QuotedIdentifier) {
2034
+ this.advance();
2035
+ return t.value ?? '';
2036
+ }
2037
+ throw new ParseError(`Expected ${what}, got ${t.kind} ${JSON.stringify(t.text)}`, t.start);
2038
+ }
2039
+ }
2040
+ /* -------------------------------------------------------------------------
2041
+ * Public API
2042
+ * ----------------------------------------------------------------------- */
2043
+ /**
2044
+ * Parse xDBML source.
2045
+ *
2046
+ * - 1-argument form `parse(source)` parses self-contained documents (any
2047
+ * module directive must carry an inline clone block; reference-only
2048
+ * directives throw).
2049
+ * - 2-argument form `parse(source, options)` accepts a `readFile`
2050
+ * resolver for cross-file `use`/`reuse` directives and a `filePath`
2051
+ * identifying the source for relative-path resolution. See
2052
+ * `ParseOptions` for the full shape.
2053
+ *
2054
+ * The function is fully synchronous. Async file loading and incremental
2055
+ * resolution are intentionally out of scope -- callers needing async I/O
2056
+ * should pre-load their module graph and supply a `readFile` callback
2057
+ * that returns from an in-memory map.
2058
+ */
2059
+ export function parse(source, options = {}) {
2060
+ const tokens = tokenize(source);
2061
+ // The initial resolution stack contains the importer's own file path
2062
+ // (so a file that tries to reuse itself triggers cycle detection at
2063
+ // the outer level too). If no filePath is provided, the stack is empty.
2064
+ const initialStack = new Set();
2065
+ if (options.filePath)
2066
+ initialStack.add(options.filePath);
2067
+ return new Parser(tokens, options, initialStack, 0).parseDocument();
2068
+ }
2069
+ /**
2070
+ * Internal `ParseFn` used by the module resolver to recursively parse a
2071
+ * referenced file. Threads the resolution stack and depth so cycle
2072
+ * detection and the depth limit cover the full transitive graph.
2073
+ *
2074
+ * NOTE: this is the recursive entry point invoked by `resolveImport()`.
2075
+ * It differs from the public `parse()` in two ways: (1) it takes the
2076
+ * full resolution-stack / depth context, and (2) it doesn't re-add
2077
+ * options.filePath to the stack (the caller already did so when
2078
+ * widening the stack with the resolved path of the referenced file).
2079
+ */
2080
+ const recursiveParse = (source, options, resolutionStack, depth) => {
2081
+ const tokens = tokenize(source);
2082
+ return new Parser(tokens, options, resolutionStack, depth).parseDocument();
2083
+ };