@xdbml/parse 0.1.0-poc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/lexer.js ADDED
@@ -0,0 +1,549 @@
1
+ /**
2
+ * xDBML lexer.
3
+ *
4
+ * Produces a stream of tokens with line/column positions. Keywords are NOT
5
+ * recognized as distinct token kinds at the lexer level. Identifiers carry
6
+ * their source value, and the parser interprets them as keywords by
7
+ * lowercased string comparison (per spec §3.8: keywords are
8
+ * case-insensitive, identifiers are case-sensitive — both end up as
9
+ * IDENTIFIER tokens here, with the parser making the keyword decision).
10
+ *
11
+ * Special multi-character punctuation handled here rather than in the parser:
12
+ * - `<>` is lexed as MANY_TO_MANY (otherwise the parser would see `<` then `>`)
13
+ * - `[*]` is lexed as ARRAY_WILDCARD only when followed-by/preceded-by a dot
14
+ * in a path context; otherwise it's left as `LBRACKET`, `OP(*)`, `RBRACKET`
15
+ * to avoid ambiguity with array settings. In practice the parser only
16
+ * uses ARRAY_WILDCARD inside fieldPath, so we lex `[*]` greedily whenever
17
+ * we see it and let the parser decide based on context.
18
+ * - `'''...'''` lexes as a single STRING_LITERAL with the multiline flag.
19
+ */
20
+ export const TokenKind = {
21
+ Identifier: 'Identifier',
22
+ QuotedIdentifier: 'QuotedIdentifier', // "double-quoted name"
23
+ StringLiteral: 'StringLiteral',
24
+ MultilineString: 'MultilineString',
25
+ NumberLiteral: 'NumberLiteral',
26
+ ExpressionLiteral: 'ExpressionLiteral', // `backtick-quoted`
27
+ LBrace: 'LBrace',
28
+ RBrace: 'RBrace',
29
+ LBracket: 'LBracket',
30
+ RBracket: 'RBracket',
31
+ LParen: 'LParen',
32
+ RParen: 'RParen',
33
+ Comma: 'Comma',
34
+ Colon: 'Colon',
35
+ Dot: 'Dot',
36
+ Tilde: 'Tilde',
37
+ Semicolon: 'Semicolon',
38
+ LAngle: 'LAngle', // <
39
+ RAngle: 'RAngle', // >
40
+ Minus: 'Minus', // -
41
+ ManyToMany: 'ManyToMany', // <>
42
+ ArrayWildcard: 'ArrayWildcard', // [*]
43
+ /**
44
+ * Standalone `*`. Used by v0.2 module-system import-all directives:
45
+ * `use * from './path'`. Note that `[*]` is a separate token
46
+ * (ArrayWildcard); this Star token is only produced when the asterisk
47
+ * appears outside that context.
48
+ */
49
+ Star: 'Star',
50
+ EOF: 'EOF',
51
+ };
52
+ export class LexError extends Error {
53
+ position;
54
+ constructor(message, position) {
55
+ super(`${message} (line ${position.line}, column ${position.column})`);
56
+ this.position = position;
57
+ }
58
+ }
59
+ export class Lexer {
60
+ text;
61
+ offset = 0;
62
+ line = 1;
63
+ column = 1;
64
+ constructor(text) {
65
+ this.text = text;
66
+ }
67
+ pos() {
68
+ return {
69
+ line: this.line,
70
+ column: this.column,
71
+ offset: this.offset,
72
+ };
73
+ }
74
+ peek(lookahead = 0) {
75
+ return this.text[this.offset + lookahead] ?? '';
76
+ }
77
+ advance() {
78
+ const c = this.text[this.offset];
79
+ this.offset += 1;
80
+ if (c === '\n') {
81
+ this.line += 1;
82
+ this.column = 1;
83
+ }
84
+ else {
85
+ this.column += 1;
86
+ }
87
+ return c;
88
+ }
89
+ matchSeq(seq) {
90
+ for (let i = 0; i < seq.length; i += 1) {
91
+ if (this.text[this.offset + i] !== seq[i]) {
92
+ return false;
93
+ }
94
+ }
95
+ return true;
96
+ }
97
+ isAtEnd() {
98
+ return this.offset >= this.text.length;
99
+ }
100
+ skipTrivia() {
101
+ while (!this.isAtEnd()) {
102
+ const c = this.peek();
103
+ if (c === ' ' || c === '\t' || c === '\r' || c === '\n') {
104
+ this.advance();
105
+ continue;
106
+ }
107
+ if (c === '/' && this.peek(1) === '/') {
108
+ // line comment
109
+ while (!this.isAtEnd() && this.peek() !== '\n') {
110
+ this.advance();
111
+ }
112
+ continue;
113
+ }
114
+ if (c === '/' && this.peek(1) === '*') {
115
+ // block comment
116
+ this.advance();
117
+ this.advance();
118
+ while (!this.isAtEnd() && !(this.peek() === '*' && this.peek(1) === '/')) {
119
+ this.advance();
120
+ }
121
+ if (!this.isAtEnd()) {
122
+ this.advance(); // *
123
+ this.advance(); // /
124
+ }
125
+ continue;
126
+ }
127
+ break;
128
+ }
129
+ }
130
+ isIdentStart(c) {
131
+ return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || c === '_';
132
+ }
133
+ isIdentCont(c) {
134
+ return this.isIdentStart(c) || (c >= '0' && c <= '9');
135
+ }
136
+ isDigit(c) {
137
+ return c >= '0' && c <= '9';
138
+ }
139
+ lexIdentifier() {
140
+ const start = this.pos();
141
+ while (!this.isAtEnd() && this.isIdentCont(this.peek())) {
142
+ this.advance();
143
+ }
144
+ const end = this.pos();
145
+ const text = this.text.substring(start.offset, end.offset);
146
+ return {
147
+ kind: TokenKind.Identifier,
148
+ text,
149
+ start,
150
+ end,
151
+ };
152
+ }
153
+ lexNumber() {
154
+ const start = this.pos();
155
+ // optional leading minus is handled at parser level (it's an operator token)
156
+ while (!this.isAtEnd() && this.isDigit(this.peek())) {
157
+ this.advance();
158
+ }
159
+ if (this.peek() === '.' && this.isDigit(this.peek(1))) {
160
+ this.advance(); // .
161
+ while (!this.isAtEnd() && this.isDigit(this.peek())) {
162
+ this.advance();
163
+ }
164
+ }
165
+ if (this.peek() === 'e' || this.peek() === 'E') {
166
+ this.advance();
167
+ if (this.peek() === '+' || this.peek() === '-') {
168
+ this.advance();
169
+ }
170
+ while (!this.isAtEnd() && this.isDigit(this.peek())) {
171
+ this.advance();
172
+ }
173
+ }
174
+ const end = this.pos();
175
+ return {
176
+ kind: TokenKind.NumberLiteral,
177
+ text: this.text.substring(start.offset, end.offset),
178
+ start,
179
+ end,
180
+ };
181
+ }
182
+ lexString() {
183
+ const start = this.pos();
184
+ // Detect triple-quoted multi-line first
185
+ if (this.matchSeq("'''")) {
186
+ this.advance();
187
+ this.advance();
188
+ this.advance();
189
+ const bodyStart = this.offset;
190
+ while (!this.isAtEnd() && !this.matchSeq("'''")) {
191
+ if (this.peek() === '\\' && this.peek(1) !== '') {
192
+ this.advance();
193
+ this.advance();
194
+ }
195
+ else {
196
+ this.advance();
197
+ }
198
+ }
199
+ if (this.isAtEnd()) {
200
+ throw new LexError('Unterminated triple-quoted string', start);
201
+ }
202
+ const bodyEnd = this.offset;
203
+ this.advance();
204
+ this.advance();
205
+ this.advance();
206
+ const end = this.pos();
207
+ const raw = this.text.substring(bodyStart, bodyEnd);
208
+ return {
209
+ kind: TokenKind.MultilineString,
210
+ text: this.text.substring(start.offset, end.offset),
211
+ value: normalizeMultiline(raw),
212
+ start,
213
+ end,
214
+ };
215
+ }
216
+ // single-quoted
217
+ this.advance(); // opening '
218
+ let value = '';
219
+ while (!this.isAtEnd() && this.peek() !== "'") {
220
+ if (this.peek() === '\\' && this.peek(1) !== '') {
221
+ const esc = this.peek(1);
222
+ if (esc === 'n') {
223
+ value += '\n';
224
+ }
225
+ else if (esc === 't') {
226
+ value += '\t';
227
+ }
228
+ else if (esc === 'r') {
229
+ value += '\r';
230
+ }
231
+ else if (esc === '\\') {
232
+ value += '\\';
233
+ }
234
+ else if (esc === "'") {
235
+ value += "'";
236
+ }
237
+ else {
238
+ value += esc;
239
+ }
240
+ this.advance();
241
+ this.advance();
242
+ }
243
+ else if (this.peek() === '\n') {
244
+ throw new LexError('Unterminated string (newline in single-quoted string)', start);
245
+ }
246
+ else {
247
+ value += this.advance();
248
+ }
249
+ }
250
+ if (this.isAtEnd()) {
251
+ throw new LexError('Unterminated string', start);
252
+ }
253
+ this.advance(); // closing '
254
+ const end = this.pos();
255
+ return {
256
+ kind: TokenKind.StringLiteral,
257
+ text: this.text.substring(start.offset, end.offset),
258
+ value,
259
+ start,
260
+ end,
261
+ };
262
+ }
263
+ lexQuotedIdent() {
264
+ const start = this.pos();
265
+ this.advance(); // opening "
266
+ let value = '';
267
+ while (!this.isAtEnd() && this.peek() !== '"') {
268
+ if (this.peek() === '\\' && this.peek(1) !== '') {
269
+ value += this.peek(1);
270
+ this.advance();
271
+ this.advance();
272
+ }
273
+ else if (this.peek() === '\n') {
274
+ throw new LexError('Unterminated quoted identifier', start);
275
+ }
276
+ else {
277
+ value += this.advance();
278
+ }
279
+ }
280
+ if (this.isAtEnd()) {
281
+ throw new LexError('Unterminated quoted identifier', start);
282
+ }
283
+ this.advance(); // closing "
284
+ const end = this.pos();
285
+ return {
286
+ kind: TokenKind.QuotedIdentifier,
287
+ text: this.text.substring(start.offset, end.offset),
288
+ value,
289
+ start,
290
+ end,
291
+ };
292
+ }
293
+ lexBacktick() {
294
+ const start = this.pos();
295
+ this.advance(); // `
296
+ const bodyStart = this.offset;
297
+ while (!this.isAtEnd() && this.peek() !== '`') {
298
+ this.advance();
299
+ }
300
+ if (this.isAtEnd()) {
301
+ throw new LexError('Unterminated expression literal', start);
302
+ }
303
+ const bodyEnd = this.offset;
304
+ this.advance(); // closing `
305
+ const end = this.pos();
306
+ return {
307
+ kind: TokenKind.ExpressionLiteral,
308
+ text: this.text.substring(start.offset, end.offset),
309
+ value: this.text.substring(bodyStart, bodyEnd),
310
+ start,
311
+ end,
312
+ };
313
+ }
314
+ /** Read a single token. Returns EOF when out of input. */
315
+ nextToken() {
316
+ this.skipTrivia();
317
+ if (this.isAtEnd()) {
318
+ const p = this.pos();
319
+ return {
320
+ kind: TokenKind.EOF,
321
+ text: '',
322
+ start: p,
323
+ end: p,
324
+ };
325
+ }
326
+ const start = this.pos();
327
+ const c = this.peek();
328
+ // Punctuation and multi-character operators
329
+ if (c === '{') {
330
+ this.advance();
331
+ return {
332
+ kind: TokenKind.LBrace,
333
+ text: '{',
334
+ start,
335
+ end: this.pos(),
336
+ };
337
+ }
338
+ if (c === '}') {
339
+ this.advance();
340
+ return {
341
+ kind: TokenKind.RBrace,
342
+ text: '}',
343
+ start,
344
+ end: this.pos(),
345
+ };
346
+ }
347
+ if (c === '[') {
348
+ // Greedy [*] match for wildcards in paths
349
+ if (this.peek(1) === '*' && this.peek(2) === ']') {
350
+ this.advance();
351
+ this.advance();
352
+ this.advance();
353
+ return {
354
+ kind: TokenKind.ArrayWildcard,
355
+ text: '[*]',
356
+ start,
357
+ end: this.pos(),
358
+ };
359
+ }
360
+ this.advance();
361
+ return {
362
+ kind: TokenKind.LBracket,
363
+ text: '[',
364
+ start,
365
+ end: this.pos(),
366
+ };
367
+ }
368
+ if (c === ']') {
369
+ this.advance();
370
+ return {
371
+ kind: TokenKind.RBracket,
372
+ text: ']',
373
+ start,
374
+ end: this.pos(),
375
+ };
376
+ }
377
+ if (c === '(') {
378
+ this.advance();
379
+ return {
380
+ kind: TokenKind.LParen,
381
+ text: '(',
382
+ start,
383
+ end: this.pos(),
384
+ };
385
+ }
386
+ if (c === ')') {
387
+ this.advance();
388
+ return {
389
+ kind: TokenKind.RParen,
390
+ text: ')',
391
+ start,
392
+ end: this.pos(),
393
+ };
394
+ }
395
+ if (c === ',') {
396
+ this.advance();
397
+ return {
398
+ kind: TokenKind.Comma,
399
+ text: ',',
400
+ start,
401
+ end: this.pos(),
402
+ };
403
+ }
404
+ if (c === ':') {
405
+ this.advance();
406
+ return {
407
+ kind: TokenKind.Colon,
408
+ text: ':',
409
+ start,
410
+ end: this.pos(),
411
+ };
412
+ }
413
+ if (c === ';') {
414
+ this.advance();
415
+ return {
416
+ kind: TokenKind.Semicolon,
417
+ text: ';',
418
+ start,
419
+ end: this.pos(),
420
+ };
421
+ }
422
+ if (c === '.') {
423
+ this.advance();
424
+ return {
425
+ kind: TokenKind.Dot,
426
+ text: '.',
427
+ start,
428
+ end: this.pos(),
429
+ };
430
+ }
431
+ if (c === '~') {
432
+ this.advance();
433
+ return {
434
+ kind: TokenKind.Tilde,
435
+ text: '~',
436
+ start,
437
+ end: this.pos(),
438
+ };
439
+ }
440
+ if (c === '*') {
441
+ this.advance();
442
+ return {
443
+ kind: TokenKind.Star,
444
+ text: '*',
445
+ start,
446
+ end: this.pos(),
447
+ };
448
+ }
449
+ if (c === '<') {
450
+ if (this.peek(1) === '>') {
451
+ this.advance();
452
+ this.advance();
453
+ return {
454
+ kind: TokenKind.ManyToMany,
455
+ text: '<>',
456
+ start,
457
+ end: this.pos(),
458
+ };
459
+ }
460
+ this.advance();
461
+ return {
462
+ kind: TokenKind.LAngle,
463
+ text: '<',
464
+ start,
465
+ end: this.pos(),
466
+ };
467
+ }
468
+ if (c === '>') {
469
+ this.advance();
470
+ return {
471
+ kind: TokenKind.RAngle,
472
+ text: '>',
473
+ start,
474
+ end: this.pos(),
475
+ };
476
+ }
477
+ if (c === '-') {
478
+ // Could be a negative number or the one-to-one operator.
479
+ // Disambiguation is positional, so we always emit Minus and let the parser decide.
480
+ this.advance();
481
+ return {
482
+ kind: TokenKind.Minus,
483
+ text: '-',
484
+ start,
485
+ end: this.pos(),
486
+ };
487
+ }
488
+ if (c === "'") {
489
+ return this.lexString();
490
+ }
491
+ if (c === '"') {
492
+ return this.lexQuotedIdent();
493
+ }
494
+ if (c === '`') {
495
+ return this.lexBacktick();
496
+ }
497
+ if (this.isDigit(c)) {
498
+ return this.lexNumber();
499
+ }
500
+ if (this.isIdentStart(c)) {
501
+ return this.lexIdentifier();
502
+ }
503
+ throw new LexError(`Unexpected character: ${JSON.stringify(c)}`, start);
504
+ }
505
+ tokenize() {
506
+ const out = [];
507
+ while (true) {
508
+ const t = this.nextToken();
509
+ out.push(t);
510
+ if (t.kind === TokenKind.EOF) {
511
+ break;
512
+ }
513
+ }
514
+ return out;
515
+ }
516
+ }
517
+ /**
518
+ * Normalize triple-quoted multi-line string content per spec §3.3.
519
+ * Strips leading newline if present, then de-indents based on the
520
+ * minimum indent of non-empty lines.
521
+ */
522
+ function normalizeMultiline(raw) {
523
+ let s = raw;
524
+ if (s.startsWith('\n')) {
525
+ s = s.slice(1);
526
+ }
527
+ else if (s.startsWith('\r\n')) {
528
+ s = s.slice(2);
529
+ }
530
+ // strip trailing whitespace-only on last line that prefixed the closing '''
531
+ s = s.replace(/[ \t]*$/, '');
532
+ const lines = s.split('\n');
533
+ let minIndent = Infinity;
534
+ for (const line of lines) {
535
+ if (line.trim() === '')
536
+ continue;
537
+ const m = line.match(/^[ \t]*/);
538
+ if (m && m[0].length < minIndent) {
539
+ minIndent = m[0].length;
540
+ }
541
+ }
542
+ if (minIndent === Infinity || minIndent === 0) {
543
+ return lines.join('\n');
544
+ }
545
+ return lines.map((l) => l.slice(minIndent)).join('\n');
546
+ }
547
+ export function tokenize(text) {
548
+ return new Lexer(text).tokenize();
549
+ }
@@ -0,0 +1,115 @@
1
+ /**
2
+ * Module-system support utilities.
3
+ *
4
+ * Per parser-design v2, the parser produces a provenance-preserving
5
+ * "Shape B" AST: `ModuleImportDirective` nodes keep imported declarations
6
+ * inside their `clone.statements` field rather than splicing them into the
7
+ * parent's statement list. This preserves the information about which file
8
+ * each declaration came from, useful for navigation, inspector panels,
9
+ * round-tripping back to source text, etc.
10
+ *
11
+ * Many downstream consumers (code generators, diagram renderers, simple
12
+ * walkers) just want a flat list of declarations and don't care about
13
+ * provenance. The `flatten()` helper produces a new XDbmlDocument where
14
+ * `ModuleImportDirective` nodes have been replaced by their `clone.statements`,
15
+ * recursively.
16
+ *
17
+ * Currently P4-only: handles clone blocks. Reference-only directives are
18
+ * rejected at parse time, so `flatten()` doesn't need to do file resolution.
19
+ * In P5, a separate `resolveModules()` function will populate clone blocks
20
+ * from referenced files before `flatten()` runs.
21
+ */
22
+ import type { CloneBlock, ModuleImportDirective, ParseOptions, XDbmlDocument } from './ast.ts';
23
+ /**
24
+ * Produce a new XDbmlDocument with all module-system directives replaced
25
+ * by their clone-block content, recursively.
26
+ *
27
+ * At the top level, each `ModuleImportDirective` is replaced by its
28
+ * `clone.statements` (each statement appears at the same position the
29
+ * directive used to occupy).
30
+ *
31
+ * Inside a Container body, each `ModuleImportDirective` is replaced by its
32
+ * `clone.statements` as `ContainerBodyItem`s. Note that the spec table in
33
+ * §26.6 guarantees that clone-block content for entity/edge/view/enum
34
+ * imports is shape-compatible with `ContainerBodyItem`; other shapes
35
+ * (e.g., a TablePartial clone inside a Container directive) would be
36
+ * semantically invalid per the spec and would surface as a downstream
37
+ * type error rather than being caught here.
38
+ *
39
+ * Field-level imports (spec §26.8) get a special transform: the clone
40
+ * block holds a bare `FieldDeclaration`, which `flatten()` lifts into a
41
+ * synthetic `TypeDeclaration` at file scope. Downstream consumers see
42
+ * a normal Named Type and can use it as a field type without learning
43
+ * about the field-import construct.
44
+ *
45
+ * Provenance information is lost in the flattened view. Consumers that
46
+ * want to know where each declaration came from should walk the original
47
+ * (non-flattened) AST instead.
48
+ */
49
+ export declare function flatten(doc: XDbmlDocument): XDbmlDocument;
50
+ /**
51
+ * A `parse`-like callback. Used by `resolveImport` to recursively parse
52
+ * a referenced file with the same parser configuration and an extended
53
+ * resolution stack. The Parser injects its own bound parse function.
54
+ */
55
+ export type ParseFn = (source: string, options: ParseOptions, resolutionStack: ReadonlySet<string>, depth: number) => XDbmlDocument;
56
+ /**
57
+ * Result of attempting to resolve a directive's referenced file.
58
+ *
59
+ * - `kind: 'resolved'` -- the file was opened, parsed, and a clone
60
+ * block was synthesized
61
+ * - `kind: 'cycle'` -- the resolution chain already contains this
62
+ * file; per spec §26.15 cycles are allowed, so we return an empty
63
+ * clone block and let name resolution (P6+) handle the actual
64
+ * cross-file linking
65
+ * - `kind: 'no-resolver'` -- no `readFile` was supplied; caller should
66
+ * fall back to the P4 rejection
67
+ */
68
+ export type ImportResolution = {
69
+ kind: 'resolved';
70
+ clone: CloneBlock;
71
+ resolvedPath: string;
72
+ } | {
73
+ kind: 'cycle';
74
+ resolvedPath: string;
75
+ } | {
76
+ kind: 'no-resolver';
77
+ };
78
+ /**
79
+ * Resolve a directive's reference. Returns an `ImportResolution` indicating
80
+ * what happened. The caller decides how to act (set `clone`, fail, etc).
81
+ *
82
+ * Path resolution: the directive's `from` is relative to the importer's
83
+ * `filePath`. If `from` doesn't end with `.xdbml`, that extension is
84
+ * appended. Returns the absolute resolved path on success.
85
+ *
86
+ * Cycles: if the resolved path is already in `resolutionStack`, returns
87
+ * `kind: 'cycle'` without recursing.
88
+ *
89
+ * Depth: if `depth + 1 > maxDepth`, throws.
90
+ */
91
+ export declare function resolveImport(directive: ModuleImportDirective, options: ParseOptions, resolutionStack: ReadonlySet<string>, depth: number, parseFn: ParseFn): ImportResolution;
92
+ /** Raised when a `from` source string is structurally disallowed. */
93
+ export declare class ModuleSourceError extends Error {
94
+ constructor(message: string);
95
+ }
96
+ export type ModuleSource = {
97
+ kind: 'relative';
98
+ from: string;
99
+ } | {
100
+ kind: 'url';
101
+ href: string;
102
+ };
103
+ /** True when a resolved key is a remote (https) URL rather than a path. */
104
+ export declare function isUrlKey(s: string): boolean;
105
+ /**
106
+ * Classify a directive's `from` source string. Returns a discriminated
107
+ * union; throws ModuleSourceError for disallowed forms. Pure and sync.
108
+ */
109
+ export declare function classifyModuleSource(from: string): ModuleSource;
110
+ /**
111
+ * Default recursion depth limit for module resolution. Enough for any
112
+ * realistic module graph; small enough to bound stack usage on
113
+ * pathological inputs.
114
+ */
115
+ export declare const DEFAULT_MAX_DEPTH = 8;