@xdbml/parse 0.1.0-poc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ast.d.ts +565 -0
- package/dist/ast.js +10 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +33 -0
- package/dist/keywords.d.ts +45 -0
- package/dist/keywords.js +277 -0
- package/dist/lexer.d.ts +91 -0
- package/dist/lexer.js +549 -0
- package/dist/module-resolver.d.ts +115 -0
- package/dist/module-resolver.js +771 -0
- package/dist/monarch.d.ts +64 -0
- package/dist/monarch.js +205 -0
- package/dist/name-resolver.d.ts +135 -0
- package/dist/name-resolver.js +854 -0
- package/dist/parser.d.ts +331 -0
- package/dist/parser.js +2083 -0
- package/package.json +33 -0
package/dist/parser.d.ts
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* xDBML parser.
|
|
3
|
+
*
|
|
4
|
+
* Hand-written recursive-descent. Reads the token stream produced by the
|
|
5
|
+
* Lexer and emits the AST defined in ./ast.ts. Pragmatic and intentionally
|
|
6
|
+
* permissive at the parse level: several spec constraints (tuple position
|
|
7
|
+
* contiguity, named-type vs. builtin shadowing, ref-path array-crossing,
|
|
8
|
+
* polymorphic alternative selectors in paths) are deferred to a future
|
|
9
|
+
* semantic-analysis pass. The grammar test cases the parser passes are
|
|
10
|
+
* the official xDBML example files in /examples.
|
|
11
|
+
*/
|
|
12
|
+
import type { ParseOptions, Position, XDbmlDocument } from './ast.ts';
|
|
13
|
+
import type { Token } from './lexer.ts';
|
|
14
|
+
export declare class ParseError extends Error {
|
|
15
|
+
position: Position;
|
|
16
|
+
constructor(message: string, position: Position);
|
|
17
|
+
}
|
|
18
|
+
export declare class Parser {
|
|
19
|
+
private tokens;
|
|
20
|
+
private idx;
|
|
21
|
+
/**
|
|
22
|
+
* Parse-time options (v0.2 / P5+). Carries the importer's filePath, the
|
|
23
|
+
* optional readFile resolver, and the maxDepth bound. Used by
|
|
24
|
+
* parseModuleDirective to resolve reference-only directives. May be an
|
|
25
|
+
* empty object when no options were supplied (the public `parse(source)`
|
|
26
|
+
* 1-arg form).
|
|
27
|
+
*/
|
|
28
|
+
private options;
|
|
29
|
+
/**
|
|
30
|
+
* The set of file paths currently being parsed in the resolution chain.
|
|
31
|
+
* Used for cycle detection: when resolving a directive whose `from` path
|
|
32
|
+
* is already in this set, the parser produces an empty clone for that
|
|
33
|
+
* directive rather than recursing (matching spec §26.15: cycles are
|
|
34
|
+
* allowed; name resolution handles them). The set is passed by reference
|
|
35
|
+
* across recursive parse() calls so all transitive levels see it.
|
|
36
|
+
*
|
|
37
|
+
* The set contains the resolved ABSOLUTE paths (post-readFile-key path
|
|
38
|
+
* computation), not the source-text `from` strings, so two directives
|
|
39
|
+
* that name the same file via different relative paths still collide.
|
|
40
|
+
*/
|
|
41
|
+
private resolutionStack;
|
|
42
|
+
/**
|
|
43
|
+
* Current recursion depth. Incremented before each recursive parse(),
|
|
44
|
+
* compared against options.maxDepth. Reaching the limit throws.
|
|
45
|
+
*/
|
|
46
|
+
private depth;
|
|
47
|
+
constructor(tokens: Token[], options?: ParseOptions, resolutionStack?: ReadonlySet<string>, depth?: number);
|
|
48
|
+
private peek;
|
|
49
|
+
private advance;
|
|
50
|
+
private check;
|
|
51
|
+
private match;
|
|
52
|
+
private expect;
|
|
53
|
+
private spanFrom;
|
|
54
|
+
parseDocument(): XDbmlDocument;
|
|
55
|
+
private parseVersionDeclaration;
|
|
56
|
+
private parseExperimentalDeclaration;
|
|
57
|
+
private parseTopLevelStatement;
|
|
58
|
+
private parseProject;
|
|
59
|
+
/**
|
|
60
|
+
* A Note inside a Project/Container/Entity body. May appear as:
|
|
61
|
+
* Note: 'short text'
|
|
62
|
+
* Note: '''long text'''
|
|
63
|
+
* Note { '''long text''' }
|
|
64
|
+
*/
|
|
65
|
+
private parseNoteBlockOrSetting;
|
|
66
|
+
private parseSettingValueExpectingString;
|
|
67
|
+
/**
|
|
68
|
+
* Top-level `Note name { '''...''' }` standalone declaration.
|
|
69
|
+
*/
|
|
70
|
+
private parseNoteDeclaration;
|
|
71
|
+
/**
|
|
72
|
+
* Parse a `name: value` line inside a Project body. Used for project
|
|
73
|
+
* settings like `targets: PostgreSQL` or `database_type: 'MySQL'`.
|
|
74
|
+
*/
|
|
75
|
+
private parseLineSetting;
|
|
76
|
+
private parseContainer;
|
|
77
|
+
private parseEntity;
|
|
78
|
+
/**
|
|
79
|
+
* Entity names may be bare (`users`), dotted (`core.users` — implicit
|
|
80
|
+
* container), or quoted (`"my-table"`).
|
|
81
|
+
*/
|
|
82
|
+
private parseEntityName;
|
|
83
|
+
private parseEntityBody;
|
|
84
|
+
private parsePartialInjection;
|
|
85
|
+
/**
|
|
86
|
+
* Parse a `records { ... }` block inside an entity body (§25.1, implicit
|
|
87
|
+
* column list). Values are stored as SettingValue cells; row boundaries
|
|
88
|
+
* are determined by source line (see `parseRecordRow`).
|
|
89
|
+
*/
|
|
90
|
+
private parseRecordsBlock;
|
|
91
|
+
/**
|
|
92
|
+
* Top-level records declaration (§25.2, new in v0.2):
|
|
93
|
+
*
|
|
94
|
+
* records users (id, name, email) { ... }
|
|
95
|
+
* records core.users (id, name, email) { ... }
|
|
96
|
+
*
|
|
97
|
+
* The entity reference can be a bare name or a dotted path for cross-
|
|
98
|
+
* container references. The column list is required; it tells the
|
|
99
|
+
* generator which columns each row's values are populating.
|
|
100
|
+
*/
|
|
101
|
+
private parseTopLevelRecords;
|
|
102
|
+
/**
|
|
103
|
+
* Parse a single row of comma-separated values.
|
|
104
|
+
*
|
|
105
|
+
* Row delimiter rule: a comma continues the row only when the next value
|
|
106
|
+
* is on the same source line as the comma. If the comma is followed by
|
|
107
|
+
* a token on a later line (or the closing `}`), the comma is treated as
|
|
108
|
+
* a trailing comma and the row ends. This rule:
|
|
109
|
+
*
|
|
110
|
+
* - Tolerates trailing commas at end of row
|
|
111
|
+
* - Supports triple-quoted multi-line string VALUES (the comma after
|
|
112
|
+
* the closing `'''` is on the line of the closing triple, and the
|
|
113
|
+
* next value sits on that same line)
|
|
114
|
+
* - Does NOT support multi-line rows where a row's values are spread
|
|
115
|
+
* across multiple source lines connected by commas
|
|
116
|
+
*/
|
|
117
|
+
private parseRecordRow;
|
|
118
|
+
/**
|
|
119
|
+
* Parse a `use` or `reuse` directive. Called from both the top-level
|
|
120
|
+
* dispatcher and the Container body dispatcher; the caller indicates
|
|
121
|
+
* which context via the `context` argument. The context affects which
|
|
122
|
+
* placements are legal (e.g., field imports must be at file scope) but
|
|
123
|
+
* does NOT affect the directive's syntactic shape.
|
|
124
|
+
*
|
|
125
|
+
* Grammar:
|
|
126
|
+
*
|
|
127
|
+
* ('use' | 'reuse') importSpec 'from' StringLiteral metadataSettings? cloneBlock?
|
|
128
|
+
*
|
|
129
|
+
* importSpec ::= '*' | '{' importItem (',' importItem)* '}'
|
|
130
|
+
* importItem ::= elementType path ('as' Identifier)?
|
|
131
|
+
* elementType ::= 'table' | 'entity' | 'collection' | 'record' |
|
|
132
|
+
* 'enum' | 'tablepartial' | 'note' | 'schema' |
|
|
133
|
+
* 'container' | 'tablegroup' | 'type' | 'edge' |
|
|
134
|
+
* 'view' | 'diagramview' | 'field'
|
|
135
|
+
* metadataSettings ::= '[' setting (',' setting)* ']'
|
|
136
|
+
* cloneBlock ::= '{' topLevelStatement* '}'
|
|
137
|
+
*/
|
|
138
|
+
private parseModuleDirective;
|
|
139
|
+
/**
|
|
140
|
+
* Parse one import item: an element-type keyword, a dotted source path,
|
|
141
|
+
* and an optional `as <alias>`.
|
|
142
|
+
*
|
|
143
|
+
* entity core.dim_customer
|
|
144
|
+
* type Email
|
|
145
|
+
* type Email as PII_Email
|
|
146
|
+
* field core.dim_customer.email (rejected in P4)
|
|
147
|
+
*/
|
|
148
|
+
private parseImportItem;
|
|
149
|
+
/**
|
|
150
|
+
* Parse a clone block. The block contains zero or more declarations
|
|
151
|
+
* that match the import items by name and element type (matching is
|
|
152
|
+
* downstream-consumer's job; the parser is permissive).
|
|
153
|
+
*
|
|
154
|
+
* Per spec §26.6, clone content uses the importing file's vocabulary
|
|
155
|
+
* (aliases already applied) and is parsed under the importing file's
|
|
156
|
+
* xdbml version directive.
|
|
157
|
+
*
|
|
158
|
+
* Most clone-block content uses TopLevelStatement shapes (Entity, Type,
|
|
159
|
+
* Container, etc.). The exception is field imports (§26.8): when the
|
|
160
|
+
* directive imports one or more fields via `field <path>` items, the
|
|
161
|
+
* clone block holds each field as a bare FieldDeclaration with no entity
|
|
162
|
+
* wrapper. The dispatch below checks whether the next token starts a
|
|
163
|
+
* known top-level keyword and falls through to FieldDeclaration when
|
|
164
|
+
* it doesn't.
|
|
165
|
+
*/
|
|
166
|
+
private parseCloneBlock;
|
|
167
|
+
/**
|
|
168
|
+
* Lookahead helper: does the current token start a top-level statement?
|
|
169
|
+
*
|
|
170
|
+
* Used by parseCloneBlock to dispatch between "this is a top-level
|
|
171
|
+
* declaration" (Entity, Type, Container, etc.) and "this is a bare
|
|
172
|
+
* field declaration" (for field imports). A field declaration starts
|
|
173
|
+
* with an identifier followed by a type expression; a top-level
|
|
174
|
+
* statement starts with one of the known top-level keywords.
|
|
175
|
+
*
|
|
176
|
+
* Mirrors the dispatch in parseTopLevelStatement(). If we add new
|
|
177
|
+
* top-level constructs there, this set should grow in parallel.
|
|
178
|
+
*/
|
|
179
|
+
private isCloneTopLevelStart;
|
|
180
|
+
/**
|
|
181
|
+
* `field_name typeExpression [settings]` or `"quoted name" typeExpression [settings]`.
|
|
182
|
+
*
|
|
183
|
+
* Critical lookahead point: we're invoked from a context where the next
|
|
184
|
+
* token MUST be a field name (Identifier or QuotedIdentifier), and the
|
|
185
|
+
* token after it is a type expression. If the next thing is a Note block
|
|
186
|
+
* or a partial injection or `indexes`, those should have been handled by
|
|
187
|
+
* the caller already.
|
|
188
|
+
*/
|
|
189
|
+
private parseFieldDeclaration;
|
|
190
|
+
/**
|
|
191
|
+
* Parse a type expression. Dispatch on the leading keyword/identifier:
|
|
192
|
+
*
|
|
193
|
+
* - `object { ... }` (and synonyms struct/record)
|
|
194
|
+
* - `array [ ... ]` (and synonym list)
|
|
195
|
+
* - `map [k, v]` (and synonyms dict/dictionary)
|
|
196
|
+
* - `set [t]`
|
|
197
|
+
* - `union [ ... ]`
|
|
198
|
+
* - `oneOf { ... }` / `anyOf { ... }` / `allOf { ... }`
|
|
199
|
+
* - `json { ... }` (and synonyms jsonb/variant; block optional)
|
|
200
|
+
* - Otherwise: scalar / named-type reference. With optional `(p, s)`.
|
|
201
|
+
*/
|
|
202
|
+
private parseTypeExpression;
|
|
203
|
+
private parseObjectType;
|
|
204
|
+
/**
|
|
205
|
+
* `array [ ... ]`. The bracket body has several forms:
|
|
206
|
+
*
|
|
207
|
+
* 1. `[varchar]` -- bare element type
|
|
208
|
+
* 2. `[varchar [not null]]` -- element type with settings
|
|
209
|
+
* 3. `[line_item object { ... }]` -- named element type (common with object)
|
|
210
|
+
* 4. `[[0] x object {...}, [1] y object {...}]` -- tuple type
|
|
211
|
+
*
|
|
212
|
+
* Disambiguation: if the first token inside the bracket is `[`, it's a
|
|
213
|
+
* tuple (each tuple element starts with `[N]`). Otherwise we look at the
|
|
214
|
+
* shape: if the first thing is an identifier and the second is also an
|
|
215
|
+
* identifier or a structural-type keyword, it's `name type` form;
|
|
216
|
+
* otherwise the first thing is the bare type.
|
|
217
|
+
*/
|
|
218
|
+
private parseArrayType;
|
|
219
|
+
/** True if the token looks like the start of a TypeExpression. */
|
|
220
|
+
private tokenStartsType;
|
|
221
|
+
private parseTupleElements;
|
|
222
|
+
private parseMapType;
|
|
223
|
+
private parseSetType;
|
|
224
|
+
/**
|
|
225
|
+
* `union [t1, t2, null]`. Members are scalars or null.
|
|
226
|
+
*/
|
|
227
|
+
private parseUnionType;
|
|
228
|
+
private parseUnionMember;
|
|
229
|
+
private parsePolymorphicType;
|
|
230
|
+
/** `alternative_name typeExpression [settings]` -- shape is the same as a field declaration, context disambiguates */
|
|
231
|
+
private parsePolymorphicAlternative;
|
|
232
|
+
private parseJsonType;
|
|
233
|
+
/**
|
|
234
|
+
* Scalar type or named-type reference. Both look like an Identifier with
|
|
235
|
+
* optional `(p, s)` parameter list. The distinction is made later at the
|
|
236
|
+
* semantic-analysis stage (named types are user-declared identifiers that
|
|
237
|
+
* resolve to a TypeDeclaration; scalars are the open set of built-ins).
|
|
238
|
+
*
|
|
239
|
+
* Resolution heuristic for the PoC: if the identifier's lowercase form is
|
|
240
|
+
* a known SQL/BSON scalar name, we tag ScalarType; otherwise we'd ideally
|
|
241
|
+
* defer to the semantic pass. For the PoC we always emit ScalarType for
|
|
242
|
+
* common scalar names and ScalarType for everything else too; callers
|
|
243
|
+
* that need to distinguish can post-process.
|
|
244
|
+
*
|
|
245
|
+
* Actually a cleaner choice: emit ScalarType when there are parameters
|
|
246
|
+
* (no named type takes `(p,s)`), and otherwise emit NamedTypeReference
|
|
247
|
+
* iff the name's first character is uppercase (heuristic) -- but that
|
|
248
|
+
* conflicts with Decimal128 etc. So: always emit ScalarType; the
|
|
249
|
+
* semantic-analysis pass walks Type declarations and rewrites scalars
|
|
250
|
+
* whose names resolve to user types as NamedTypeReference. The PoC keeps
|
|
251
|
+
* the AST shape consistent regardless.
|
|
252
|
+
*/
|
|
253
|
+
private parseScalarOrNamedType;
|
|
254
|
+
private parseTypeParam;
|
|
255
|
+
private parseTypeDecl;
|
|
256
|
+
/**
|
|
257
|
+
* Finish parsing a v0.1 object-form Type after the name (and optional
|
|
258
|
+
* pre-body settings) have been consumed. Handles the `{ ...body }` part.
|
|
259
|
+
*/
|
|
260
|
+
private finishObjectTypeDecl;
|
|
261
|
+
private parseEdge;
|
|
262
|
+
private parseView;
|
|
263
|
+
private parseSourceQueryItem;
|
|
264
|
+
private parseEnum;
|
|
265
|
+
private parseRef;
|
|
266
|
+
private parseRefSpec;
|
|
267
|
+
private parseCardinalityOperator;
|
|
268
|
+
private parseRefEndpoint;
|
|
269
|
+
/**
|
|
270
|
+
* Parse a dotted path with the §18 segment vocabulary:
|
|
271
|
+
*
|
|
272
|
+
* IDENTIFIER -- a field segment
|
|
273
|
+
* .IDENTIFIER -- field
|
|
274
|
+
* .[N] -- array index (positional)
|
|
275
|
+
* .[*] -- array wildcard (via ArrayWildcard token)
|
|
276
|
+
* ."quoted name" -- quoted-identifier field
|
|
277
|
+
* .["literal key"] -- map literal key
|
|
278
|
+
*
|
|
279
|
+
* We start by consuming an identifier/qualified head, then walk pathTail.
|
|
280
|
+
* The JSONPath-alias forms `[N]`, `[*]` without a leading dot are
|
|
281
|
+
* recognized as well; they normalize to the dot-prefixed form.
|
|
282
|
+
*
|
|
283
|
+
* For the PoC we stop at the first token that doesn't continue a path
|
|
284
|
+
* (e.g., a cardinality operator, a comma, a settings bracket).
|
|
285
|
+
*/
|
|
286
|
+
private parsePathSegments;
|
|
287
|
+
private parseTablePartial;
|
|
288
|
+
private parseTableGroup;
|
|
289
|
+
private parseIndexes;
|
|
290
|
+
private parseIndexEntry;
|
|
291
|
+
private parseIndexComponent;
|
|
292
|
+
private parseChecks;
|
|
293
|
+
private parseCheckEntry;
|
|
294
|
+
private maybeSettingsBlock;
|
|
295
|
+
/**
|
|
296
|
+
* A single setting. Forms:
|
|
297
|
+
* flag -- bare identifier(s), e.g. `pk`, `not null`
|
|
298
|
+
* name: value -- key/value, e.g. `default: 'x'`, `synonyms: [...]`
|
|
299
|
+
* ref: > target -- inline ref
|
|
300
|
+
*
|
|
301
|
+
* The grammar for "flag" is annoying because `not null` is two words but is
|
|
302
|
+
* still one flag. We handle the multi-word flags by greedy lowercase prefix
|
|
303
|
+
* match: `not` followed by `null` becomes `not null`; `primary` followed by
|
|
304
|
+
* `key` becomes `primary key`.
|
|
305
|
+
*/
|
|
306
|
+
private parseSetting;
|
|
307
|
+
/**
|
|
308
|
+
* A setting value. Open-vocabulary:
|
|
309
|
+
* string literal, multi-line string, number, boolean, null, identifier
|
|
310
|
+
* (or dotted identifier path), expression literal, list `[...]`
|
|
311
|
+
*/
|
|
312
|
+
private parseSettingValue;
|
|
313
|
+
private parseIdentLikeName;
|
|
314
|
+
}
|
|
315
|
+
/**
|
|
316
|
+
* Parse xDBML source.
|
|
317
|
+
*
|
|
318
|
+
* - 1-argument form `parse(source)` parses self-contained documents (any
|
|
319
|
+
* module directive must carry an inline clone block; reference-only
|
|
320
|
+
* directives throw).
|
|
321
|
+
* - 2-argument form `parse(source, options)` accepts a `readFile`
|
|
322
|
+
* resolver for cross-file `use`/`reuse` directives and a `filePath`
|
|
323
|
+
* identifying the source for relative-path resolution. See
|
|
324
|
+
* `ParseOptions` for the full shape.
|
|
325
|
+
*
|
|
326
|
+
* The function is fully synchronous. Async file loading and incremental
|
|
327
|
+
* resolution are intentionally out of scope -- callers needing async I/O
|
|
328
|
+
* should pre-load their module graph and supply a `readFile` callback
|
|
329
|
+
* that returns from an in-memory map.
|
|
330
|
+
*/
|
|
331
|
+
export declare function parse(source: string, options?: ParseOptions): XDbmlDocument;
|