@xdbml/parse 0.1.0-poc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,331 @@
1
+ /**
2
+ * xDBML parser.
3
+ *
4
+ * Hand-written recursive-descent. Reads the token stream produced by the
5
+ * Lexer and emits the AST defined in ./ast.ts. Pragmatic and intentionally
6
+ * permissive at the parse level: several spec constraints (tuple position
7
+ * contiguity, named-type vs. builtin shadowing, ref-path array-crossing,
8
+ * polymorphic alternative selectors in paths) are deferred to a future
9
+ * semantic-analysis pass. The grammar test cases the parser passes are
10
+ * the official xDBML example files in /examples.
11
+ */
12
+ import type { ParseOptions, Position, XDbmlDocument } from './ast.ts';
13
+ import type { Token } from './lexer.ts';
14
+ export declare class ParseError extends Error {
15
+ position: Position;
16
+ constructor(message: string, position: Position);
17
+ }
18
+ export declare class Parser {
19
+ private tokens;
20
+ private idx;
21
+ /**
22
+ * Parse-time options (v0.2 / P5+). Carries the importer's filePath, the
23
+ * optional readFile resolver, and the maxDepth bound. Used by
24
+ * parseModuleDirective to resolve reference-only directives. May be an
25
+ * empty object when no options were supplied (the public `parse(source)`
26
+ * 1-arg form).
27
+ */
28
+ private options;
29
+ /**
30
+ * The set of file paths currently being parsed in the resolution chain.
31
+ * Used for cycle detection: when resolving a directive whose `from` path
32
+ * is already in this set, the parser produces an empty clone for that
33
+ * directive rather than recursing (matching spec §26.15: cycles are
34
+ * allowed; name resolution handles them). The set is passed by reference
35
+ * across recursive parse() calls so all transitive levels see it.
36
+ *
37
+ * The set contains the resolved ABSOLUTE paths (post-readFile-key path
38
+ * computation), not the source-text `from` strings, so two directives
39
+ * that name the same file via different relative paths still collide.
40
+ */
41
+ private resolutionStack;
42
+ /**
43
+ * Current recursion depth. Incremented before each recursive parse(),
44
+ * compared against options.maxDepth. Reaching the limit throws.
45
+ */
46
+ private depth;
47
+ constructor(tokens: Token[], options?: ParseOptions, resolutionStack?: ReadonlySet<string>, depth?: number);
48
+ private peek;
49
+ private advance;
50
+ private check;
51
+ private match;
52
+ private expect;
53
+ private spanFrom;
54
+ parseDocument(): XDbmlDocument;
55
+ private parseVersionDeclaration;
56
+ private parseExperimentalDeclaration;
57
+ private parseTopLevelStatement;
58
+ private parseProject;
59
+ /**
60
+ * A Note inside a Project/Container/Entity body. May appear as:
61
+ * Note: 'short text'
62
+ * Note: '''long text'''
63
+ * Note { '''long text''' }
64
+ */
65
+ private parseNoteBlockOrSetting;
66
+ private parseSettingValueExpectingString;
67
+ /**
68
+ * Top-level `Note name { '''...''' }` standalone declaration.
69
+ */
70
+ private parseNoteDeclaration;
71
+ /**
72
+ * Parse a `name: value` line inside a Project body. Used for project
73
+ * settings like `targets: PostgreSQL` or `database_type: 'MySQL'`.
74
+ */
75
+ private parseLineSetting;
76
+ private parseContainer;
77
+ private parseEntity;
78
+ /**
79
+ * Entity names may be bare (`users`), dotted (`core.users` — implicit
80
+ * container), or quoted (`"my-table"`).
81
+ */
82
+ private parseEntityName;
83
+ private parseEntityBody;
84
+ private parsePartialInjection;
85
+ /**
86
+ * Parse a `records { ... }` block inside an entity body (§25.1, implicit
87
+ * column list). Values are stored as SettingValue cells; row boundaries
88
+ * are determined by source line (see `parseRecordRow`).
89
+ */
90
+ private parseRecordsBlock;
91
+ /**
92
+ * Top-level records declaration (§25.2, new in v0.2):
93
+ *
94
+ * records users (id, name, email) { ... }
95
+ * records core.users (id, name, email) { ... }
96
+ *
97
+ * The entity reference can be a bare name or a dotted path for cross-
98
+ * container references. The column list is required; it tells the
99
+ * generator which columns each row's values are populating.
100
+ */
101
+ private parseTopLevelRecords;
102
+ /**
103
+ * Parse a single row of comma-separated values.
104
+ *
105
+ * Row delimiter rule: a comma continues the row only when the next value
106
+ * is on the same source line as the comma. If the comma is followed by
107
+ * a token on a later line (or the closing `}`), the comma is treated as
108
+ * a trailing comma and the row ends. This rule:
109
+ *
110
+ * - Tolerates trailing commas at end of row
111
+ * - Supports triple-quoted multi-line string VALUES (the comma after
112
+ * the closing `'''` is on the line of the closing triple, and the
113
+ * next value sits on that same line)
114
+ * - Does NOT support multi-line rows where a row's values are spread
115
+ * across multiple source lines connected by commas
116
+ */
117
+ private parseRecordRow;
118
+ /**
119
+ * Parse a `use` or `reuse` directive. Called from both the top-level
120
+ * dispatcher and the Container body dispatcher; the caller indicates
121
+ * which context via the `context` argument. The context affects which
122
+ * placements are legal (e.g., field imports must be at file scope) but
123
+ * does NOT affect the directive's syntactic shape.
124
+ *
125
+ * Grammar:
126
+ *
127
+ * ('use' | 'reuse') importSpec 'from' StringLiteral metadataSettings? cloneBlock?
128
+ *
129
+ * importSpec ::= '*' | '{' importItem (',' importItem)* '}'
130
+ * importItem ::= elementType path ('as' Identifier)?
131
+ * elementType ::= 'table' | 'entity' | 'collection' | 'record' |
132
+ * 'enum' | 'tablepartial' | 'note' | 'schema' |
133
+ * 'container' | 'tablegroup' | 'type' | 'edge' |
134
+ * 'view' | 'diagramview' | 'field'
135
+ * metadataSettings ::= '[' setting (',' setting)* ']'
136
+ * cloneBlock ::= '{' topLevelStatement* '}'
137
+ */
138
+ private parseModuleDirective;
139
+ /**
140
+ * Parse one import item: an element-type keyword, a dotted source path,
141
+ * and an optional `as <alias>`.
142
+ *
143
+ * entity core.dim_customer
144
+ * type Email
145
+ * type Email as PII_Email
146
+ * field core.dim_customer.email (rejected in P4)
147
+ */
148
+ private parseImportItem;
149
+ /**
150
+ * Parse a clone block. The block contains zero or more declarations
151
+ * that match the import items by name and element type (matching is
152
+ * downstream-consumer's job; the parser is permissive).
153
+ *
154
+ * Per spec §26.6, clone content uses the importing file's vocabulary
155
+ * (aliases already applied) and is parsed under the importing file's
156
+ * xdbml version directive.
157
+ *
158
+ * Most clone-block content uses TopLevelStatement shapes (Entity, Type,
159
+ * Container, etc.). The exception is field imports (§26.8): when the
160
+ * directive imports one or more fields via `field <path>` items, the
161
+ * clone block holds each field as a bare FieldDeclaration with no entity
162
+ * wrapper. The dispatch below checks whether the next token starts a
163
+ * known top-level keyword and falls through to FieldDeclaration when
164
+ * it doesn't.
165
+ */
166
+ private parseCloneBlock;
167
+ /**
168
+ * Lookahead helper: does the current token start a top-level statement?
169
+ *
170
+ * Used by parseCloneBlock to dispatch between "this is a top-level
171
+ * declaration" (Entity, Type, Container, etc.) and "this is a bare
172
+ * field declaration" (for field imports). A field declaration starts
173
+ * with an identifier followed by a type expression; a top-level
174
+ * statement starts with one of the known top-level keywords.
175
+ *
176
+ * Mirrors the dispatch in parseTopLevelStatement(). If we add new
177
+ * top-level constructs there, this set should grow in parallel.
178
+ */
179
+ private isCloneTopLevelStart;
180
+ /**
181
+ * `field_name typeExpression [settings]` or `"quoted name" typeExpression [settings]`.
182
+ *
183
+ * Critical lookahead point: we're invoked from a context where the next
184
+ * token MUST be a field name (Identifier or QuotedIdentifier), and the
185
+ * token after it is a type expression. If the next thing is a Note block
186
+ * or a partial injection or `indexes`, those should have been handled by
187
+ * the caller already.
188
+ */
189
+ private parseFieldDeclaration;
190
+ /**
191
+ * Parse a type expression. Dispatch on the leading keyword/identifier:
192
+ *
193
+ * - `object { ... }` (and synonyms struct/record)
194
+ * - `array [ ... ]` (and synonym list)
195
+ * - `map [k, v]` (and synonyms dict/dictionary)
196
+ * - `set [t]`
197
+ * - `union [ ... ]`
198
+ * - `oneOf { ... }` / `anyOf { ... }` / `allOf { ... }`
199
+ * - `json { ... }` (and synonyms jsonb/variant; block optional)
200
+ * - Otherwise: scalar / named-type reference. With optional `(p, s)`.
201
+ */
202
+ private parseTypeExpression;
203
+ private parseObjectType;
204
+ /**
205
+ * `array [ ... ]`. The bracket body has several forms:
206
+ *
207
+ * 1. `[varchar]` -- bare element type
208
+ * 2. `[varchar [not null]]` -- element type with settings
209
+ * 3. `[line_item object { ... }]` -- named element type (common with object)
210
+ * 4. `[[0] x object {...}, [1] y object {...}]` -- tuple type
211
+ *
212
+ * Disambiguation: if the first token inside the bracket is `[`, it's a
213
+ * tuple (each tuple element starts with `[N]`). Otherwise we look at the
214
+ * shape: if the first thing is an identifier and the second is also an
215
+ * identifier or a structural-type keyword, it's `name type` form;
216
+ * otherwise the first thing is the bare type.
217
+ */
218
+ private parseArrayType;
219
+ /** True if the token looks like the start of a TypeExpression. */
220
+ private tokenStartsType;
221
+ private parseTupleElements;
222
+ private parseMapType;
223
+ private parseSetType;
224
+ /**
225
+ * `union [t1, t2, null]`. Members are scalars or null.
226
+ */
227
+ private parseUnionType;
228
+ private parseUnionMember;
229
+ private parsePolymorphicType;
230
+ /** `alternative_name typeExpression [settings]` -- shape is the same as a field declaration, context disambiguates */
231
+ private parsePolymorphicAlternative;
232
+ private parseJsonType;
233
+ /**
234
+ * Scalar type or named-type reference. Both look like an Identifier with
235
+ * optional `(p, s)` parameter list. The distinction is made later at the
236
+ * semantic-analysis stage (named types are user-declared identifiers that
237
+ * resolve to a TypeDeclaration; scalars are the open set of built-ins).
238
+ *
239
+ * Resolution heuristic for the PoC: if the identifier's lowercase form is
240
+ * a known SQL/BSON scalar name, we tag ScalarType; otherwise we'd ideally
241
+ * defer to the semantic pass. For the PoC we always emit ScalarType for
242
+ * common scalar names and ScalarType for everything else too; callers
243
+ * that need to distinguish can post-process.
244
+ *
245
+ * Actually a cleaner choice: emit ScalarType when there are parameters
246
+ * (no named type takes `(p,s)`), and otherwise emit NamedTypeReference
247
+ * iff the name's first character is uppercase (heuristic) -- but that
248
+ * conflicts with Decimal128 etc. So: always emit ScalarType; the
249
+ * semantic-analysis pass walks Type declarations and rewrites scalars
250
+ * whose names resolve to user types as NamedTypeReference. The PoC keeps
251
+ * the AST shape consistent regardless.
252
+ */
253
+ private parseScalarOrNamedType;
254
+ private parseTypeParam;
255
+ private parseTypeDecl;
256
+ /**
257
+ * Finish parsing a v0.1 object-form Type after the name (and optional
258
+ * pre-body settings) have been consumed. Handles the `{ ...body }` part.
259
+ */
260
+ private finishObjectTypeDecl;
261
+ private parseEdge;
262
+ private parseView;
263
+ private parseSourceQueryItem;
264
+ private parseEnum;
265
+ private parseRef;
266
+ private parseRefSpec;
267
+ private parseCardinalityOperator;
268
+ private parseRefEndpoint;
269
+ /**
270
+ * Parse a dotted path with the §18 segment vocabulary:
271
+ *
272
+ * IDENTIFIER -- a field segment
273
+ * .IDENTIFIER -- field
274
+ * .[N] -- array index (positional)
275
+ * .[*] -- array wildcard (via ArrayWildcard token)
276
+ * ."quoted name" -- quoted-identifier field
277
+ * .["literal key"] -- map literal key
278
+ *
279
+ * We start by consuming an identifier/qualified head, then walk pathTail.
280
+ * The JSONPath-alias forms `[N]`, `[*]` without a leading dot are
281
+ * recognized as well; they normalize to the dot-prefixed form.
282
+ *
283
+ * For the PoC we stop at the first token that doesn't continue a path
284
+ * (e.g., a cardinality operator, a comma, a settings bracket).
285
+ */
286
+ private parsePathSegments;
287
+ private parseTablePartial;
288
+ private parseTableGroup;
289
+ private parseIndexes;
290
+ private parseIndexEntry;
291
+ private parseIndexComponent;
292
+ private parseChecks;
293
+ private parseCheckEntry;
294
+ private maybeSettingsBlock;
295
+ /**
296
+ * A single setting. Forms:
297
+ * flag -- bare identifier(s), e.g. `pk`, `not null`
298
+ * name: value -- key/value, e.g. `default: 'x'`, `synonyms: [...]`
299
+ * ref: > target -- inline ref
300
+ *
301
+ * The grammar for "flag" is annoying because `not null` is two words but is
302
+ * still one flag. We handle the multi-word flags by greedy lowercase prefix
303
+ * match: `not` followed by `null` becomes `not null`; `primary` followed by
304
+ * `key` becomes `primary key`.
305
+ */
306
+ private parseSetting;
307
+ /**
308
+ * A setting value. Open-vocabulary:
309
+ * string literal, multi-line string, number, boolean, null, identifier
310
+ * (or dotted identifier path), expression literal, list `[...]`
311
+ */
312
+ private parseSettingValue;
313
+ private parseIdentLikeName;
314
+ }
315
+ /**
316
+ * Parse xDBML source.
317
+ *
318
+ * - 1-argument form `parse(source)` parses self-contained documents (any
319
+ * module directive must carry an inline clone block; reference-only
320
+ * directives throw).
321
+ * - 2-argument form `parse(source, options)` accepts a `readFile`
322
+ * resolver for cross-file `use`/`reuse` directives and a `filePath`
323
+ * identifying the source for relative-path resolution. See
324
+ * `ParseOptions` for the full shape.
325
+ *
326
+ * The function is fully synchronous. Async file loading and incremental
327
+ * resolution are intentionally out of scope -- callers needing async I/O
328
+ * should pre-load their module graph and supply a `readFile` callback
329
+ * that returns from an in-memory map.
330
+ */
331
+ export declare function parse(source: string, options?: ParseOptions): XDbmlDocument;