@xdbml/parse 0.1.0-poc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,64 @@
1
+ /**
2
+ * Monaco language configuration and Monarch tokens provider for xDBML.
3
+ *
4
+ * Monarch is a regex-based state-machine syntax highlighter that ships
5
+ * with Monaco. This file produces the highlighting rules; the real parser
6
+ * lives in ./parser.ts. The two operate independently: Monaco asks
7
+ * Monarch "what color is each character" for highlighting, and asks the
8
+ * parser "is the file valid and what's its AST" for everything else.
9
+ *
10
+ * The two must stay in sync grammatically (a keyword the parser
11
+ * recognizes should also be highlighted as a keyword) but they don't
12
+ * share code or state.
13
+ *
14
+ * Reference: https://microsoft.github.io/monaco-editor/monarch.html
15
+ *
16
+ * Types are loose `unknown` rather than depending on monaco-editor here,
17
+ * because @xdbml/parse should be usable without forcing the Monaco
18
+ * dependency on consumers that just want to parse. The Monaco types
19
+ * are structurally compatible -- the consumer casts at the boundary.
20
+ */
21
+ /** Subset of Monaco's LanguageConfiguration -- only the fields we set. */
22
+ export interface XDbmlLanguageConfiguration {
23
+ comments: {
24
+ lineComment: string;
25
+ blockComment: [string, string];
26
+ };
27
+ brackets: [string, string][];
28
+ autoClosingPairs: {
29
+ open: string;
30
+ close: string;
31
+ }[];
32
+ surroundingPairs: {
33
+ open: string;
34
+ close: string;
35
+ }[];
36
+ indentationRules?: {
37
+ increaseIndentPattern: RegExp;
38
+ decreaseIndentPattern: RegExp;
39
+ };
40
+ }
41
+ /** Subset of Monaco's IMonarchLanguage -- the shape Monarch consumes. */
42
+ export interface XDbmlMonarchLanguage {
43
+ tokenPostfix: string;
44
+ brackets: {
45
+ open: string;
46
+ close: string;
47
+ token: string;
48
+ }[];
49
+ decls: string[];
50
+ containerKeywords: string[];
51
+ entityKeywords: string[];
52
+ structuralTypeKeywords: string[];
53
+ polymorphismKeywords: string[];
54
+ scalarTypes: string[];
55
+ bsonTypes: string[];
56
+ settingFlags: string[];
57
+ settingKeys: string[];
58
+ granularityValues: string[];
59
+ ignoreCase: boolean;
60
+ unicode: boolean;
61
+ tokenizer: Record<string, unknown[]>;
62
+ }
63
+ export declare const xdbmlLanguageConfig: XDbmlLanguageConfiguration;
64
+ export declare const xdbmlMonarchTokensProvider: XDbmlMonarchLanguage;
@@ -0,0 +1,205 @@
1
+ /**
2
+ * Monaco language configuration and Monarch tokens provider for xDBML.
3
+ *
4
+ * Monarch is a regex-based state-machine syntax highlighter that ships
5
+ * with Monaco. This file produces the highlighting rules; the real parser
6
+ * lives in ./parser.ts. The two operate independently: Monaco asks
7
+ * Monarch "what color is each character" for highlighting, and asks the
8
+ * parser "is the file valid and what's its AST" for everything else.
9
+ *
10
+ * The two must stay in sync grammatically (a keyword the parser
11
+ * recognizes should also be highlighted as a keyword) but they don't
12
+ * share code or state.
13
+ *
14
+ * Reference: https://microsoft.github.io/monaco-editor/monarch.html
15
+ *
16
+ * Types are loose `unknown` rather than depending on monaco-editor here,
17
+ * because @xdbml/parse should be usable without forcing the Monaco
18
+ * dependency on consumers that just want to parse. The Monaco types
19
+ * are structurally compatible -- the consumer casts at the boundary.
20
+ */
21
+ import { CONTAINER_KEYWORDS, ENTITY_KEYWORDS, DECLARATION_KEYWORDS, STRUCTURAL_TYPE_KEYWORDS, POLYMORPHISM_KEYWORDS, SCALAR_TYPES, BSON_TYPES, SETTING_FLAGS, SETTING_KEYS, GRANULARITY_VALUES, } from "./keywords.js";
22
+ /* -------------------------------------------------------------------------
23
+ * Language configuration
24
+ * ----------------------------------------------------------------------- */
25
+ export const xdbmlLanguageConfig = {
26
+ comments: {
27
+ lineComment: '//',
28
+ blockComment: ['/*', '*/'],
29
+ },
30
+ brackets: [
31
+ ['{', '}'],
32
+ ['[', ']'],
33
+ ['(', ')'],
34
+ ],
35
+ autoClosingPairs: [
36
+ { open: '{', close: '}' },
37
+ { open: '[', close: ']' },
38
+ { open: '(', close: ')' },
39
+ { open: '"', close: '"' },
40
+ { open: "'", close: "'" },
41
+ { open: '`', close: '`' },
42
+ ],
43
+ surroundingPairs: [
44
+ { open: '{', close: '}' },
45
+ { open: '[', close: ']' },
46
+ { open: '(', close: ')' },
47
+ { open: '"', close: '"' },
48
+ { open: "'", close: "'" },
49
+ { open: '`', close: '`' },
50
+ ],
51
+ indentationRules: {
52
+ increaseIndentPattern: /^(.*\{[^}]*|\s*[{[].*)$/,
53
+ decreaseIndentPattern: /^(.*\}.*|\s*[}\]].*)$/,
54
+ },
55
+ };
56
+ /* -------------------------------------------------------------------------
57
+ * Monarch tokens provider
58
+ *
59
+ * Token type conventions (these map to the Monaco theme rules):
60
+ * keyword -- core construct keywords (Project, Container, ...)
61
+ * keyword.declaration -- top-level declaration keywords
62
+ * keyword.entity -- Entity/Table/Collection/Record
63
+ * keyword.container -- Container/Schema/Database/...
64
+ * keyword.type -- structural type keywords (object/array/...)
65
+ * keyword.polymorphism -- oneOf/anyOf/allOf/union
66
+ * keyword.directive -- xdbml:, experimental:
67
+ * type -- scalar type names (int, varchar, decimal, ...)
68
+ * type.bson -- BSON type names (objectId, Decimal128, ...)
69
+ * identifier -- user-supplied names
70
+ * identifier.custom-property -- x_-prefixed custom property names
71
+ * string / string.multiline -- string literals
72
+ * string.backtick -- expression literals
73
+ * string.quoted-ident -- double-quoted identifiers
74
+ * number -- numeric literals
75
+ * comment / comment.block -- comments
76
+ * operators -- cardinality operators (< > - <>)
77
+ * keyword.wildcard -- [*]
78
+ * keyword.partial -- ~ in partial injection
79
+ * delimiter / @bracket -- punctuation
80
+ * ----------------------------------------------------------------------- */
81
+ export const xdbmlMonarchTokensProvider = {
82
+ tokenPostfix: '.xdbml',
83
+ brackets: [
84
+ { open: '[', close: ']', token: 'delimiter.square' },
85
+ { open: '(', close: ')', token: 'delimiter.parenthesis' },
86
+ { open: '{', close: '}', token: 'delimiter.curly' },
87
+ ],
88
+ // Keyword vocabulary is sourced from ./keywords.ts so the TextMate
89
+ // grammar (in tools/textmate/) and this Monarch tokenizer stay in
90
+ // sync. To add a keyword: edit ./keywords.ts (one place), then
91
+ // re-run the TextMate build script. See keywords.ts for the full
92
+ // workflow.
93
+ // Top-level declaration keywords
94
+ decls: [...DECLARATION_KEYWORDS],
95
+ containerKeywords: [...CONTAINER_KEYWORDS],
96
+ entityKeywords: [...ENTITY_KEYWORDS],
97
+ // Structural type expression keywords (used as type expressions inside fields)
98
+ structuralTypeKeywords: [...STRUCTURAL_TYPE_KEYWORDS],
99
+ polymorphismKeywords: [...POLYMORPHISM_KEYWORDS],
100
+ // SQL scalar types -- recognized for highlighting; the parser accepts any
101
+ // identifier as a scalar type, so this list is for color, not validation.
102
+ scalarTypes: [...SCALAR_TYPES],
103
+ // BSON / document-store types
104
+ bsonTypes: [...BSON_TYPES],
105
+ // Bare-flag settings: `pk`, `unique`, `not null`, etc.
106
+ // `not null` and `primary key` are two words but tokenized one at a time
107
+ // here -- the highlighter colors each as `keyword.setting`.
108
+ // `required` is a synonym for `not null` (spec §8); the parser
109
+ // normalizes it to `not null` in the AST, but for highlighting
110
+ // purposes both spellings get the same `keyword.setting` color.
111
+ settingFlags: [...SETTING_FLAGS],
112
+ // Setting keys appearing as `name: value` -- recognized for highlighting.
113
+ // Open-vocabulary at the parser level; this list drives coloring only.
114
+ settingKeys: [...SETTING_KEYS],
115
+ granularityValues: [...GRANULARITY_VALUES],
116
+ ignoreCase: true,
117
+ unicode: true,
118
+ tokenizer: {
119
+ root: [
120
+ // xDBML / experimental directives at the very top of files
121
+ [/^(\s*)(xdbml|experimental)(\s*)(:)/, [
122
+ '',
123
+ 'keyword.directive',
124
+ '',
125
+ 'delimiter',
126
+ ]],
127
+ // [*] -- array wildcard token (must precede generic bracket rules)
128
+ [/\[\*\]/, 'keyword.wildcard'],
129
+ // ~ prefix for partial injection
130
+ [/~/, 'keyword.partial'],
131
+ // Brackets, parens, braces
132
+ [/[{}[\]()]/, '@bracket'],
133
+ // Punctuation
134
+ [/[,.:;]/, 'delimiter'],
135
+ // Cardinality operators
136
+ [/<>/, 'operators.cardinality'],
137
+ [/[<>-](?![A-Za-z_])/, 'operators.cardinality'],
138
+ // Comments
139
+ [/\/\/.*$/, 'comment'],
140
+ [/\/\*/, 'comment.block', '@comment'],
141
+ // Triple-quoted multi-line strings -- must precede single-quoted
142
+ [/'''/, { token: 'string.multiline', next: '@multilineString' }],
143
+ // Quoted identifiers (double-quote)
144
+ [/"/, { token: 'string.quoted-ident', next: '@quotedIdent' }],
145
+ // Single-quoted strings
146
+ [/'/, { token: 'string', next: '@singleString' }],
147
+ // Backtick expression literals
148
+ [/`/, { token: 'string.backtick', next: '@backtickExpr' }],
149
+ // Numbers
150
+ [/0[xX][0-9a-fA-F]+/, 'number.hex'],
151
+ [/-?\d+\.\d+([eE][+-]?\d+)?/, 'number.float'],
152
+ [/-?\d+([eE][+-]?\d+)?/, 'number'],
153
+ [/#[0-9A-Fa-f]{3,8}\b/, 'number.hex'],
154
+ // x_ custom property identifiers
155
+ [/\bx_[a-zA-Z0-9_]+/, 'identifier.custom-property'],
156
+ // Identifiers and keyword recognition.
157
+ // The parser is the authority on keyword vs identifier disambiguation;
158
+ // Monarch does coarse highlighting based on lowercase comparison.
159
+ [/[a-zA-Z_][\w$]*/, {
160
+ cases: {
161
+ '@containerKeywords': 'keyword.container',
162
+ '@entityKeywords': 'keyword.entity',
163
+ '@structuralTypeKeywords': 'keyword.type',
164
+ '@polymorphismKeywords': 'keyword.polymorphism',
165
+ '@decls': 'keyword.declaration',
166
+ '@scalarTypes': 'type',
167
+ '@bsonTypes': 'type.bson',
168
+ '@settingFlags': 'keyword.setting',
169
+ '@settingKeys': 'keyword.setting',
170
+ '@granularityValues': 'keyword.value',
171
+ 'true': 'keyword.literal',
172
+ 'false': 'keyword.literal',
173
+ 'null': 'keyword.literal',
174
+ '@default': 'identifier',
175
+ },
176
+ }],
177
+ // Whitespace
178
+ [/[ \t\r\n]+/, ''],
179
+ ],
180
+ comment: [
181
+ [/[^/*]+/, 'comment.block'],
182
+ [/\*\//, 'comment.block', '@pop'],
183
+ [/[/*]/, 'comment.block'],
184
+ ],
185
+ singleString: [
186
+ [/[^\\']+/, 'string'],
187
+ [/\\./, 'string.escape'],
188
+ [/'/, { token: 'string', next: '@pop' }],
189
+ ],
190
+ multilineString: [
191
+ [/[^']+/, 'string.multiline'],
192
+ [/'''/, { token: 'string.multiline', next: '@pop' }],
193
+ [/'/, 'string.multiline'],
194
+ ],
195
+ quotedIdent: [
196
+ [/[^\\"]+/, 'string.quoted-ident'],
197
+ [/\\./, 'string.quoted-ident'],
198
+ [/"/, { token: 'string.quoted-ident', next: '@pop' }],
199
+ ],
200
+ backtickExpr: [
201
+ [/[^`]+/, 'string.backtick'],
202
+ [/`/, { token: 'string.backtick', next: '@pop' }],
203
+ ],
204
+ },
205
+ };
@@ -0,0 +1,135 @@
1
+ /**
2
+ * Name resolution pass (spec §26.10 / §26.15, parser batch P6).
3
+ *
4
+ * `resolveNames(doc)` walks an xDBML document (the flattened view; clone
5
+ * blocks have been merged) and produces:
6
+ *
7
+ * - A symbol table mapping qualified names to declarations
8
+ * - A list of diagnostics: unresolved references and name conflicts
9
+ *
10
+ * The resolver does NOT mutate the AST. It is a pure side computation
11
+ * that downstream consumers can run for validation, IDE support, code
12
+ * generation, etc. Spans on diagnostics point to the offending construct
13
+ * in the source, so callers can surface them as editor markers.
14
+ *
15
+ * Per the spec, name resolution is a two-pass process:
16
+ *
17
+ * Pass 1: Collect declarations.
18
+ * Walk all top-level + container-body declarations and add them to
19
+ * the symbol table. Duplicates (same qualified name + same kind)
20
+ * produce a `duplicate-declaration` diagnostic and the LATER
21
+ * declaration is silently dropped from the table.
22
+ *
23
+ * Pass 2: Resolve references.
24
+ * Walk all reference sites and look up targets in the symbol table.
25
+ * References that don't resolve produce diagnostics. Built-in scalar
26
+ * and BSON types are recognized via SCALAR_TYPES / BSON_TYPES and
27
+ * never produce unresolved-type diagnostics.
28
+ *
29
+ * Two passes handle forward references (a Type declared at end of file
30
+ * can be referenced from a field declared at the top) and circular
31
+ * imports (cycles are already collapsed by `flatten()` / cycles in P5
32
+ * resolution; the resolver just sees the merged namespace).
33
+ *
34
+ * The resolver flattens its input internally, so callers don't need to
35
+ * `flatten()` first. Callers that want to surface diagnostics tied to
36
+ * the original (provenance-preserving) AST can map positions back via
37
+ * span comparison; in practice the cloned declarations' spans point
38
+ * into the importing file's clone block, which is where the user can
39
+ * edit them, so the natural workflow works correctly.
40
+ */
41
+ import type { EntityDeclaration, EnumDeclaration, Position, Span, TopLevelStatement, TypeDeclaration, XDbmlDocument } from './ast.ts';
42
+ /**
43
+ * Kind of declaration a symbol refers to. Mirrors the declaration AST
44
+ * shape vocabulary; useful for diagnostics that want to say
45
+ * "expected an entity, found a type" or similar.
46
+ */
47
+ export type SymbolKind = 'entity' | 'type' | 'enum' | 'container' | 'edge' | 'view' | 'tablegroup' | 'tablepartial' | 'note';
48
+ /**
49
+ * One entry in the symbol table. Carries the declaration node (so
50
+ * downstream consumers can navigate), the canonical qualified name, and
51
+ * a source position for diagnostics.
52
+ */
53
+ export interface SymbolEntry {
54
+ /** The fully-qualified, dot-separated name (e.g., `core.dim_customer`). */
55
+ qualifiedName: string;
56
+ /** The bare, unqualified name as it appears in source. */
57
+ name: string;
58
+ /** Container the symbol lives in, if any (top-level entries leave this undefined). */
59
+ containerName?: string;
60
+ kind: SymbolKind;
61
+ /** The declaration node. Type narrows on `kind`. */
62
+ declaration: EntityDeclaration | TypeDeclaration | EnumDeclaration | TopLevelStatement;
63
+ /** Source position of the declaration (start of declaration). */
64
+ position: Position;
65
+ }
66
+ /**
67
+ * Stable diagnostic code. Tooling can match on these to filter or style
68
+ * messages without parsing the human-readable text.
69
+ */
70
+ export type DiagnosticCode = 'duplicate-declaration' | 'unresolved-type' | 'unresolved-entity' | 'unresolved-field' | 'unresolved-partial' | 'unresolved-tablegroup-member' | 'unresolved-records-entity' | 'unresolved-records-column' | 'empty-import' | 'invalid-nested-path';
71
+ /**
72
+ * A single resolution diagnostic. Severity is currently always `error`,
73
+ * but the field is included to leave room for future warnings (e.g.,
74
+ * style concerns like "redundant alias matches original name").
75
+ *
76
+ * Position is given as a Span (start + end) rather than a single Position
77
+ * so editor integrations (Monaco markers, LSP servers) can underline the
78
+ * exact offending construct rather than guessing where the squiggle
79
+ * should end. Each diagnostic's span corresponds to an AST node's own
80
+ * span (a field's type expression, a path endpoint, an entity name).
81
+ */
82
+ export interface Diagnostic {
83
+ severity: 'error' | 'warning';
84
+ code: DiagnosticCode;
85
+ message: string;
86
+ span: Span;
87
+ }
88
+ /**
89
+ * The result of `resolveNames(doc)`. The symbol table is consultable for
90
+ * downstream queries (e.g., "given a name, find the declaration"); the
91
+ * diagnostics list is for surfacing problems.
92
+ */
93
+ export interface ResolutionResult {
94
+ diagnostics: Diagnostic[];
95
+ symbols: SymbolTable;
96
+ }
97
+ /**
98
+ * Read-only handle on the collected symbol table.
99
+ *
100
+ * Lookup is by qualified name (e.g., `core.dim_customer`). The class
101
+ * also exposes a `lookupBare()` for the common case where the name has
102
+ * no container prefix and the caller wants to find the unique match
103
+ * (returns undefined if ambiguous or missing).
104
+ */
105
+ export declare class SymbolTable {
106
+ private readonly byQualified;
107
+ private readonly byBare;
108
+ constructor(entries: ReadonlyArray<SymbolEntry>);
109
+ /** Look up by canonical qualified name. */
110
+ lookup(qualifiedName: string): SymbolEntry | undefined;
111
+ /**
112
+ * Look up by bare name. Returns the unique entry if exactly one match,
113
+ * or undefined when missing or ambiguous (multiple containers contain
114
+ * an entry with this bare name). For ambiguous cases, callers should
115
+ * inspect `lookupAllBare()` if they want to disambiguate.
116
+ */
117
+ lookupBare(name: string): SymbolEntry | undefined;
118
+ /** Look up by bare name; returns all matches. */
119
+ lookupAllBare(name: string): ReadonlyArray<SymbolEntry>;
120
+ /** Iterate all entries in declaration order. */
121
+ entries(): IterableIterator<SymbolEntry>;
122
+ /** Total number of entries. */
123
+ get size(): number;
124
+ }
125
+ /**
126
+ * Resolve names in an xDBML document. Flattens the AST internally
127
+ * (so callers don't need to call `flatten()` first), then runs the
128
+ * two-pass resolution algorithm. Returns diagnostics and the symbol
129
+ * table.
130
+ *
131
+ * Cost is roughly linear in (declarations + reference sites). For
132
+ * typical schemas (10s-100s of entities) this is fast enough to run
133
+ * on every keystroke in an interactive editor.
134
+ */
135
+ export declare function resolveNames(doc: XDbmlDocument): ResolutionResult;