@xdbml/parse 0.1.0-poc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,277 @@
1
+ /**
2
+ * Shared keyword vocabulary for xDBML tokenizers and highlighters.
3
+ *
4
+ * This module is the single source of truth for the keyword lists
5
+ * that drive syntax highlighting across multiple surfaces:
6
+ *
7
+ * - parser/src/monarch.ts the playground's in-editor highlighter
8
+ * - tools/textmate/... the TextMate grammar for VS Code,
9
+ * Shiki (xdbml.org code blocks, Claude
10
+ * chat code blocks, etc.), and GitHub
11
+ *
12
+ * The grammar in grammar/xDBML.g4 is the language's canonical
13
+ * specification; this file mirrors its keyword vocabulary in a form
14
+ * convenient for consumers. A keyword-consistency test (in
15
+ * parser/test/) asserts that every keyword listed here is recognized
16
+ * by the parser.
17
+ *
18
+ * When adding a keyword to xDBML:
19
+ * 1. Update the grammar in grammar/xDBML.g4
20
+ * 2. Update parser/src/parser.ts if the parser needs to recognize it
21
+ * 3. Add it to the right array below
22
+ * 4. Re-run the TextMate grammar build script
23
+ * (tools/textmate/scripts/build.mjs)
24
+ * 5. Run `npm test` in the parser package to verify all three
25
+ * consumers see the keyword
26
+ *
27
+ * All keywords here are case-insensitive in xDBML. Stored in
28
+ * lower-case as the canonical form; matchers should be case-insensitive.
29
+ */
30
+ /* -------------------------------------------------------------------------
31
+ * Declaration keywords
32
+ *
33
+ * The keywords that introduce a top-level declaration. The full set
34
+ * includes both container-style keywords (Container, Schema, etc.)
35
+ * and entity-style keywords (Table, Entity, etc.) plus other
36
+ * top-level constructs (Ref, View, Note, Enum, etc.).
37
+ * ----------------------------------------------------------------------- */
38
+ export const CONTAINER_KEYWORDS = [
39
+ 'container',
40
+ 'schema',
41
+ 'database',
42
+ 'keyspace',
43
+ 'namespace',
44
+ 'dataset',
45
+ 'bucket',
46
+ ];
47
+ export const ENTITY_KEYWORDS = [
48
+ 'table',
49
+ 'entity',
50
+ 'collection',
51
+ 'record',
52
+ ];
53
+ /**
54
+ * The full set of declaration keywords. Includes containers,
55
+ * entities, and other top-level constructs.
56
+ */
57
+ export const DECLARATION_KEYWORDS = [
58
+ 'project',
59
+ ...CONTAINER_KEYWORDS,
60
+ ...ENTITY_KEYWORDS,
61
+ 'type',
62
+ 'edge',
63
+ 'view',
64
+ 'enum',
65
+ 'ref',
66
+ 'note',
67
+ 'tablepartial',
68
+ 'tablegroup',
69
+ 'diagramview',
70
+ ];
71
+ /* -------------------------------------------------------------------------
72
+ * Type expression keywords
73
+ *
74
+ * Keywords that appear inside a field's type expression, after the
75
+ * colon. Includes structural types (object, array, map) and
76
+ * polymorphism markers (oneOf, anyOf, allOf, union).
77
+ * ----------------------------------------------------------------------- */
78
+ export const STRUCTURAL_TYPE_KEYWORDS = [
79
+ 'object',
80
+ 'struct',
81
+ 'array',
82
+ 'list',
83
+ 'map',
84
+ 'dict',
85
+ 'dictionary',
86
+ 'set',
87
+ 'json',
88
+ 'jsonb',
89
+ 'variant',
90
+ ];
91
+ export const POLYMORPHISM_KEYWORDS = [
92
+ 'union',
93
+ 'oneof',
94
+ 'anyof',
95
+ 'allof',
96
+ ];
97
+ /* -------------------------------------------------------------------------
98
+ * Scalar and BSON types
99
+ *
100
+ * The parser accepts any identifier as a scalar type (open vocabulary),
101
+ * so these lists drive color, not validation. Themes color them under
102
+ * `storage.type` (TextMate) or `type` (Monarch).
103
+ * ----------------------------------------------------------------------- */
104
+ export const SCALAR_TYPES = [
105
+ // Integers
106
+ 'tinyint', 'smallint', 'mediumint', 'int', 'integer', 'bigint',
107
+ 'int32', 'int64',
108
+ // Floating point and decimal
109
+ 'float', 'double', 'decimal', 'dec', 'numeric', 'real',
110
+ // Boolean
111
+ 'bit', 'bool', 'boolean',
112
+ // Strings
113
+ 'char', 'varchar', 'varchar2', 'nvarchar', 'nvarchar2', 'nchar',
114
+ 'text', 'mediumtext', 'longtext', 'string', 'ntext',
115
+ // Binary
116
+ 'binary', 'varbinary', 'blob', 'mediumblob', 'longblob', 'tinyblob',
117
+ 'tinytext',
118
+ // Document
119
+ 'json', 'jsonb', 'variant', 'xml',
120
+ // Date/time
121
+ 'date', 'time', 'datetime', 'datetime2', 'timestamp',
122
+ 'timestamptz', 'year',
123
+ // Identity and network
124
+ 'uuid', 'inet6',
125
+ // Money
126
+ 'money', 'smallmoney',
127
+ // Enumeration as a type position
128
+ 'enum',
129
+ ];
130
+ export const BSON_TYPES = [
131
+ 'objectid',
132
+ 'decimal128',
133
+ 'bindata',
134
+ 'minkey',
135
+ 'maxkey',
136
+ 'symbol',
137
+ 'regex',
138
+ 'long',
139
+ 'double',
140
+ ];
141
+ /* -------------------------------------------------------------------------
142
+ * Settings vocabulary
143
+ *
144
+ * Settings appear inside `[...]` brackets after a field declaration
145
+ * or other construct. Two forms:
146
+ *
147
+ * - Flag settings: bare names like `pk`, `unique`, `not null`
148
+ * - Keyed settings: `name: value` pairs
149
+ *
150
+ * `required` is normalized to `not null` by the parser per spec §8;
151
+ * both spellings get the same color.
152
+ *
153
+ * Setting keys are open-vocabulary at the parser level (any identifier
154
+ * followed by `:` is accepted as a setting), so this list drives
155
+ * highlighting only.
156
+ * ----------------------------------------------------------------------- */
157
+ export const SETTING_FLAGS = [
158
+ 'pk',
159
+ 'primary',
160
+ 'key',
161
+ 'unique',
162
+ 'null',
163
+ 'not',
164
+ 'required',
165
+ 'increment',
166
+ 'inactive', // v0.2 §11.9: Ref flag for visualization-only deactivation
167
+ ];
168
+ export const SETTING_KEYS = [
169
+ // General
170
+ 'note',
171
+ 'default',
172
+ 'ref',
173
+ 'name',
174
+ 'color',
175
+ 'headercolor',
176
+ 'as',
177
+ 'check',
178
+ // xDBML-specific
179
+ 'type',
180
+ 'target',
181
+ 'targets',
182
+ 'database_type',
183
+ 'source',
184
+ 'source_cardinality',
185
+ 'target_cardinality',
186
+ 'min_source',
187
+ 'max_source',
188
+ 'min_target',
189
+ 'max_target',
190
+ 'undirected',
191
+ 'discriminator',
192
+ 'source_query',
193
+ 'materialized',
194
+ 'refresh_schedule',
195
+ 'refresh_on',
196
+ 'source_database',
197
+ 'storage_options',
198
+ // Validation
199
+ 'pattern',
200
+ 'format',
201
+ 'minlength',
202
+ 'maxlength',
203
+ 'minimum',
204
+ 'maximum',
205
+ 'exclusiveminimum',
206
+ 'exclusivemaximum',
207
+ 'multipleof',
208
+ 'minitems',
209
+ 'maxitems',
210
+ 'uniqueitems',
211
+ 'minproperties',
212
+ 'maxproperties',
213
+ // AI-readiness
214
+ 'synonyms',
215
+ 'business_term',
216
+ 'granularity',
217
+ 'tags',
218
+ // Referential actions
219
+ 'delete',
220
+ 'update',
221
+ // Block keywords (entity-body or top-level)
222
+ 'indexes',
223
+ 'checks', // v0.2 §10: entity-level checks block
224
+ // Container settings
225
+ 'replication',
226
+ 'location',
227
+ 'default_charset',
228
+ // Module system (v0.2 §26)
229
+ 'cloned_at', // v0.2 directive setting: ISO 8601 timestamp on a use/reuse directive
230
+ ];
231
+ /* -------------------------------------------------------------------------
232
+ * Value vocabularies
233
+ *
234
+ * Constants that appear as right-hand-side values for specific
235
+ * settings (granularity for time-series, etc.). The parser doesn't
236
+ * validate these; they're highlighted as constants when recognized.
237
+ * ----------------------------------------------------------------------- */
238
+ export const GRANULARITY_VALUES = [
239
+ 'year', 'quarter', 'month', 'week', 'day',
240
+ 'hour', 'minute', 'second',
241
+ 'millisecond', 'microsecond', 'nanosecond',
242
+ ];
243
+ /* -------------------------------------------------------------------------
244
+ * Top-level directives
245
+ *
246
+ * Appear only at the very top of a file: `xdbml: 0.1` or
247
+ * `experimental: ...`. Distinct from declaration keywords because
248
+ * their syntax (and meaning) is different.
249
+ * ----------------------------------------------------------------------- */
250
+ export const DIRECTIVE_KEYWORDS = [
251
+ 'xdbml',
252
+ 'experimental',
253
+ ];
254
+ /* -------------------------------------------------------------------------
255
+ * Module-system keywords (v0.2, spec §26)
256
+ *
257
+ * The keywords that introduce a module-system directive:
258
+ *
259
+ * reuse { entity X, type Y as Z } from './path' [cloned_at: '...'] { ... }
260
+ * use * from './path'
261
+ *
262
+ * `use` and `reuse` start the directive; `from` precedes the path
263
+ * literal; `as` appears inside the import-item list when renaming.
264
+ *
265
+ * `as` is also a contextual keyword in upstream DBML (`Table users as u`
266
+ * for aliasing), so it's already implicitly highlighted via its
267
+ * presence in SETTING_KEYS; including it here as well doesn't change
268
+ * behavior but documents the module-system role.
269
+ *
270
+ * Highlighted with the scope `keyword.control.module.xdbml`.
271
+ * ----------------------------------------------------------------------- */
272
+ export const MODULE_KEYWORDS = [
273
+ 'use',
274
+ 'reuse',
275
+ 'from',
276
+ 'as',
277
+ ];
@@ -0,0 +1,91 @@
1
+ /**
2
+ * xDBML lexer.
3
+ *
4
+ * Produces a stream of tokens with line/column positions. Keywords are NOT
5
+ * recognized as distinct token kinds at the lexer level. Identifiers carry
6
+ * their source value, and the parser interprets them as keywords by
7
+ * lowercased string comparison (per spec §3.8: keywords are
8
+ * case-insensitive, identifiers are case-sensitive — both end up as
9
+ * IDENTIFIER tokens here, with the parser making the keyword decision).
10
+ *
11
+ * Special multi-character punctuation handled here rather than in the parser:
12
+ * - `<>` is lexed as MANY_TO_MANY (otherwise the parser would see `<` then `>`)
13
+ * - `[*]` is lexed as ARRAY_WILDCARD only when followed-by/preceded-by a dot
14
+ * in a path context; otherwise it's left as `LBRACKET`, `OP(*)`, `RBRACKET`
15
+ * to avoid ambiguity with array settings. In practice the parser only
16
+ * uses ARRAY_WILDCARD inside fieldPath, so we lex `[*]` greedily whenever
17
+ * we see it and let the parser decide based on context.
18
+ * - `'''...'''` lexes as a single STRING_LITERAL with the multiline flag.
19
+ */
20
+ import type { Position } from './ast.ts';
21
+ export declare const TokenKind: {
22
+ readonly Identifier: "Identifier";
23
+ readonly QuotedIdentifier: "QuotedIdentifier";
24
+ readonly StringLiteral: "StringLiteral";
25
+ readonly MultilineString: "MultilineString";
26
+ readonly NumberLiteral: "NumberLiteral";
27
+ readonly ExpressionLiteral: "ExpressionLiteral";
28
+ readonly LBrace: "LBrace";
29
+ readonly RBrace: "RBrace";
30
+ readonly LBracket: "LBracket";
31
+ readonly RBracket: "RBracket";
32
+ readonly LParen: "LParen";
33
+ readonly RParen: "RParen";
34
+ readonly Comma: "Comma";
35
+ readonly Colon: "Colon";
36
+ readonly Dot: "Dot";
37
+ readonly Tilde: "Tilde";
38
+ readonly Semicolon: "Semicolon";
39
+ readonly LAngle: "LAngle";
40
+ readonly RAngle: "RAngle";
41
+ readonly Minus: "Minus";
42
+ readonly ManyToMany: "ManyToMany";
43
+ readonly ArrayWildcard: "ArrayWildcard";
44
+ /**
45
+ * Standalone `*`. Used by v0.2 module-system import-all directives:
46
+ * `use * from './path'`. Note that `[*]` is a separate token
47
+ * (ArrayWildcard); this Star token is only produced when the asterisk
48
+ * appears outside that context.
49
+ */
50
+ readonly Star: "Star";
51
+ readonly EOF: "EOF";
52
+ };
53
+ export type TokenKind = typeof TokenKind[keyof typeof TokenKind];
54
+ export interface Token {
55
+ kind: TokenKind;
56
+ /** The raw source text of the token */
57
+ text: string;
58
+ /** For StringLiteral / MultilineString / QuotedIdentifier: text with quotes stripped & escapes processed */
59
+ value?: string;
60
+ start: Position;
61
+ end: Position;
62
+ }
63
+ export declare class LexError extends Error {
64
+ position: Position;
65
+ constructor(message: string, position: Position);
66
+ }
67
+ export declare class Lexer {
68
+ private text;
69
+ private offset;
70
+ private line;
71
+ private column;
72
+ constructor(text: string);
73
+ private pos;
74
+ private peek;
75
+ private advance;
76
+ private matchSeq;
77
+ private isAtEnd;
78
+ private skipTrivia;
79
+ private isIdentStart;
80
+ private isIdentCont;
81
+ private isDigit;
82
+ private lexIdentifier;
83
+ private lexNumber;
84
+ private lexString;
85
+ private lexQuotedIdent;
86
+ private lexBacktick;
87
+ /** Read a single token. Returns EOF when out of input. */
88
+ private nextToken;
89
+ tokenize(): Token[];
90
+ }
91
+ export declare function tokenize(text: string): Token[];