@dforge-core/metadata 0.0.20 → 0.0.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,36 @@
1
+ // @dforge-core/metadata/dsl — the action-DSL analyzer.
2
+ //
3
+ // A separate entry point, not part of the root barrel: this is a lexer, a
4
+ // parser and a rule engine, while the root export is registries and types that
5
+ // the web app and the VS Code webview pull into a browser bundle. Importing
6
+ // `@dforge-core/metadata` must not drag a tokenizer in behind it.
7
+ //
8
+ // Consumers: the language server (diagnostics, completion, hover,
9
+ // go-to-definition) and the module validator that runs before a pack.
10
+
11
+ export { tokenize, stringValue } from "./lexer";
12
+ export type { Token, TokenKind } from "./lexer";
13
+
14
+ export { parseDsl, BLOCK_KINDS } from "./parse";
15
+ export type {
16
+ BlockKind,
17
+ CallRef,
18
+ DslBlock,
19
+ DslDocument,
20
+ DslParam,
21
+ FieldRef,
22
+ GlobalRef,
23
+ Span,
24
+ } from "./parse";
25
+
26
+ export { BUILTINS, BUILTIN_BY_NAME, BUILTIN_VALUES } from "./builtins";
27
+ export type { Builtin } from "./builtins";
28
+
29
+ export { checkDsl } from "./check";
30
+ export type {
31
+ ColumnLookup,
32
+ DslContext,
33
+ DslIssue,
34
+ DslSeverity,
35
+ EntityShape,
36
+ } from "./types";
@@ -0,0 +1,364 @@
1
+ // Hand-rolled lexer for the dForge action DSL.
2
+ //
3
+ // Deliberately not the tree-sitter grammar: that one exists to be compiled into
4
+ // Zed, and pulling a native parser (or a wasm runtime) into an npx-launched
5
+ // language server — or into a validator that runs before a pack — costs far
6
+ // more than it buys for the handful of constructs the rules need.
7
+
8
+ export type TokenKind =
9
+ | "ident"
10
+ | "number"
11
+ | "string"
12
+ | "template"
13
+ | "regex"
14
+ | "punct"
15
+ | "comment";
16
+
17
+ export interface Token {
18
+ kind: TokenKind;
19
+ /** Source text, verbatim — quotes included for strings. */
20
+ text: string;
21
+ start: number;
22
+ end: number;
23
+ line: number;
24
+ character: number;
25
+ /** True when a line break sits between this token and the previous one. */
26
+ startsLine: boolean;
27
+ }
28
+
29
+ const PUNCT3 = ["===", "!=="];
30
+ const PUNCT2 = [
31
+ "==", "!=", "<=", ">=", "&&", "||", "++", "--",
32
+ "+=", "-=", "*=", "/=", "%=", "${", "=>",
33
+ ];
34
+
35
+ /**
36
+ * Template literals and the `${…}` holes in them. Inside a template we are
37
+ * reading literal text; inside a hole we are reading code again, and the brace
38
+ * count says which `}` closes it.
39
+ */
40
+ type Frame =
41
+ | { kind: "template"; chunkStart: number }
42
+ | { kind: "interp"; braces: number };
43
+
44
+ /**
45
+ * Keywords whose `(…)` holds a condition rather than a parameter list. What
46
+ * follows that `)` is a statement — so a regex may open there, and the brace
47
+ * after it is not a function body.
48
+ */
49
+ export const CONTROL_KEYWORDS = new Set([
50
+ "if",
51
+ "while",
52
+ "for",
53
+ "switch",
54
+ "catch",
55
+ "with",
56
+ ]);
57
+
58
+ export function tokenize(text: string): Token[] {
59
+ const tokens: Token[] = [];
60
+ let i = 0;
61
+ let line = 0;
62
+ let lineStart = 0;
63
+ let pendingLineBreak = true;
64
+ const frames: Frame[] = [];
65
+ // Whether each open paren belongs to a control condition, so that the `)`
66
+ // closing it is known to be followed by a statement.
67
+ const parenIsControl: boolean[] = [];
68
+ let closedControlParen = false;
69
+ // Whether each open brace began a statement block rather than an object,
70
+ // so that the `}` closing it is known to leave a statement position.
71
+ const braceIsBlock: boolean[] = [];
72
+ let closedBlockBrace = false;
73
+
74
+ const push = (kind: TokenKind, start: number, end: number) => {
75
+ tokens.push({
76
+ kind,
77
+ text: text.slice(start, end),
78
+ start,
79
+ end,
80
+ line,
81
+ character: start - lineStart,
82
+ startsLine: pendingLineBreak,
83
+ });
84
+ pendingLineBreak = false;
85
+ };
86
+
87
+ while (i < text.length) {
88
+ const frame = frames[frames.length - 1];
89
+
90
+ // ── literal text inside a template ──────────────────────────────
91
+ // Emitted as one token per chunk, so the analyzer sees the code in
92
+ // `${…}` but never the prose around it. The `${` closes the chunk
93
+ // before it and the `}` opens the chunk after, which keeps the brace
94
+ // count of anything reading these tokens balanced.
95
+ if (frame?.kind === "template") {
96
+ const startLine = line;
97
+ const startLineStart = lineStart;
98
+ while (i < text.length) {
99
+ const ch = text[i]!;
100
+ if (ch === "\\") {
101
+ i += 2;
102
+ continue;
103
+ }
104
+ if (ch === "`" || (ch === "$" && text[i + 1] === "{")) break;
105
+ if (ch === "\n") {
106
+ line++;
107
+ lineStart = i + 1;
108
+ }
109
+ i++;
110
+ }
111
+ if (text[i] === "`") {
112
+ i++;
113
+ frames.pop();
114
+ } else if (i < text.length) {
115
+ i += 2;
116
+ frames.push({ kind: "interp", braces: 0 });
117
+ } else {
118
+ frames.pop(); // unterminated
119
+ }
120
+ const savedLine = line;
121
+ const savedLineStart = lineStart;
122
+ line = startLine;
123
+ lineStart = startLineStart;
124
+ push("template", frame.chunkStart, i);
125
+ line = savedLine;
126
+ lineStart = savedLineStart;
127
+ continue;
128
+ }
129
+
130
+ const c = text[i]!;
131
+
132
+ // A leading BOM is not a column. The compiler strips it before anchoring
133
+ // block headers at `^execute:`, so the first real token must still land
134
+ // at character 0 here or every block-scoped rule stands down.
135
+ if (c === "\uFEFF" && i === 0) {
136
+ i++;
137
+ lineStart = i;
138
+ continue;
139
+ }
140
+
141
+ if (c === "\n") {
142
+ i++;
143
+ line++;
144
+ lineStart = i;
145
+ pendingLineBreak = true;
146
+ continue;
147
+ }
148
+ if (c === " " || c === "\t" || c === "\r") {
149
+ i++;
150
+ continue;
151
+ }
152
+
153
+ if (c === "/" && text[i + 1] === "/") {
154
+ const start = i;
155
+ while (i < text.length && text[i] !== "\n") i++;
156
+ push("comment", start, i);
157
+ continue;
158
+ }
159
+ if (c === "/" && text[i + 1] === "*") {
160
+ const start = i;
161
+ const startLine = line;
162
+ const startLineStart = lineStart;
163
+ i += 2;
164
+ while (i < text.length && !(text[i] === "*" && text[i + 1] === "/")) {
165
+ if (text[i] === "\n") {
166
+ line++;
167
+ lineStart = i + 1;
168
+ }
169
+ i++;
170
+ }
171
+ i = Math.min(i + 2, text.length);
172
+ // Report the comment at its opening position, not its end.
173
+ const savedLine = line;
174
+ const savedLineStart = lineStart;
175
+ line = startLine;
176
+ lineStart = startLineStart;
177
+ push("comment", start, i);
178
+ line = savedLine;
179
+ lineStart = savedLineStart;
180
+ continue;
181
+ }
182
+
183
+ if (c === "'" || c === '"') {
184
+ const start = i;
185
+ const quote = c;
186
+ i++;
187
+ while (i < text.length && text[i] !== quote && text[i] !== "\n") {
188
+ if (text[i] === "\\") i++;
189
+ i++;
190
+ }
191
+ if (text[i] === quote) i++;
192
+ push("string", start, i);
193
+ continue;
194
+ }
195
+
196
+ if (c === "`") {
197
+ frames.push({ kind: "template", chunkStart: i });
198
+ i++;
199
+ continue;
200
+ }
201
+
202
+ if (
203
+ c === "/" &&
204
+ regexAllowedAfter(
205
+ tokens[tokens.length - 1],
206
+ closedControlParen,
207
+ closedBlockBrace,
208
+ )
209
+ ) {
210
+ const end = scanRegex(text, i);
211
+ if (end > 0) {
212
+ push("regex", i, end);
213
+ i = end;
214
+ continue;
215
+ }
216
+ }
217
+
218
+ if (/[0-9]/.test(c)) {
219
+ const start = i;
220
+ while (i < text.length && /[0-9._eE]/.test(text[i]!)) i++;
221
+ push("number", start, i);
222
+ continue;
223
+ }
224
+
225
+ if (/[A-Za-z_$]/.test(c)) {
226
+ const start = i;
227
+ while (i < text.length && /[A-Za-z0-9_$]/.test(text[i]!)) i++;
228
+ push("ident", start, i);
229
+ continue;
230
+ }
231
+
232
+ if (c === "(") {
233
+ const prev = tokens[tokens.length - 1];
234
+ parenIsControl.push(
235
+ prev?.kind === "ident" && CONTROL_KEYWORDS.has(prev.text),
236
+ );
237
+ } else if (c === ")") {
238
+ closedControlParen = parenIsControl.pop() ?? false;
239
+ }
240
+
241
+ if (frame?.kind === "interp") {
242
+ if (c === "}" && frame.braces === 0) {
243
+ frames.pop();
244
+ // The `}` opens the next literal chunk rather than becoming a
245
+ // token of its own — see the chunk comment above.
246
+ (frames[frames.length - 1] as { chunkStart: number }).chunkStart = i;
247
+ i++;
248
+ continue;
249
+ }
250
+ if (c === "{") frame.braces++;
251
+ else if (c === "}") frame.braces--;
252
+ }
253
+
254
+ if (c === "{") {
255
+ braceIsBlock.push(isBlockBrace(tokens[tokens.length - 1]));
256
+ } else if (c === "}") {
257
+ closedBlockBrace = braceIsBlock.pop() ?? false;
258
+ }
259
+
260
+ const three = text.slice(i, i + 3);
261
+ if (PUNCT3.includes(three)) {
262
+ push("punct", i, i + 3);
263
+ i += 3;
264
+ continue;
265
+ }
266
+ const two = text.slice(i, i + 2);
267
+ if (PUNCT2.includes(two)) {
268
+ push("punct", i, i + 2);
269
+ i += 2;
270
+ continue;
271
+ }
272
+ push("punct", i, i + 1);
273
+ i++;
274
+ }
275
+
276
+ return tokens;
277
+ }
278
+
279
+ /**
280
+ * Whether a `/` here opens a regex literal rather than dividing. The usual
281
+ * rule: a regex may start where an expression may start, so it follows an
282
+ * operator or a keyword but not a value.
283
+ *
284
+ * A `)` depends on what it closed: after a control condition a statement
285
+ * follows, so `if (x) /re/.test(s)` is a regex, while `(a + b) / 2` divides.
286
+ * A template chunk likewise depends on which end it is — one that stops at
287
+ * `${` opens an expression, one that ran to the closing backtick is a value.
288
+ *
289
+ * A `}` likewise depends on what it closed: ending a statement block leaves a
290
+ * statement position, so `if (x) {}` may be followed by a regex, while an
291
+ * object literal is a value and divides.
292
+ */
293
+ function regexAllowedAfter(
294
+ prev: Token | undefined,
295
+ closedControlParen: boolean,
296
+ closedBlockBrace: boolean,
297
+ ): boolean {
298
+ if (!prev) return true;
299
+ if (prev.kind === "number" || prev.kind === "string") return false;
300
+ if (prev.kind === "template") return prev.text.endsWith("${");
301
+ if (prev.kind === "ident") return REGEX_PRECEDING_KEYWORDS.has(prev.text);
302
+ if (prev.kind === "regex") return false;
303
+ if (prev.kind === "comment") return true;
304
+ if (prev.text === ")") return closedControlParen;
305
+ if (prev.text === "}") return closedBlockBrace;
306
+ return !REGEX_BLOCKING_PUNCT.has(prev.text);
307
+ }
308
+
309
+ /**
310
+ * Whether the `{` after `prev` opens a statement block rather than an object
311
+ * literal. A block follows a condition, another block, a statement end or one
312
+ * of the keywords that take a bare body; anything else — an `=`, a `(`, a `,`,
313
+ * a `=>` — is expression position, where the brace is a value.
314
+ */
315
+ function isBlockBrace(prev: Token | undefined): boolean {
316
+ if (!prev) return true;
317
+ if (prev.kind === "ident") return BLOCK_PRECEDING_KEYWORDS.has(prev.text);
318
+ if (prev.kind !== "punct") return false;
319
+ return prev.text === ")" || prev.text === "}" || prev.text === "{" || prev.text === ";";
320
+ }
321
+
322
+ const BLOCK_PRECEDING_KEYWORDS = new Set(["else", "do", "try", "finally"]);
323
+
324
+ const REGEX_PRECEDING_KEYWORDS = new Set([
325
+ "return", "typeof", "instanceof", "in", "of", "new", "delete", "void",
326
+ "case", "do", "else", "throw", "yield", "await",
327
+ ]);
328
+
329
+ const REGEX_BLOCKING_PUNCT = new Set(["]", "++", "--"]);
330
+
331
+ /**
332
+ * End offset of the regex literal starting at `start`, or -1 if it is not one.
333
+ * A `/` inside a character class does not close the literal, and a literal
334
+ * never spans a line — hitting one means this was a division after all.
335
+ */
336
+ function scanRegex(text: string, start: number): number {
337
+ let i = start + 1;
338
+ let inClass = false;
339
+ // An empty pattern would be the `//` comment the caller already handled.
340
+ if (text[i] === "/" || text[i] === "*") return -1;
341
+ for (; i < text.length; i++) {
342
+ const c = text[i]!;
343
+ if (c === "\\") {
344
+ i++;
345
+ continue;
346
+ }
347
+ if (c === "\n") return -1;
348
+ if (c === "[") inClass = true;
349
+ else if (c === "]") inClass = false;
350
+ else if (c === "/" && !inClass) {
351
+ i++;
352
+ while (i < text.length && /[a-z]/.test(text[i]!)) i++;
353
+ return i;
354
+ }
355
+ }
356
+ return -1;
357
+ }
358
+
359
+ /** Strip the surrounding quotes from a string token's text. */
360
+ export function stringValue(token: Token): string {
361
+ if (token.kind !== "string") return token.text;
362
+ const body = token.text.slice(1, token.text.endsWith(token.text[0]!) && token.text.length > 1 ? -1 : undefined);
363
+ return body.replace(/\\(.)/g, "$1");
364
+ }