@rohal12/spindle 0.53.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +3 -1
  2. package/dist/pkg/format.js +1 -1
  3. package/dist/pkg/headless.js +10058 -2075
  4. package/dist/pkg/macro-registry.json +4 -4
  5. package/dist/pkg/story-variables.js +10197 -1925
  6. package/dist/pkg/tooling.js +15 -1
  7. package/package.json +3 -1
  8. package/src/code-check.ts +314 -0
  9. package/src/components/Passage.tsx +3 -3
  10. package/src/components/PassageDialog.tsx +2 -4
  11. package/src/components/StoryInterface.tsx +2 -4
  12. package/src/components/macros/Do.tsx +1 -1
  13. package/src/components/macros/Goto.tsx +19 -8
  14. package/src/components/macros/Include.tsx +23 -11
  15. package/src/components/macros/Set.tsx +1 -1
  16. package/src/components/macros/VarDisplay.tsx +1 -1
  17. package/src/components/macros/Widget.tsx +5 -33
  18. package/src/components/macros/macro-args.ts +16 -7
  19. package/src/define-macro.ts +2 -0
  20. package/src/expression.ts +25 -20
  21. package/src/index.tsx +63 -37
  22. package/src/interpolation.ts +4 -5
  23. package/src/js-lexer.ts +981 -1342
  24. package/src/markup/ast.ts +11 -226
  25. package/src/markup/code-attributes.ts +3 -3
  26. package/src/markup/code-end.ts +121 -0
  27. package/src/markup/parse.ts +118 -0
  28. package/src/markup/render.tsx +1 -1
  29. package/src/markup/spindle.d.peggy.ts +24 -0
  30. package/src/markup/spindle.peggy +463 -0
  31. package/src/markup/tokens.ts +95 -0
  32. package/src/markup/validate.ts +281 -0
  33. package/src/parser.ts +19 -4
  34. package/src/registry.ts +49 -14
  35. package/src/runtime-errors.ts +10 -0
  36. package/src/store.ts +8 -2
  37. package/src/story-init.ts +3 -5
  38. package/src/story-variables.ts +49 -8
  39. package/src/tooling.ts +58 -0
  40. package/src/types-drift-check.ts +28 -1
  41. package/src/widgets/widget-def.ts +60 -0
  42. package/types/index.d.ts +7 -3
  43. package/types/tooling.d.ts +53 -3
  44. package/src/markup/tokenizer.ts +0 -1112
package/src/js-lexer.ts CHANGED
@@ -1,27 +1,38 @@
1
1
  /**
2
- * Lexical scanner for the JavaScript in expressions and macro arguments.
2
+ * The JavaScript in story markup, read with acorn.
3
3
  *
4
- * It is the one place that knows where string, template and regex literals
5
- * and comments begin and end. The expression transformer (`expression.ts`)
6
- * and the macro argument splitters (`components/macros/arg-utils.ts`) both
7
- * walk source text through `lexJs` and differ only in what they do with the
8
- * pieces it reports. The passage tokenizer (`markup/tokenizer.ts`) and
9
- * attribute interpolation (`interpolation.ts`) find where the code in a
10
- * `{…}` ends with `findCodeEnd`.
4
+ * acorn reads the code: where string, template and regex literals and
5
+ * comments begin and end, whether `/` opens a regex, and (`parseCode`) the
6
+ * syntax tree. A plugin (`SigilParser`) teaches its tokenizer the two sigils
7
+ * that are no JavaScript: `@name` (locals) anywhere, and `%name`
8
+ * (transients) where an operand is expected, the same place acorn reads `/`
9
+ * as a regex; `%` elsewhere is the modulo operator. `$name` and `_name` are
10
+ * identifiers already; the syntax tree tells references from property names
11
+ * and declarations.
11
12
  *
12
- * It also finds the sigil variable references in code: `$name`, `_name` and
13
- * `@name` where an identifier starts — not inside `a$b` or `ñ_x`, and not
14
- * where a property name stands: after `.`, as an object literal key or a
15
- * class member name — and `%name`.
13
+ * - `parseCode`: the references and string text of well-formed code, or a
14
+ * `CodeSyntaxError` that says what is wrong and where. The expression
15
+ * engine (`expression.ts`) rewrites the references it finds, and the
16
+ * story-start check (`markup/validate.ts`) reports the errors.
17
+ * - `findCodeEnd`: where the code in a `{…}` ends, for the markup grammar
18
+ * (`markup/code-end.ts`). -1 when acorn can't read it: the grammar then
19
+ * scans it leniently, as prose-like macro arguments (`{link Don't go}`)
20
+ * need.
21
+ * - `lexJs`: a lenient token walk for the macro argument splitters
22
+ * (`components/macros/arg-utils.ts`), which also read text that is no
23
+ * JavaScript.
16
24
  *
17
- * Besides literals, the scanner tracks whether the next token is an operand
18
- * or an operator, which decides two ambiguities: `/` opens a regex in operand
19
- * position and divides otherwise, and `%name` is a transient reference in
20
- * operand position while `%` after an operand — `($n)%3`, `$a[i] %2`,
21
- * `_i++ %n` — is the modulo operator. For that it tells blocks from object
22
- * literals: `}` closing a block may be followed by a statement, `}` closing
23
- * an object literal by an operator.
25
+ * Restrictions (docs/variables.md "Code in passages"): sigil variables can't
26
+ * be declared (`let _x`, `(_a) => …`) or be property names (`a.@x`), and
27
+ * code must be valid modern JavaScript (a non-strict script).
24
28
  */
29
+ import {
30
+ Parser,
31
+ tokTypes as tt,
32
+ getLineInfo,
33
+ type Options,
34
+ type TokenType,
35
+ } from 'acorn';
25
36
 
26
37
  /** The sigil of a variable reference: story, temporary, local, transient. */
27
38
  export type Sigil = '$' | '_' | '@' | '%';
@@ -35,9 +46,9 @@ export interface JsLexHandlers {
35
46
  code?(ch: string, index: number, nesting: number): void;
36
47
  /**
37
48
  * Literal text passed through verbatim: a string or regex literal, a
38
- * comment, or a piece of a template literal (its backticks, text, escapes
39
- * and the `${` / `}` delimiters around interpolations, whose code is
40
- * reported through `code`).
49
+ * comment, or a piece of a template literal (its backticks, text and the
50
+ * `${` / `}` around interpolations, whose code is reported through
51
+ * `code`).
41
52
  */
42
53
  literal?(text: string, index: number, nesting: number): void;
43
54
  /**
@@ -53,1454 +64,1082 @@ export interface JsLexHandlers {
53
64
  */
54
65
  export type JsGoal = 'expression' | 'statements';
55
66
 
56
- /** Transient name after `%`: an identifier, so `%3` is never a reference. */
57
- const TRANS_NAME_RE = /[A-Za-z_]\w*/y;
58
- /** An assignment operator: `=`, `+=`, `??=`, … but not `==` or `=>`. */
59
- const ASSIGN_OP_RE = /(?:\*\*|<<|>>>?|&&|\|\||\?\?|[-+*/%&|^])?=(?![=>])/y;
60
- /** Flags after the closing `/` of a regex literal. */
61
- const REGEX_FLAGS_RE = /\w*/y;
62
- /**
63
- * Characters of identifiers (with the `\u…` escapes they may contain) and
64
- * numbers. Surrogates stand for the astral identifier characters they encode.
65
- */
66
- const WORD_CHAR_RE = /[\p{ID_Continue}$\u200c\u200d\\\ud800-\udfff]/u;
67
- /** An identifier, or the rest of one after a `$`, `_` or `@` sigil. */
68
- const IDENT_RE = /[\p{ID_Continue}$\u200c\u200d]*/uy;
69
- /** A sigil variable name: the whole identifier after the sigil. */
70
- const VAR_NAME_RE = /^\w+$/;
71
- /** What may start a property name after a `get`, `set`, … modifier. */
72
- const KEY_START_RE = /[\p{ID_Continue}$\\"'[*#]/u;
73
- const SPACE_RE = /\s/;
74
- const LINE_TERMINATOR_RE = /[\n\r\u2028\u2029]/;
75
- const LINE_TERMINATOR_G = /[\n\r\u2028\u2029]/g;
76
- /**
77
- * Keywords followed by an operand rather than an operator (a declaration's
78
- * binding counts as one: `const of of xs`, `let { _a: x } = o`).
79
- */
80
- const OPERAND_KEYWORDS = new Set([
81
- 'await',
82
- 'case',
83
- 'const',
84
- 'delete',
85
- 'do',
86
- 'else',
87
- 'in',
88
- 'instanceof',
89
- 'let',
90
- 'new',
91
- 'return',
92
- 'throw',
93
- 'typeof',
94
- 'var',
95
- 'void',
96
- 'yield',
97
- ]);
98
- /** Keywords whose parenthesised header is followed by a statement. */
99
- const HEADER_KEYWORDS = new Set(['for', 'if', 'while', 'with']);
100
- /** Keywords that a line break ends the statement after. */
101
- const RESTRICTED_KEYWORDS = new Set(['break', 'continue', 'return']);
102
- /** Words that may precede a property name in an object literal or class. */
103
- const MODIFIERS = new Set(['async', 'get', 'set', 'static']);
67
+ /** Code runs through `new Function`: a non-strict script body. */
68
+ const OPTIONS: Options = {
69
+ ecmaVersion: 'latest',
70
+ sourceType: 'script',
71
+ allowReturnOutsideFunction: true,
72
+ };
73
+
74
+ /** `@name`: a local; the name is the whole identifier after the `@`. */
75
+ const AT_NAME_RE = /\w+(?![\p{ID_Continue}$\u200c\u200d])/uy;
76
+ /** `%name`: a transient; `%3` is never one. */
77
+ const TRANS_NAME_RE = /[A-Za-z_]\w*(?![\p{ID_Continue}$\u200c\u200d])/uy;
78
+ /** A `$`/`_` identifier that is a variable reference: `$a`, not `$a$b`. */
79
+ const SIGIL_IDENT_RE = /^[$_]\w+$/;
80
+ const LINE_BREAK_RE = /[\n\r\u2028\u2029]/;
81
+
82
+ /** acorn's tokenizer state, which its typings leave out. */
83
+ interface ParserState {
84
+ input: string;
85
+ pos: number;
86
+ type: TokenType;
87
+ value: unknown;
88
+ start: number;
89
+ end: number;
90
+ exprAllowed: boolean;
91
+ lastTokEnd: number;
92
+ context: unknown[];
93
+ nextToken(): void;
94
+ finishToken(type: TokenType, value?: unknown): void;
95
+ parseExpression(): AnyNode;
96
+ }
104
97
 
105
- /**
106
- * Scan the `"…"` or `'…'` string literal opening at `start`. `end` is the
107
- * index just past its closing quote, or `src.length` when it is unterminated
108
- * (`closed` false). A backslash escapes the character after it, so a quote
109
- * after an even run of backslashes closes the string and one after an odd
110
- * run does not.
111
- */
112
- export function scanStringLiteral(
113
- src: string,
114
- start: number,
115
- ): { end: number; closed: boolean } {
116
- const quote = src.charAt(start);
117
- let i = start + 1;
118
- while (i < src.length) {
119
- const c = src.charAt(i);
120
- if (c === '\\') i += 2;
121
- else if (c === quote) return { end: i + 1, closed: true };
122
- else i++;
123
- }
124
- return { end: src.length, closed: false };
98
+ type Base = new (options: Options, input: string, start?: number) => Parser;
99
+
100
+ /** What follows a member name, but never `function` or `class` keywords. */
101
+ const NOT_A_BODY_RE = /\s*(?:[;},:)]|=(?![=>]))/y;
102
+
103
+ /** Does what follows `pos` show the keyword before it is a name? */
104
+ function notABody(src: string, pos: number): boolean {
105
+ NOT_A_BODY_RE.lastIndex = pos;
106
+ return NOT_A_BODY_RE.test(src);
125
107
  }
126
108
 
109
+ /** Tokens that may end an operand, and so a statement before a line break. */
110
+ const OPERAND_ENDS: ReadonlySet<TokenType> = new Set([
111
+ tt.name,
112
+ tt.num,
113
+ tt.string,
114
+ tt.regexp,
115
+ tt.backQuote,
116
+ tt.bracketR,
117
+ tt.braceR,
118
+ tt.parenR,
119
+ tt.incDec,
120
+ tt._this,
121
+ tt._null,
122
+ tt._true,
123
+ tt._false,
124
+ ]);
125
+
127
126
  /**
128
- * Scan the regex literal (with flags) opening at `start`. `end` is the index
129
- * just past it; an unterminated one (`closed` false) ends at the line break
130
- * (escaped or not) or the end of the source.
127
+ * acorn with the `@name` and `%name` sigils, and three fixes to its guess
128
+ * whether a `{` opens a block or an object literal, which decides whether a
129
+ * `/` after its `}` opens a regex and a `%` a transient: after the `:` of a
130
+ * conditional it opens an object literal (`a ? b : {} / 2`), not a block as
131
+ * after a label; after a block's `}` it opens another block (`{}{} %n = 1`),
132
+ * and so it does on a new line after a statement (`p⏎{} %n = 1`).
131
133
  */
132
- function scanRegex(
133
- src: string,
134
- start: number,
135
- cache?: JsScanCache,
136
- ): { end: number; closed: boolean } {
137
- // The rest of a regex reads the same from just past a class, however the
138
- // scan got there: record how it ends there, and use what is recorded
139
- const pending: number[] = [];
140
- const done = (end: number, closed: boolean) => {
141
- if (cache) {
142
- for (const at of pending) cache.regexes.set(at, end * 2 + +closed);
143
- }
144
- return { end, closed };
145
- };
146
- let i = start + 1;
147
- while (i < src.length) {
148
- // Unterminated at the end of the line: leave the rest to the parser
149
- const lineEnd = regexLineEnd(src, i);
150
- if (lineEnd !== -1) return done(lineEnd, false);
151
- const c = src.charAt(i);
152
- if (c === '\\') {
153
- i += 2;
154
- continue;
155
- }
156
- if (c === '[') {
157
- // A class, where `/` does not close
158
- i = scanRegexClass(src, i, cache);
159
- if (src.charAt(i) !== ']') return done(i, false);
160
- i++;
161
- const known = cache?.regexes.get(i);
162
- if (known !== undefined)
163
- return done(Math.floor(known / 2), known % 2 === 1);
164
- pending.push(i);
165
- continue;
166
- }
167
- if (c === '/') {
168
- REGEX_FLAGS_RE.lastIndex = i + 1;
169
- const flags = REGEX_FLAGS_RE.exec(src)?.[0].length ?? 0;
170
- return done(i + 1 + flags, true);
134
+ const SigilParser = class extends (Parser as unknown as Base) {
135
+ /** Whether an operand could start before the last token, and this one. */
136
+ operandBeforeLast = true;
137
+ operandHere = true;
138
+ /** Whether a line break came before the last token, and this one. */
139
+ breakBeforeLast = false;
140
+ breakHere = false;
141
+ /** Conditional `?`s awaiting their `:`, by context depth. */
142
+ ternaries: number[] = [];
143
+ /** The last `:` ended a conditional's `?`. */
144
+ colonEndsTernary = false;
145
+ /** The last `}` closed a block. */
146
+ closedBlock = false;
147
+ /** The value of the token before the one being finished. */
148
+ prevValue: unknown;
149
+ /** Look ahead after a `%name` starting a line (off in look-aheads). */
150
+ lookahead = true;
151
+ /**
152
+ * Where a `%` is a transient (true) or the modulo operator (false)
153
+ * whatever the tokens before it (see parseCode).
154
+ */
155
+ percentAt?: ReadonlyMap<number, boolean>;
156
+ /** Where it read a `%name` as a transient. */
157
+ transientsRead?: Set<number>;
158
+
159
+ readToken(code: number): void {
160
+ const self = this as unknown as ParserState;
161
+ // acorn updates it only when parsing; its guesses read it when
162
+ // tokenizing too (`return {` on one line opens an object literal)
163
+ self.lastTokEnd = self.end;
164
+ this.operandBeforeLast = this.operandHere;
165
+ this.operandHere = self.exprAllowed;
166
+ this.breakBeforeLast = this.breakHere;
167
+ this.breakHere = LINE_BREAK_RE.test(self.input.slice(self.end, self.pos));
168
+ if (code === 64 || code === 37) {
169
+ const at = self.pos;
170
+ const re = code === 64 ? AT_NAME_RE : TRANS_NAME_RE;
171
+ re.lastIndex = at + 1;
172
+ const name = re.exec(self.input)?.[0];
173
+ if (name) {
174
+ const end = at + 1 + name.length;
175
+ // acorn reads `/` after a prefix `++` as division; `%` there is
176
+ // still a sigil (`++%n`), but after a postfix one modulo (`n++ % 2`)
177
+ const afterPrefix =
178
+ self.type === tt.incDec &&
179
+ (this.operandBeforeLast || this.breakBeforeLast);
180
+ const forced = code === 37 ? this.percentAt?.get(at) : undefined;
181
+ if (
182
+ code === 64 ||
183
+ (forced ??
184
+ (self.exprAllowed ||
185
+ afterPrefix ||
186
+ (this.lookahead && transientAssignment(self, end))))
187
+ ) {
188
+ if (code === 37) this.transientsRead?.add(at);
189
+ self.pos = end;
190
+ self.finishToken(tt.name, self.input.charAt(at) + name);
191
+ return;
192
+ }
193
+ }
171
194
  }
172
- i++;
195
+ // @ts-expect-error acorn internals
196
+ super.readToken(code);
173
197
  }
174
- return done(src.length, false);
175
- }
176
198
 
177
- /**
178
- * Index of the `]` closing the regex character class opening at `open`, or
179
- * of the line break (or the end of the source) that leaves it unterminated.
180
- * Every `[` in a class reads the rest of it the same way, so the result is
181
- * recorded for them too: scans of `/[/[/[…` from each `/` share one pass.
182
- */
183
- function scanRegexClass(
184
- src: string,
185
- open: number,
186
- cache?: JsScanCache,
187
- ): number {
188
- const known = cache?.classes.get(open);
189
- if (known !== undefined) return known;
190
- const opens = [open];
191
- let i = open + 1;
192
- while (i < src.length) {
193
- const lineEnd = regexLineEnd(src, i);
194
- if (lineEnd !== -1) {
195
- i = lineEnd;
196
- break;
199
+ finishToken(type: TokenType, value?: unknown): void {
200
+ const { context } = this as unknown as ParserState;
201
+ const depth = context.length;
202
+ if (type === tt.braceR) {
203
+ const closed = context[depth - 1] as { token: string; isExpr: boolean };
204
+ this.closedBlock = closed.token === '{' && !closed.isExpr;
205
+ } else if (type === tt.question) {
206
+ this.ternaries[depth] = (this.ternaries[depth] ?? 0) + 1;
207
+ } else if (type === tt.colon) {
208
+ const open = this.ternaries[depth] ?? 0;
209
+ this.colonEndsTernary = open > 0;
210
+ if (open > 0) this.ternaries[depth] = open - 1;
197
211
  }
198
- const c = src.charAt(i);
199
- if (c === '\\') {
200
- i += 2;
201
- continue;
212
+ // A keyword after `.` or `?.` is a property name (`a?.typeof`,
213
+ // `p.in⏎function f() {}`): read it as a name, so what follows it reads
214
+ // as after an operand, a function after it as a declaration
215
+ // and so is a `function` or `class` that no body follows: a class field
216
+ // or an object key (`class D { function; }`, `{ class: 1 }`)
217
+ const self = this as unknown as ParserState;
218
+ const prevType = self.type;
219
+ this.prevValue = self.value;
220
+ const property =
221
+ (type.keyword !== undefined &&
222
+ (prevType === tt.dot || prevType === tt.questionDot)) ||
223
+ ((type === tt._function || type === tt._class) &&
224
+ notABody(self.input, self.pos));
225
+ // @ts-expect-error acorn internals
226
+ super.finishToken(property ? tt.name : type, value);
227
+ // A variable named `of` (`const of of list`): acorn takes it for the
228
+ // `of` of a for-of loop, after which an operand comes
229
+ if (
230
+ type === tt.name &&
231
+ value === 'of' &&
232
+ (prevType === tt._const ||
233
+ prevType === tt._var ||
234
+ this.prevValue === 'let')
235
+ ) {
236
+ self.exprAllowed = false;
202
237
  }
203
- if (c === ']') break;
204
- if (c === '[' && cache) opens.push(i);
205
- i++;
206
238
  }
207
- i = Math.min(i, src.length);
208
- if (cache) for (const o of opens) cache.classes.set(o, i);
209
- return i;
210
- }
211
239
 
212
- /**
213
- * Index of the line break at `i`, escaped (`\` then a line break) or not,
214
- * where a regex literal or class ends unterminated; -1 if there is none.
215
- */
216
- function regexLineEnd(src: string, i: number): number {
217
- const c = src.charAt(i);
218
- if (LINE_TERMINATOR_RE.test(c)) return i;
219
- if (c === '\\' && LINE_TERMINATOR_RE.test(src.charAt(i + 1))) return i + 1;
220
- return -1;
221
- }
240
+ braceIsBlock(prevType: TokenType): boolean {
241
+ if (prevType === tt.colon && this.colonEndsTernary) return false;
242
+ if (prevType === tt.braceR && this.closedBlock) return true;
243
+ // A class's static initialization block: `static { … }`
244
+ if (prevType === tt.name && this.prevValue === 'static') return true;
245
+ // A statement ended by a line break (`p⏎{ }`): a block starts
246
+ if (this.breakHere && OPERAND_ENDS.has(prevType)) {
247
+ const { context } = this as unknown as ParserState;
248
+ const parent = context[context.length - 1] as {
249
+ token: string;
250
+ isExpr: boolean;
251
+ };
252
+ if (parent.token === '{' && !parent.isExpr) return true;
253
+ }
254
+ // @ts-expect-error acorn internals
255
+ return super.braceIsBlock(prevType);
256
+ }
257
+ };
222
258
 
223
259
  /**
224
- * The next index from `from` on where `find` matches (-1 for none), for
225
- * searches from increasing positions: a search from within the stretch the
226
- * last one covered has the same answer.
260
+ * Is the `%name` ending at `end`, at the start of a line after an operand,
261
+ * assigned to (`$x = 5⏎%a[i].b = 1`, not the `5 % a` of `$x = 5⏎%a`)?
262
+ * Its target may go on with `.name` and `[…]`.
227
263
  */
228
- function nextMatch(
229
- memo: NextMatch | undefined,
230
- from: number,
231
- find: (from: number) => number,
232
- ): number {
233
- if (memo && from >= memo.from && (memo.at < 0 || from <= memo.at)) {
234
- return memo.at;
235
- }
236
- const at = find(from);
237
- if (memo) {
238
- memo.from = from;
239
- memo.at = at;
264
+ function transientAssignment(p: ParserState, end: number): boolean {
265
+ if (!LINE_BREAK_RE.test(p.input.slice(p.end, p.pos))) return false;
266
+ const q = tokenizerAt(p.input, end, 'statements', { afterOperand: true });
267
+ (q as unknown as { lookahead: boolean }).lookahead = false;
268
+ let depth = 0;
269
+ try {
270
+ for (;;) {
271
+ q.nextToken();
272
+ const t = q.type;
273
+ if (t === tt.eof) return false;
274
+ if (depth > 0) {
275
+ if (OPENERS.has(t)) depth++;
276
+ else if (CLOSERS.has(t)) depth--;
277
+ } else if (t === tt.bracketL) {
278
+ depth = 1;
279
+ } else if (t === tt.dot) {
280
+ q.nextToken();
281
+ if (q.type !== tt.name && !q.type.keyword) return false;
282
+ } else {
283
+ return t === tt.eq || t === tt.assign;
284
+ }
285
+ }
286
+ } catch {
287
+ return false;
240
288
  }
241
- return at;
242
289
  }
243
290
 
244
- /** A search memo for `nextMatch`. */
245
- interface NextMatch {
246
- from: number;
247
- at: number;
248
- }
291
+ const OPENERS = new Set([tt.parenL, tt.bracketL, tt.braceL, tt.dollarBraceL]);
292
+ const CLOSERS = new Set([tt.parenR, tt.bracketR, tt.braceR]);
249
293
 
250
294
  /**
251
- * Index just past the comment opening at `start` (`//` or `/*`), or -1 for
252
- * an unterminated `/*` comment.
295
+ * A tokenizer (and parser) for `src` from `start` on. In an expression, the
296
+ * first token reads as after a `(`: a `{` opens an object literal, and a
297
+ * `function` or `class` is an expression, which an operator may follow.
298
+ * After an operand, a `/` divides.
253
299
  */
254
- function findCommentEnd(
300
+ function tokenizerAt(
255
301
  src: string,
256
302
  start: number,
257
- cache?: JsScanCache,
258
- ): number {
259
- if (src.charAt(start + 1) === '/') {
260
- const end = nextLineBreak(src, start, cache);
261
- return end < 0 ? src.length : end;
262
- }
263
- const end = nextMatch(cache?.commentClose, start + 2, (from) =>
264
- src.indexOf('*/', from),
265
- );
266
- return end < 0 ? -1 : end + 2;
303
+ goal: JsGoal,
304
+ { onComment, afterOperand = false }: TokenizerOptions = {},
305
+ ): ParserState {
306
+ const options = onComment ? { ...OPTIONS, onComment } : OPTIONS;
307
+ const p = new SigilParser(options, src, start) as unknown as ParserState;
308
+ if (goal === 'expression') p.type = tt.parenL;
309
+ if (afterOperand) p.exprAllowed = false;
310
+ return p;
267
311
  }
268
312
 
269
- /** Next line break from `from` on (-1 for none). */
270
- function nextLineBreak(src: string, from: number, cache?: JsScanCache): number {
271
- return nextMatch(cache?.lineEnd, from, (at) => {
272
- LINE_TERMINATOR_G.lastIndex = at;
273
- return LINE_TERMINATOR_G.exec(src)?.index ?? -1;
274
- });
275
- }
276
-
277
- /** Is there a line break between `from` and `to`? */
278
- function lineBreakIn(
279
- src: string,
280
- from: number,
281
- to: number,
282
- cache?: JsScanCache,
283
- ): boolean {
284
- if (!cache) return LINE_TERMINATOR_RE.test(src.slice(from, to));
285
- const at = nextLineBreak(src, from, cache);
286
- return at >= 0 && at < to;
313
+ interface TokenizerOptions {
314
+ onComment?: Options['onComment'];
315
+ afterOperand?: boolean;
287
316
  }
288
317
 
289
- /** Index just past the comment opening at `start` (`//` or `/*`). */
290
- function skipComment(src: string, start: number): number {
291
- const end = findCommentEnd(src, start);
292
- return end < 0 ? src.length : end;
293
- }
318
+ // ---------------------------------------------------------------------------
319
+ // Errors
320
+ // ---------------------------------------------------------------------------
294
321
 
295
- /** Index of the first character from `i` on that is no space or comment. */
296
- function skipTrivia(src: string, i: number): number {
297
- while (i < src.length) {
298
- const c = src.charAt(i);
299
- if (SPACE_RE.test(c)) i++;
300
- else if (c === '/' && '/*'.includes(src.charAt(i + 1)))
301
- i = skipComment(src, i);
302
- else break;
303
- }
304
- return i;
322
+ /** The bracket left open where a syntax error is. */
323
+ export interface OpenBracket {
324
+ /** `(`, `[`, `{` or `${`. */
325
+ open: string;
326
+ closer: string;
327
+ /** Its index in the source. */
328
+ pos: number;
305
329
  }
306
330
 
307
331
  /**
308
- * Lex the template literal opening at `start` (a backtick): its backticks,
309
- * text and escapes are reported as literal text and the code of its `${…}`
310
- * interpolations as `lexJs` does, one nesting level deeper. Returns the index
311
- * just past the closing backtick, or `src.length` if it is unterminated.
332
+ * A syntax error in story code: `reason` at index `pos` of `source`, and the
333
+ * bracket left open there when that is the likely cause. Its message reads
334
+ * `Unexpected end of code at column 16: ($gold + $count▶ (missing ")" for
335
+ * the "(" at column 1)`.
312
336
  */
313
- export function lexTemplate(
314
- src: string,
315
- start: number,
316
- handlers: JsLexHandlers = {},
317
- nesting = 0,
318
- ): number {
319
- handlers.literal?.('`', start, nesting);
320
- const outer = frame('template', '`', start);
321
- return scan(src, handlers, start + 1, nesting, outer, newContext());
322
- }
337
+ export class CodeSyntaxError extends SyntaxError {
338
+ constructor(
339
+ readonly reason: string,
340
+ readonly source: string,
341
+ readonly pos: number,
342
+ readonly bracket?: OpenBracket,
343
+ ) {
344
+ super(
345
+ `${reason} ${where(source, pos)}` +
346
+ bracketHint(bracket, (at) => columnOf(source, at)),
347
+ );
348
+ this.name = 'SyntaxError';
349
+ }
323
350
 
324
- /**
325
- * State shared by the scans of one source: the main scan and the look-ahead
326
- * scans that find where a bracketed assignment target ends.
327
- */
328
- interface ScanContext {
329
351
  /**
330
- * Look ahead after a `%name` starting a line for an assignment. Off in
331
- * look-ahead scans, which only need to match brackets — and whose own
332
- * look-ahead could rescan the same text over and over.
352
+ * What is wrong, for the code found at `offset` in `text` (a passage):
353
+ * `Unexpected "{" (missing ")" for the "(" at line 2, column 4)`. The
354
+ * error itself is at `offset + pos` there.
333
355
  */
334
- lookahead: boolean;
335
- /**
336
- * Index of the `]` matching the `[` at an index (`src.length` if there is
337
- * none), as found by look-ahead scans: each text is scanned ahead once.
338
- */
339
- brackets: Map<number, number>;
340
- /** Set for `findCodeEnd`: the scan stops at the first lexical error. */
341
- strict?: StrictScan;
356
+ reasonIn(text: string, offset: number): string {
357
+ return (
358
+ this.reason +
359
+ bracketHint(this.bracket, (pos) => {
360
+ const { line, column } = getLineInfo(text, offset + pos);
361
+ return `at line ${line}, column ${column + 1}`;
362
+ })
363
+ );
364
+ }
342
365
  }
343
366
 
344
- interface StrictScan {
345
- /** Stop at a `{` in code for which this holds. */
346
- stop?: (index: number) => boolean;
347
- /** The scan ran into a lexical error. */
348
- malformed: boolean;
349
- /** The scan stopped where `stop` held. */
350
- stopped: boolean;
351
- /** Results shared with other scans of the source. */
352
- cache?: JsScanCache;
353
- /** `cache.braces`, unless the scan has a `stop`. */
354
- braces?: Map<number, number>;
355
- /** `cache.checkpoints` for this kind of scan. */
356
- checkpoints?: Map<CheckpointKey, number>;
357
- /** `cache.parens` for this kind of scan. */
358
- parens?: Map<number, number>;
359
- /** Checkpoints this scan passed, to record its result at. */
360
- passed: CheckpointKey[];
367
+ function bracketHint(
368
+ b: OpenBracket | undefined,
369
+ at: (pos: number) => string,
370
+ ): string {
371
+ return b ? ` (missing "${b.closer}" for the "${b.open}" ${at(b.pos)})` : '';
361
372
  }
362
373
 
363
- /** Characters that a scan checkpoint follows (`checkpointKey`). */
364
- const CHECKPOINT_AFTER = new Set([
365
- '}',
366
- '"',
367
- "'",
368
- '`',
369
- '/',
370
- '\n',
371
- '\r',
372
- '\u2028',
373
- '\u2029',
374
- ' ',
375
- '\t',
376
- ]);
377
-
378
- /**
379
- * How far into a scan its results start to be shared within brackets
380
- * (`checkpointKey`, `knownParen`). Most scans end sooner, so they don't pay
381
- * for recording results no other scan will use; a long one pays this much
382
- * before it can use what earlier scans recorded, which keeps scans from
383
- * many starts about linear.
384
- */
385
- const SHARE_AFTER = 256;
386
-
387
- /**
388
- * A scan checkpoint: a position and the scan state there (`checkpointKey`).
389
- * A number at the top level, a string within brackets.
390
- */
391
- type CheckpointKey = number | string;
374
+ /** `at column 4`, or `at line 2, column 4` in code over several lines. */
375
+ function columnOf(src: string, pos: number): string {
376
+ const { line, column } = getLineInfo(src, pos);
377
+ return LINE_BREAK_RE.test(src)
378
+ ? `at line ${line}, column ${column + 1}`
379
+ : `at column ${column + 1}`;
380
+ }
392
381
 
393
- /** A frame result: it is still open at the end of the source. */
394
- const UNCLOSED = -1;
395
- /** A frame result: a lexical error inside it ends the scan. */
396
- const MALFORMED = -2;
397
- /** A frame result: the scan stops at index `s` inside it (`STOPPED - s`). */
398
- const STOPPED = -3;
382
+ /** `at column 9: $name = ▶"Bob`: the place, and its line marked there. */
383
+ function where(src: string, pos: number): string {
384
+ const { column } = getLineInfo(src, pos);
385
+ const lineStart = pos - column;
386
+ const after = src.slice(lineStart).search(LINE_BREAK_RE);
387
+ const lineEnd = after < 0 ? src.length : lineStart + after;
388
+ const from = Math.max(lineStart, pos - 30);
389
+ const to = Math.min(lineEnd, pos + 30);
390
+ const excerpt =
391
+ (from > lineStart ? '…' : '') +
392
+ src.slice(from, pos) +
393
+ '▶' +
394
+ src.slice(pos, to) +
395
+ (to < lineEnd ? '…' : '');
396
+ return `${columnOf(src, pos)}: ${excerpt.trim()}`;
397
+ }
399
398
 
400
- /**
401
- * The result of a frame that closes at `end`. `bodyStarted`: the code inside
402
- * started a function or class body. That body replaced the one to come, and
403
- * is gone once the frame is closed, so after the frame no body is to come —
404
- * a scan that skips the frame must know, as must the frames around it.
405
- */
406
- const frameEnd = (end: number, bodyStarted: boolean) => end * 2 + +bodyStarted;
399
+ /** The last element of `list`. */
400
+ const last = <T>(list: readonly T[]): T | undefined => list[list.length - 1];
407
401
 
408
- /** The closers whose search for their frame a frame result may depend on. */
409
- const CLOSERS = [')', ']', '}'] as const;
410
- const FRAME_KINDS: readonly Frame['kind'][] = [
411
- 'block',
412
- 'object',
413
- 'class',
414
- 'template',
415
- ];
402
+ const CLOSER: Record<string, string> = {
403
+ '(': ')',
404
+ '[': ']',
405
+ '{': '}',
406
+ '${': '}',
407
+ };
416
408
 
417
409
  /**
418
- * Key of a frame result, for the frames whose code lexes the same whatever
419
- * surrounds them: braces (a block, an object literal, a class body), a
420
- * template literal and its interpolations. Undefined for parentheses and
421
- * square brackets, which a stray closer inside may close.
410
+ * Turn an acorn error into a `CodeSyntaxError`: name the token it stopped
411
+ * at, and the bracket left open when that is the likely cause.
422
412
  */
423
- function frameKey(f: Frame): number | undefined {
424
- const kind = f.interpolation ? 4 : FRAME_KINDS.indexOf(f.kind);
425
- return kind < 0 ? undefined : f.open * 5 + kind;
413
+ function syntaxError(src: string, error: unknown): CodeSyntaxError {
414
+ if (!(error instanceof SyntaxError) || !('pos' in error)) throw error;
415
+ const pos = (error as SyntaxError & { pos: number }).pos;
416
+ let reason = error.message.replace(/ \(\d+:\d+\)$/, '');
417
+ const { open, token } = bracketsBefore(src, pos);
418
+ const atEnd = pos >= src.length || token === '';
419
+ let unclosed = false;
420
+ if (reason === 'Unexpected token') {
421
+ reason = atEnd ? 'Unexpected end of code' : `Unexpected "${token}"`;
422
+ unclosed =
423
+ !!open &&
424
+ CLOSER[open.text] !== token &&
425
+ (atEnd || [')', ']', '}', ';', '{'].includes(token));
426
+ } else if (reason.startsWith('Unterminated template')) {
427
+ // A backtick in an interpolation opens a template of its own: the `}`
428
+ // ending the interpolation is missing (`${$name`)
429
+ unclosed = open?.text === '${';
430
+ }
431
+ const bracket =
432
+ unclosed && open
433
+ ? { open: open.text, closer: CLOSER[open.text]!, pos: open.pos }
434
+ : undefined;
435
+ return new CodeSyntaxError(reason, src, pos, bracket);
426
436
  }
427
437
 
428
- /**
429
- * Results that `findCodeEnd` scans of one source share, so that scanning it
430
- * from many starts doesn't lex the same code over and over.
431
- */
432
- export interface JsScanCache {
433
- /**
434
- * Where each `{…}` frame, template literal or `${…}` interpolation closes,
435
- * as a frame result (`frameEnd`): its `}` or closing backtick, and whether
436
- * the code inside started a function or class body; or `UNCLOSED` or
437
- * `MALFORMED`. The code inside such a frame lexes the same whatever
438
- * surrounds it, given where it opens and its kind — a stray `)` or `]`
439
- * inside never closes it — so a later scan entering the same frame skips
440
- * to its end.
441
- */
442
- braces: Map<number, number>;
443
- /** Look-ahead bracket matches (`ScanContext.brackets`). */
444
- brackets: Map<number, number>;
445
- /** Regex character class ends (`scanRegexClass`). */
446
- classes: Map<number, number>;
447
- /** How a regex goes on from just past a class: `end * 2 + closed`. */
448
- regexes: Map<number, number>;
449
- /** The last search for a line break ending a `//` comment. */
450
- lineEnd: NextMatch;
451
- /** The last search for the end of a block comment. */
452
- commentClose: NextMatch;
453
- /**
454
- * How scans go on from points, by the kind of scan (its goal and how it
455
- * ends) and then by the point and the scan state there, the brackets open
456
- * around it included: the index the scan ends at, `UNCLOSED` or
457
- * `MALFORMED`. The points are where no word is being read (see
458
- * `checkpointKey`). Scans from different starts soon pass such points in
459
- * the same state, and from there on go the same way.
460
- */
461
- checkpoints: Map<string, Map<CheckpointKey, number>>;
462
- /**
463
- * Ids of the stacks of open brackets that checkpoints have seen (see
464
- * `scan`), by the id of the stack below the innermost bracket, the state
465
- * of that one and the innermost bracket.
466
- */
467
- stacks: Map<string, number>;
468
- /**
469
- * How `(…)` and `[…]` frames end, by the kind of scan and then by the frame
470
- * (`parenKey`): a frame result (`frameEnd`), or `UNCLOSED`, `MALFORMED` or
471
- * `STOPPED - s`. Unlike braces, a stray closer inside may close a frame
472
- * around them, so the code inside lexes the same only around frames for
473
- * which the closers it met find nothing to close; the key says which
474
- * closers met none.
475
- */
476
- parens: Map<string, Map<number, number>>;
477
- /**
478
- * How far into a scan its results start to be shared within brackets
479
- * (`SHARE_AFTER`; tests set 0 to share them all).
480
- */
481
- shareAfter: number;
438
+ /** The innermost bracket open at `pos`, and the token there. */
439
+ function bracketsBefore(
440
+ src: string,
441
+ pos: number,
442
+ ): { open?: { text: string; pos: number }; token: string } {
443
+ const stack: { text: string; pos: number }[] = [];
444
+ for (const tok of tokens(src, 0, 'statements')) {
445
+ if ('error' in tok) break;
446
+ if (tok.start >= pos || tok.type === tt.eof) {
447
+ return { open: last(stack), token: src.slice(tok.start, tok.end) };
448
+ }
449
+ const text = src.slice(tok.start, tok.end);
450
+ if (text in CLOSER) stack.push({ text, pos: tok.start });
451
+ else if (stack.length && CLOSER[last(stack)!.text] === text) stack.pop();
452
+ }
453
+ return { open: last(stack), token: src.charAt(pos) };
482
454
  }
483
455
 
484
- export function createJsScanCache(): JsScanCache {
485
- return {
486
- checkpoints: new Map(),
487
- stacks: new Map(),
488
- parens: new Map(),
489
- shareAfter: SHARE_AFTER,
490
- braces: new Map(),
491
- brackets: new Map(),
492
- classes: new Map(),
493
- regexes: new Map(),
494
- lineEnd: { from: Infinity, at: -1 },
495
- commentClose: { from: Infinity, at: -1 },
496
- };
456
+ // ---------------------------------------------------------------------------
457
+ // Parsing: references, string text, errors
458
+ // ---------------------------------------------------------------------------
459
+
460
+ export interface VariableRef {
461
+ sigil: Sigil;
462
+ /** The name after the sigil. */
463
+ name: string;
464
+ /** Range of the sigil and name in the source. */
465
+ start: number;
466
+ end: number;
467
+ /** A shorthand property (`{ $gold }`): its key is to be written out. */
468
+ shorthand: boolean;
497
469
  }
498
470
 
499
- const newContext = (): ScanContext => ({
500
- lookahead: true,
501
- brackets: new Map(),
502
- });
503
-
504
- /** Index of the `]` matching the `[` at `open`, or `src.length`. */
505
- function matchBracket(src: string, open: number, ctx: ScanContext): number {
506
- let close = ctx.brackets.get(open);
507
- if (close === undefined) {
508
- const ahead = { lookahead: false, brackets: ctx.brackets };
509
- close = scan(src, {}, open + 1, 0, frame('expr', ']', open), ahead);
510
- ctx.brackets.set(open, close);
511
- }
512
- return close;
471
+ export interface ParsedCode {
472
+ /** The variable references, in source order. */
473
+ refs: VariableRef[];
474
+ /** The raw text of string literals and template literal pieces. */
475
+ strings: string[];
513
476
  }
514
477
 
515
478
  /**
516
- * Is the code at `i`, just past a `%name`, the rest of an assignment target
517
- * and its operator: ` = 1`, `.a[b] += 2`, but not `== 1`, `=> 1` or `% 2`?
479
+ * Parse `src` as `goal`, and find its variable references and string text.
480
+ * Throws a `CodeSyntaxError` for code that is not well-formed, and for a
481
+ * sigil variable declared (`let _x`) or used as a property name (`a.@x`).
518
482
  */
519
- function assignmentFollows(src: string, i: number, ctx: ScanContext): boolean {
483
+ export function parseCode(
484
+ src: string,
485
+ goal: JsGoal = 'expression',
486
+ ): ParsedCode {
487
+ // Whether `%name` is a transient or `%` modulo is guessed from the tokens
488
+ // before it, as acorn guesses whether `/` opens a regex. Where the guess
489
+ // makes the parser fail at a `%`, the other reading is tried, and the
490
+ // code parsed again (`for (const of of %list)`, `a⏎of % p`).
491
+ const percentAt = new Map<number, boolean>();
520
492
  for (;;) {
521
- i = skipTrivia(src, i);
522
- const c = src.charAt(i);
523
- if (c === '.' && src.charAt(i + 1) !== '.') {
524
- IDENT_RE.lastIndex = skipTrivia(src, i + 1);
525
- const name = IDENT_RE.exec(src)?.[0];
526
- if (!name) return false;
527
- i = IDENT_RE.lastIndex;
528
- } else if (c === '[') {
529
- i = matchBracket(src, i, ctx);
530
- if (i >= src.length) return false;
531
- i++;
532
- } else {
533
- break;
493
+ const transientsRead = new Set<number>();
494
+ try {
495
+ const ast = parseAst(src, goal, percentAt, transientsRead);
496
+ const out: ParsedCode = { refs: [], strings: [] };
497
+ walk(ast, src, out, false, false);
498
+ out.refs.sort((a, b) => a.start - b.start);
499
+ return out;
500
+ } catch (error) {
501
+ const pos = (error as { pos?: number }).pos ?? -1;
502
+ if (!(error instanceof SyntaxError) || percentAt.has(pos)) throw error;
503
+ TRANS_NAME_RE.lastIndex = pos + 1;
504
+ if (src.charAt(pos) !== '%' || !TRANS_NAME_RE.test(src)) throw error;
505
+ percentAt.set(pos, !transientsRead.has(pos));
534
506
  }
535
507
  }
536
- ASSIGN_OP_RE.lastIndex = i;
537
- return ASSIGN_OP_RE.test(src);
538
508
  }
539
509
 
540
510
  /**
541
- * The code between a pair of brackets, or the whole source.
542
- *
543
- * - `block`: statements (a block, a function body, the whole source as
544
- * statements),
545
- * - `object`: the property definitions of an object literal,
546
- * - `class`: the member definitions of a class body,
547
- * - `expr`: an expression (parentheses, square brackets, a template literal
548
- * interpolation, the whole source as an expression),
549
- * - `template`: the text of a template literal.
511
+ * Parse `src` as `goal`, reading `%` as `percentAt` says where it says, and
512
+ * noting where it read a transient in `transientsRead`.
550
513
  */
551
- interface Frame {
552
- kind: 'block' | 'object' | 'class' | 'expr' | 'template';
553
- /** The character closing it, or '' for the whole source. */
554
- closer: string;
555
- /** Index of its opening bracket. */
556
- open: number;
557
- /** A parenthesised `if`/`for`/`while`/`with` header: a statement follows. */
558
- header: boolean;
559
- /** Its closing `}` ends an operand (object literal, function expression). */
560
- operand: boolean;
561
- /** Conditional-expression `?`s awaiting their `:`. */
562
- ternary: number;
563
- /** A template literal's `${…}` interpolation. */
564
- interpolation: boolean;
514
+ function parseAst(
515
+ src: string,
516
+ goal: JsGoal,
517
+ percentAt: ReadonlyMap<number, boolean>,
518
+ transientsRead: Set<number>,
519
+ ): AnyNode {
520
+ try {
521
+ const p = tokenizerAt(src, 0, goal) as ParserState & {
522
+ parse(): AnyNode;
523
+ percentAt?: ReadonlyMap<number, boolean>;
524
+ transientsRead?: Set<number>;
525
+ };
526
+ p.percentAt = percentAt;
527
+ p.transientsRead = transientsRead;
528
+ if (goal === 'statements') return p.parse();
529
+ // As acorn's parseExpressionAt, and then the code must end
530
+ p.nextToken();
531
+ const ast = p.parseExpression();
532
+ if (p.type !== tt.eof) {
533
+ throw Object.assign(new SyntaxError('Unexpected token'), {
534
+ pos: p.start,
535
+ });
536
+ }
537
+ return ast;
538
+ } catch (error) {
539
+ throw syntaxError(src, error);
540
+ }
565
541
  }
566
542
 
567
- function frame(kind: Frame['kind'], closer: string, open = -1): Frame {
568
- return {
569
- kind,
570
- closer,
571
- open,
572
- header: false,
573
- operand: false,
574
- ternary: 0,
575
- interpolation: false,
576
- };
543
+ interface AnyNode {
544
+ type: string;
545
+ start: number;
546
+ end: number;
547
+ [key: string]: unknown;
548
+ }
549
+
550
+ const isNode = (v: unknown): v is AnyNode =>
551
+ typeof v === 'object' &&
552
+ v !== null &&
553
+ typeof (v as AnyNode).type === 'string';
554
+
555
+ const NAMESPACE_NAME: Record<Sigil, string> = {
556
+ $: 'story',
557
+ _: 'temporary',
558
+ '@': 'local',
559
+ '%': 'transient',
560
+ };
561
+
562
+ /** The sigil and name of a sigil identifier, as written. */
563
+ function sigilOf(node: AnyNode, src: string): [Sigil, string] | null {
564
+ const name = node.name as string;
565
+ const c = src.charAt(node.start);
566
+ if (c !== name.charAt(0)) return null; // written with an escape
567
+ if (c === '@' || c === '%') return [c, name.slice(1)];
568
+ if ((c === '$' || c === '_') && SIGIL_IDENT_RE.test(name)) {
569
+ return [c, name.slice(1)];
570
+ }
571
+ return null;
572
+ }
573
+
574
+ /** A property or member name: never a reference, and never `@x`/`%x`. */
575
+ function propertyName(node: unknown, src: string): void {
576
+ if (!isNode(node) || node.type !== 'Identifier') return;
577
+ const c = src.charAt(node.start);
578
+ if (c === '@' || c === '%') {
579
+ throw new CodeSyntaxError(
580
+ `"${node.name as string}" can't be a property name`,
581
+ src,
582
+ node.start,
583
+ );
584
+ }
577
585
  }
578
586
 
579
587
  /**
580
- * Walk `src`, reporting code characters, literal text and variable
581
- * references to `handlers` in source order. Every character is reported
582
- * exactly once (a variable reference covers its sigil and name). Returns
583
- * `src.length`.
588
+ * Collect the references and string text under `node`. `binding`: the
589
+ * identifiers here are declared (a `let`, a parameter), where a sigil
590
+ * variable can't be. `shorthand`: the node is a shorthand property's value.
584
591
  */
585
- export function lexJs(
592
+ function walk(
593
+ node: unknown,
586
594
  src: string,
587
- handlers: JsLexHandlers,
588
- goal: JsGoal = 'expression',
589
- ): number {
590
- const outer = frame(goal === 'statements' ? 'block' : 'expr', '');
591
- return scan(src, handlers, 0, 0, outer, newContext());
595
+ out: ParsedCode,
596
+ binding: boolean,
597
+ shorthand: boolean,
598
+ ): void {
599
+ if (Array.isArray(node)) {
600
+ for (const n of node) walk(n, src, out, binding, false);
601
+ return;
602
+ }
603
+ if (!isNode(node)) return;
604
+ const sub = (child: unknown, bind = false, short = false) =>
605
+ walk(child, src, out, bind, short);
606
+ switch (node.type) {
607
+ case 'Identifier': {
608
+ const sigil = sigilOf(node, src);
609
+ if (!sigil) return;
610
+ if (binding) {
611
+ throw new CodeSyntaxError(
612
+ `"${node.name as string}" is a ${NAMESPACE_NAME[sigil[0]]} variable and can't be declared`,
613
+ src,
614
+ node.start,
615
+ );
616
+ }
617
+ out.refs.push({
618
+ sigil: sigil[0],
619
+ name: sigil[1],
620
+ start: node.start,
621
+ end: node.end,
622
+ shorthand,
623
+ });
624
+ return;
625
+ }
626
+ case 'Literal':
627
+ if (typeof node.value === 'string') {
628
+ out.strings.push(src.slice(node.start + 1, node.end - 1));
629
+ }
630
+ return;
631
+ case 'TemplateElement':
632
+ out.strings.push((node.value as { raw: string }).raw);
633
+ return;
634
+ case 'MemberExpression':
635
+ sub(node.object);
636
+ if (node.computed) sub(node.property);
637
+ else propertyName(node.property, src);
638
+ return;
639
+ case 'Property':
640
+ if (node.computed) sub(node.key);
641
+ else if (!node.shorthand) propertyName(node.key, src);
642
+ if (node.shorthand) {
643
+ const value = node.value as AnyNode;
644
+ if (value.type === 'AssignmentPattern') {
645
+ sub(value.left, binding, true);
646
+ sub(value.right);
647
+ } else {
648
+ sub(value, binding, true);
649
+ }
650
+ } else {
651
+ sub(node.value, binding);
652
+ }
653
+ return;
654
+ case 'MethodDefinition':
655
+ case 'PropertyDefinition':
656
+ if (node.computed) sub(node.key);
657
+ else propertyName(node.key, src);
658
+ sub(node.value);
659
+ return;
660
+ case 'LabeledStatement':
661
+ sub(node.body);
662
+ return;
663
+ case 'BreakStatement':
664
+ case 'ContinueStatement':
665
+ case 'MetaProperty':
666
+ case 'PrivateIdentifier':
667
+ return;
668
+ case 'VariableDeclarator':
669
+ sub(node.id, true);
670
+ sub(node.init);
671
+ return;
672
+ case 'FunctionDeclaration':
673
+ case 'FunctionExpression':
674
+ case 'ArrowFunctionExpression':
675
+ sub(node.id, true);
676
+ for (const p of node.params as unknown[]) sub(p, true);
677
+ sub(node.body);
678
+ return;
679
+ case 'ClassDeclaration':
680
+ case 'ClassExpression':
681
+ sub(node.id, true);
682
+ sub(node.superClass);
683
+ sub(node.body);
684
+ return;
685
+ case 'CatchClause':
686
+ sub(node.param, true);
687
+ sub(node.body);
688
+ return;
689
+ case 'AssignmentPattern':
690
+ sub(node.left, binding);
691
+ sub(node.right);
692
+ return;
693
+ case 'ObjectPattern':
694
+ case 'ArrayPattern':
695
+ case 'RestElement':
696
+ for (const key of ['properties', 'elements', 'argument']) {
697
+ sub(node[key], binding);
698
+ }
699
+ return;
700
+ default:
701
+ for (const key in node) {
702
+ if (key === 'type' || key === 'start' || key === 'end') continue;
703
+ const v = node[key];
704
+ if (typeof v === 'object' && v !== null) sub(v);
705
+ }
706
+ }
592
707
  }
593
708
 
709
+ // ---------------------------------------------------------------------------
710
+ // Where code in markup ends
711
+ // ---------------------------------------------------------------------------
712
+
594
713
  export interface FindCodeEndOptions {
595
714
  /** What the code is (default `expression`). */
596
715
  goal?: JsGoal;
597
716
  /** End the code at a `{` in code, at any depth, for which this holds. */
598
717
  stop?: (index: number) => boolean;
599
- /**
600
- * Names `stop` for `cache`: scans with the same key share results (scans
601
- * with a `stop` but no key don't use `cache.checkpoints`).
602
- */
603
- stopKey?: string;
604
- /**
605
- * Results to share with other scans of the same source. Scanning a source
606
- * from n starts then takes about linear time instead of n scans of the
607
- * rest of it.
608
- */
609
- cache?: JsScanCache;
610
718
  }
611
719
 
720
+ /** The rest of a word, after a number (`2s`). */
721
+ const WORD_RE = /[\p{ID_Continue}$.]*/uy;
722
+ /** The flags after a regex literal, whatever they are. */
723
+ const FLAGS_RE = /[\p{ID_Continue}$]*/uy;
724
+
725
+ /** Words a string may follow with no space between (`of'x'`, `get"y"`). */
726
+ const WORDS_BEFORE_STRING = new Set(['of', 'get', 'set', 'static', 'async']);
727
+
612
728
  /**
613
- * Find where the code starting at `start` ends, lexing it as JavaScript:
614
- * braces, quotes and backticks inside string, template and regex literals and
615
- * comments don't count.
729
+ * Find where the code starting at `start` ends, reading it as JavaScript:
730
+ * braces, quotes and backticks inside literals and comments don't count.
616
731
  *
617
732
  * Without `stop`, the code ends at the first `}` in code outside the
618
733
  * brackets it opened (the `}` closing a `{…}` around it); with `stop`, at
619
734
  * the first `{` in code, at any depth, for which `stop` holds. Returns the
620
735
  * index of that `}` or `{`.
621
736
  *
622
- * Returns -1 when there is no such end, or when the code before it is not
623
- * well-formed JavaScript as far as a lexer can tell: an unterminated string,
624
- * regex literal or block comment, or a quote directly after an identifier or
625
- * number (`don't`, but not `typeof'x'`). Callers fall back to a more lenient
737
+ * A backtick inside a `${…}` that the code never closes is read as closing
738
+ * the template literal around it, once (`{set $s = \`Hi ${$name\`}` ends
739
+ * at its last `}`), so the error is reported in the block the author wrote.
740
+ *
741
+ * Returns -1 when there is no such end, or when the code before it can't be
742
+ * JavaScript: an unterminated string, regex literal or comment, or a quote
743
+ * directly after a word (`don't`). Callers fall back to a more lenient
626
744
  * reading there, so text that only looks like code is not swallowed by an
627
- * apostrophe or a stray quote.
745
+ * apostrophe or a stray quote. Characters no JavaScript has (a lone `@`,
746
+ * `→`) are skipped, a number with a unit (`2s`) is one word, and a quoted
747
+ * string may span lines, as quoted macro labels may: the parse reports
748
+ * what in it is no JavaScript.
628
749
  */
629
750
  export function findCodeEnd(
630
751
  src: string,
631
752
  start: number,
632
- { goal = 'expression', stop, stopKey, cache }: FindCodeEndOptions = {},
753
+ { goal = 'expression', stop }: FindCodeEndOptions = {},
633
754
  ): number {
634
- const strict: StrictScan = {
635
- stop,
636
- malformed: false,
637
- stopped: false,
638
- passed: [],
639
- };
640
- strict.cache = cache;
641
- // Where a `{…}` frame ends depends on `stop`
642
- if (cache && !stop) strict.braces = cache.braces;
643
- if (cache && (!stop || stopKey !== undefined)) {
644
- const kindKey = `${goal} ${stop ? `stop ${stopKey}` : '}'}`;
645
- let checkpoints = cache.checkpoints.get(kindKey);
646
- if (!checkpoints) cache.checkpoints.set(kindKey, (checkpoints = new Map()));
647
- strict.checkpoints = checkpoints;
648
- let parens = cache.parens.get(kindKey);
649
- if (!parens) cache.parens.set(kindKey, (parens = new Map()));
650
- strict.parens = parens;
651
- }
652
- const ctx: ScanContext = {
653
- lookahead: true,
654
- brackets: cache?.brackets ?? new Map(),
655
- strict,
656
- };
657
- const kind = goal === 'statements' ? 'block' : 'expr';
658
- const outer = frame(kind, stop ? '' : '}', start - 1);
659
- const end = scan(src, {}, start, 0, outer, ctx);
660
- if (strict.checkpoints) {
661
- const ended = stop ? strict.stopped : end < src.length;
662
- const result = strict.malformed ? MALFORMED : ended ? end : UNCLOSED;
663
- for (const at of strict.passed) strict.checkpoints.set(at, result);
664
- }
665
- if (strict.malformed) return -1;
666
- if (stop) return strict.stopped ? end : -1;
667
- return end < src.length ? end : -1;
755
+ const end = scanCodeEnd(src, start, goal, stop);
756
+ return end < 0 && !stop ? parsedCodeEnd(src, start, goal) : end;
668
757
  }
669
758
 
670
759
  /**
671
- * Scan from `start` within `outer`. Returns `src.length`, or the index of
672
- * the `]` closing an `outer` square bracket, or the index just past the
673
- * backtick closing an `outer` template literal.
674
- *
675
- * Nested brackets, template literals and their interpolations are frames
676
- * on a stack, not recursive calls: however deep the nesting, the scan
677
- * cannot overflow the call stack.
760
+ * Where the code from `start` ends, as the parser finds it: the `}` it
761
+ * stops at, or -1. The token scan of `findCodeEnd` guesses whether a `/`
762
+ * opens a regex from the tokens before it, and a few rare constructs (a
763
+ * class field named `class` inside a template literal) mislead it; the
764
+ * parser knows. Only asked when the scan finds no end.
678
765
  */
679
- function scan(
680
- src: string,
681
- handlers: JsLexHandlers,
682
- start: number,
683
- nesting: number,
684
- outer: Frame,
685
- ctx: ScanContext,
686
- ): number {
687
- const frames: Frame[] = [outer];
688
- const top = () => frames[frames.length - 1]!;
689
- /**
690
- * Indices of the open frames closed by `)`, `]` and `}` (above `outer`),
691
- * innermost last: the frame a closer closes is found without walking the
692
- * stack.
693
- */
694
- const closers: Record<string, number[]> = { ')': [], ']': [], '}': [] };
695
- const innermost = (c: string) => {
696
- const list = closers[c]!;
697
- return list.length ? list[list.length - 1]! : 0;
766
+ function parsedCodeEnd(src: string, start: number, goal: JsGoal): number {
767
+ const p = tokenizerAt(src, start, goal) as ParserState & {
768
+ parse(): unknown;
698
769
  };
699
- let i = start;
700
-
701
- let operandNext = true; // an operand (not an operator) comes next
702
- let stmtNext = outer.kind === 'block'; // a statement starts here
703
- let arrowBody = false; // an arrow function body starts here
704
- let keyNext = false; // a property name may come next
705
- let word = ''; // identifier/number currently being read
706
- let afterDot = false; // the next word is a property name, never a keyword
707
- let dots = 0; // length of the current run of `.` tokens
708
- let afterHeaderKeyword = false; // last token was if/while/for/with
709
- let lastPunct = '';
710
- let lastKeyword = ''; // the last token, if it was a keyword
711
- let lineBreak = false; // a line break since the last token
712
- /** The next `{` at this depth opens a function or class body. */
713
- let pendingBody: { depth: number; frame: Frame } | undefined;
714
-
715
- const strict = ctx.strict;
716
- const braces = strict?.braces;
717
- const stacks = strict?.checkpoints && strict.cache?.stacks;
718
- /**
719
- * With checkpoints, per open frame the id of the stack up to it, found
720
- * when a checkpoint needs it: the kind and flags of each frame, and the
721
- * conditional-expression count of each but the innermost (which changes
722
- * only while it is innermost, and is part of the checkpoint state). How a
723
- * scan goes on depends on these, not on where the frames opened.
724
- */
725
- const stackIds: (number | undefined)[] | undefined = stacks ? [0] : undefined;
726
-
727
- /** The id of the stack of open frames (`stackIds`). */
728
- function stackId(): number {
729
- let k = frames.length - 1;
730
- while (stackIds![k] === undefined) k--;
731
- for (k++; k < frames.length; k++) {
732
- const f = frames[k]!;
733
- const key =
734
- `${stackIds![k - 1]} ${frames[k - 1]!.ternary} ${f.kind} ${f.closer}` +
735
- ` ${+f.header}${+f.operand}${+f.interpolation}`;
736
- let id = stacks!.get(key);
737
- if (id === undefined) stacks!.set(key, (id = stacks!.size + 1));
738
- stackIds![k] = id;
770
+ try {
771
+ if (goal === 'statements') {
772
+ p.parse();
773
+ } else {
774
+ p.nextToken();
775
+ p.parseExpression();
776
+ if (p.type === tt.braceR) return p.start;
739
777
  }
740
- return stackIds![frames.length - 1]!;
778
+ } catch (error) {
779
+ const pos = (error as { pos?: number }).pos ?? -1;
780
+ if (src.charAt(pos) === '}') return pos;
741
781
  }
742
- /** A `{…}` frame result found in `braces`, to skip to. */
743
- let skipTo: { frame: Frame; end: number } | undefined;
744
-
745
- const parens = strict?.parens;
746
- const shareAfter = strict?.cache?.shareAfter ?? SHARE_AFTER;
747
- /**
748
- * With `parens`, per open frame and per closer in `CLOSERS`: the lowest
749
- * frame index the closer looked for its frame at, while this frame or one
750
- * opened in it was innermost. Below a frame's own index, the code in it
751
- * depended on what is around it. A frame's entry takes in those of the
752
- * frames opened in it as they close.
753
- */
754
- const lowest: number[][] | undefined = parens
755
- ? [[Infinity, Infinity, Infinity]]
756
- : undefined;
757
- /**
758
- * How many function or class bodies to come the scan has started, and with
759
- * `braces` or `parens`, that count when each open frame opened. Code in a
760
- * frame that starts one replaces the one to come, and the new one can't
761
- * follow once the frame is closed: none is to come then. A scan skipping
762
- * the frame learns this from its result (`frameEnd`, `resume`), and counts
763
- * the body as started too, for the frames around it to record.
764
- */
765
- let bodyStarts = 0;
766
- const startsAt: number[] | undefined = braces || parens ? [0] : undefined;
767
- /** A `(…)` or `[…]` frame result found in `parens`, to skip to. */
768
- let parenTo: { frame: Frame; result: number } | undefined;
769
-
770
- /** The result of the frame at index `k`, closing at `i`. */
771
- const closedAt = (k: number) =>
772
- frameEnd(i, startsAt !== undefined && bodyStarts !== startsAt[k]);
782
+ return -1;
783
+ }
773
784
 
774
- /**
775
- * Go on after a frame an earlier scan lexed, whose result (`frameEnd`) is
776
- * `result`: returns the index it closes at. A body started in it is gone
777
- * after it, with the one it replaced.
778
- */
779
- function resume(result: number): number {
780
- if (result % 2 === 1) {
781
- pendingBody = undefined;
782
- bodyStarts++;
785
+ /** `findCodeEnd` by the tokens: see there. */
786
+ function scanCodeEnd(
787
+ src: string,
788
+ start: number,
789
+ goal: JsGoal,
790
+ stop: ((index: number) => boolean) | undefined,
791
+ ): number {
792
+ const stack: string[] = [];
793
+ let prevEnd = -1;
794
+ let prevWord = false;
795
+ let from = start;
796
+ let afterOperand = false;
797
+ let templateSlip = false;
798
+ for (;;) {
799
+ let resumeAt = -1;
800
+ for (const tok of tokens(src, from, goal, { afterOperand })) {
801
+ if ('error' in tok) {
802
+ const { message } = tok.error;
803
+ // In a template literal, a fresh tokenizer would lose its place
804
+ const inTemplate = stack.includes('${');
805
+ if (!/^Unterminated/.test(message) && !inTemplate) {
806
+ // Something no JavaScript has (a lone `@`, `#`, a bad regex flag):
807
+ // on after its first character, as after an operator
808
+ resumeAt = Math.max(tok.pos, from) + 1;
809
+ afterOperand = false;
810
+ if (resumeAt >= src.length) return -1;
811
+ break;
812
+ }
813
+ const k = stack.lastIndexOf('${');
814
+ const unclosed = /^Unterminated template/.test(message);
815
+ // One such slip per block: more is no code an author meant. (A raw
816
+ // body that is no code ends at its first closer.)
817
+ if (!unclosed || k < 0 || templateSlip || stop) return -1;
818
+ // The template's text starts just past the backtick that opened it
819
+ templateSlip = true;
820
+ stack.length = k;
821
+ resumeAt = tok.pos;
822
+ afterOperand = true;
823
+ break;
824
+ }
825
+ if (tok.restarted && stack.includes('${')) return -1;
826
+ const t = tok.type;
827
+ if (t === tt.eof) return -1;
828
+ if (t === tt.string && prevWord && tok.start === prevEnd) return -1;
829
+ prevWord =
830
+ (t === tt.name && !WORDS_BEFORE_STRING.has(tok.value as string)) ||
831
+ t === tt.num;
832
+ prevEnd = tok.end;
833
+ if (t === tt.braceL) {
834
+ if (stop?.(tok.start)) return tok.start;
835
+ stack.push('{');
836
+ } else if (t === tt.dollarBraceL) stack.push('${');
837
+ else if (t === tt.parenL) stack.push('(');
838
+ else if (t === tt.bracketL) stack.push('[');
839
+ else if (t === tt.parenR || t === tt.bracketR) {
840
+ if (last(stack) === (t === tt.parenR ? '(' : '[')) stack.pop();
841
+ } else if (t === tt.braceR) {
842
+ const k = Math.max(stack.lastIndexOf('{'), stack.lastIndexOf('${'));
843
+ if (k >= 0) stack.length = k;
844
+ else if (!stop) return tok.start;
845
+ }
783
846
  }
784
- return Math.floor(result / 2);
785
- }
786
-
787
- /** The closer `c` looked for its frame and found index `k`. */
788
- function lookedFor(c: (typeof CLOSERS)[number], k: number) {
789
- if (!lowest) return;
790
- const entry = lowest[lowest.length - 1]!;
791
- const n = CLOSERS.indexOf(c);
792
- if (k < entry[n]!) entry[n] = k;
793
- }
794
-
795
- /** Take the entry of the frame at index `k` into the one below it. */
796
- function foldLowest(k: number) {
797
- const from = lowest![k]!;
798
- const into = lowest![k - 1]!;
799
- for (let n = 0; n < 3; n++) if (from[n]! < into[n]!) into[n] = from[n]!;
847
+ from = resumeAt;
848
+ prevWord = false;
800
849
  }
850
+ }
801
851
 
802
- /** Key of a `(…)` or `[…]` frame result, but for the closers it met. */
803
- const parenKey = (f: Frame) =>
804
- (f.open * 3 + (f.closer === ']' ? 2 : +f.header)) * 8;
805
-
852
+ interface Tok {
853
+ type: TokenType;
854
+ start: number;
855
+ end: number;
856
+ value: unknown;
806
857
  /**
807
- * Record how the frame at index `k` ends, if it is a `(…)` or `[…]`, its
808
- * `lowest` entry complete: under the closers that looked below it, which
809
- * found nothing to close (else the frame was closed with them).
858
+ * Read past an error, by a fresh tokenizer after it: one that knows
859
+ * nothing of the template literals around it.
810
860
  */
811
- function recordParen(k: number, result: number) {
812
- const f = frames[k]!;
813
- if (k === 0 || (f.closer !== ')' && f.closer !== ']')) return;
814
- if (f.open - start < shareAfter) return;
815
- let met = 0;
816
- for (let n = 0; n < 3; n++) if (lowest![k]![n]! < k) met |= 1 << n;
817
- parens!.set(parenKey(f) + met, result);
818
- }
819
-
820
- /** Record how the open `(…)` and `[…]` frames end: the scan ends in them. */
821
- function recordOpenParens(result: number) {
822
- if (!lowest) return;
823
- for (let k = frames.length - 1; k > 0; k--) {
824
- recordParen(k, result);
825
- foldLowest(k);
826
- }
827
- }
861
+ restarted?: boolean;
862
+ }
828
863
 
829
- /**
830
- * How an earlier scan found a `(…)` or `[…]` frame opening here to end,
831
- * if it did: a result recorded under closers that find nothing to close
832
- * around the frame here too. The frame is not open yet.
833
- */
834
- function knownParen(f: Frame): number | undefined {
835
- if (!parens || f.open - start < shareAfter) return undefined;
836
- // Where each closer would look for its frame, and whether it would find
837
- // none to close (it is stray)
838
- const at = [
839
- Math.max(innermost(')'), innermost('}')),
840
- Math.max(innermost(']'), innermost('}')),
841
- innermost('}'),
842
- ];
843
- let stray = 0;
844
- for (let n = 0; n < 3; n++) {
845
- const k = at[n]!;
846
- if (k === 0 || (n < 2 && frames[k]!.closer !== CLOSERS[n])) {
847
- stray |= 1 << n;
848
- }
849
- }
850
- const key = parenKey(f);
851
- for (let met = stray; ; met = (met - 1) & stray) {
852
- const result = parens.get(key + met);
853
- if (result !== undefined) {
854
- // The closers it met look here too
855
- for (let n = 0; n < 3; n++) {
856
- if (met & (1 << n)) lookedFor(CLOSERS[n]!, at[n]!);
864
+ /**
865
+ * acorn's tokens from `start` on, up to and including `eof`, or up to an
866
+ * error it raises (`pos` is where). A quoted string with a line break in it
867
+ * is one string token: quoted macro labels may span lines, though no
868
+ * JavaScript string does.
869
+ */
870
+ function* tokens(
871
+ src: string,
872
+ start: number,
873
+ goal: JsGoal,
874
+ options: TokenizerOptions = {},
875
+ ): Generator<Tok | { error: Error; pos: number }> {
876
+ let p = tokenizerAt(src, start, goal, options);
877
+ for (;;) {
878
+ try {
879
+ p.nextToken();
880
+ } catch (error) {
881
+ const pos = Math.min(
882
+ (error as { pos?: number }).pos ?? p.pos,
883
+ src.length,
884
+ );
885
+ const message = (error as Error).message;
886
+ /** Go on after `[from, end)`, read as an operand of `type`. */
887
+ const operand = (type: TokenType, from: number, end: number) => {
888
+ p = tokenizerAt(src, end, 'statements', {
889
+ onComment: options.onComment,
890
+ afterOperand: true,
891
+ });
892
+ return { type, start: from, end, value: undefined, restarted: true };
893
+ };
894
+ const first = src.charAt(p.start);
895
+ const unterminated = /^Unterminated/.test(message);
896
+ if (/^Unterminated string/.test(message)) {
897
+ const { end, closed } = scanStringLiteral(src, pos);
898
+ if (closed) {
899
+ yield operand(tt.string, pos, end);
900
+ continue;
901
+ }
902
+ } else if (
903
+ !unterminated &&
904
+ p.start <= pos &&
905
+ (first === '"' || first === "'")
906
+ ) {
907
+ // A string with an escape no JavaScript has (`"\x"`): one string
908
+ const { end, closed } = scanStringLiteral(src, p.start);
909
+ if (closed) {
910
+ yield operand(tt.string, p.start, end);
911
+ continue;
857
912
  }
858
- return result;
913
+ } else if (!unterminated && p.start <= pos && first === '/') {
914
+ // A regex with flags no JavaScript has (`/a/gb`): one regex
915
+ const end = regexEnd(src, p.start);
916
+ if (end > 0) {
917
+ yield operand(tt.regexp, p.start, end);
918
+ continue;
919
+ }
920
+ } else if (
921
+ !unterminated &&
922
+ p.start <= pos &&
923
+ /[\d.]/.test(first) &&
924
+ /\d/.test(src.slice(p.start, p.start + 2))
925
+ ) {
926
+ // A number with a unit (`2s`, `5ms`), as macro arguments have them,
927
+ // or some other word starting with a digit (`0_$`): one word
928
+ WORD_RE.lastIndex = p.start;
929
+ WORD_RE.test(src);
930
+ yield operand(tt.num, p.start, WORD_RE.lastIndex);
931
+ continue;
859
932
  }
860
- if (met === 0) return undefined;
933
+ yield { error: error as Error, pos };
934
+ return;
861
935
  }
936
+ yield { type: p.type, start: p.start, end: p.end, value: p.value };
937
+ if (p.type === tt.eof) return;
862
938
  }
939
+ }
863
940
 
864
- /** Record how the open frames end, when the scan ends in them. */
865
- function recordOpen(result: number) {
866
- if (!braces) return;
867
- for (let k = 1; k < frames.length; k++) record(frames[k]!, result);
868
- }
941
+ // ---------------------------------------------------------------------------
942
+ // Lenient token walk
943
+ // ---------------------------------------------------------------------------
869
944
 
870
- /** Record how a frame ends. */
871
- function record(f: Frame, result: number) {
872
- const key = braces && frameKey(f);
873
- if (key !== undefined) braces!.set(key, result);
874
- }
945
+ /**
946
+ * Walk `src`, reporting code characters, literal text and variable
947
+ * references to `handlers` in source order, every character exactly once.
948
+ * Text acorn can't tokenize is read leniently: an unterminated literal runs
949
+ * to the end, and a character no JavaScript has is code. Returns
950
+ * `src.length`.
951
+ */
952
+ export function lexJs(
953
+ src: string,
954
+ handlers: JsLexHandlers,
955
+ goal: JsGoal = 'expression',
956
+ ): number {
957
+ return walkTokens(src, 0, handlers, 0, goal, false);
958
+ }
875
959
 
876
- /** How an earlier scan found the frame `f` to end, if it did. */
877
- const known = (f: Frame) => {
878
- const key = braces && frameKey(f);
879
- return key === undefined ? undefined : braces!.get(key);
880
- };
960
+ /**
961
+ * Lex the template literal opening at `start` (a backtick) as `lexJs` does,
962
+ * its interpolations one nesting level deeper. Returns the index just past
963
+ * its closing backtick, or `src.length` if it is unterminated.
964
+ */
965
+ export function lexTemplate(
966
+ src: string,
967
+ start: number,
968
+ handlers: JsLexHandlers = {},
969
+ nesting = 0,
970
+ ): number {
971
+ return walkTokens(src, start, handlers, nesting, 'expression', true);
972
+ }
881
973
 
882
- /** End a strict scan at a lexical error. */
883
- const malformed = () => {
884
- recordOpen(MALFORMED);
885
- recordOpenParens(MALFORMED);
886
- strict!.malformed = true;
887
- return src.length;
974
+ function walkTokens(
975
+ src: string,
976
+ from: number,
977
+ handlers: JsLexHandlers,
978
+ nesting: number,
979
+ goal: JsGoal,
980
+ oneTemplate: boolean,
981
+ ): number {
982
+ const comments: [number, number][] = [];
983
+ const onComment = (_block: boolean, _text: string, s: number, e: number) => {
984
+ comments.push([s, e]);
888
985
  };
889
-
890
- function push(f: Frame) {
891
- stackIds?.push(undefined);
892
- closers[f.closer]?.push(frames.length);
893
- frames.push(f);
894
- lowest?.push([Infinity, Infinity, Infinity]);
895
- startsAt?.push(bodyStarts);
896
- }
897
-
898
- /**
899
- * Just past a token or a space, line break or comment, but not within a
900
- * word: the key of the position and the whole scan state there, the open
901
- * frames included, which decide how the scan goes on (or undefined
902
- * elsewhere). No word is being read there, and the character before is no
903
- * word character, so a quote after it starts a string in any scan.
904
- *
905
- * Within brackets too, once the scan is long (`SHARE_AFTER`). Scans that
906
- * never passed such a point in the same state each ran on to the end of
907
- * the source, so many of them took quadratic time: inside an unclosed `(`
908
- * in `{(}{(}{(…` (there was no point within brackets), or after the regex
909
- * literals of `{a'</a </p>…` (there was no point after a space).
910
- */
911
- function checkpointKey(): CheckpointKey | undefined {
912
- const ternary = top().ternary;
913
- if (
914
- ternary > 3 ||
915
- !CHECKPOINT_AFTER.has(src.charAt(i - 1)) ||
916
- (frames.length > 1 && i - start < shareAfter)
917
- ) {
918
- return undefined;
919
- }
920
- let state = ternary;
921
- if (operandNext) state |= 1 << 2;
922
- if (stmtNext) state |= 1 << 3;
923
- if (keyNext) state |= 1 << 4;
924
- if (pendingBody) {
925
- state |= pendingBody.frame.kind === 'class' ? 1 << 5 : 1 << 6;
926
- if (pendingBody.frame.operand) state |= 1 << 7;
927
- }
928
- if (arrowBody) state |= 1 << 8;
929
- if (afterDot) state |= 1 << 9;
930
- if (afterHeaderKeyword) state |= 1 << 10;
931
- if (lineBreak) state |= 1 << 11;
932
- if (RESTRICTED_KEYWORDS.has(lastKeyword)) state |= 1 << 12;
933
- if (lastPunct === '=') state |= 1 << 13;
934
- // A run of dots: one or two (a spread may follow), three, or more
935
- if (lastPunct === '.') state |= Math.min(dots, 4) << 14;
936
- const at = i * (1 << 17) + state;
937
- if (frames.length === 1) return at;
938
- // A function or class body to come may open at an outer level
939
- const body = pendingBody ? frames.length - pendingBody.depth : '';
940
- return `${at} ${stackId()} ${body}`;
941
- }
942
-
943
- /**
944
- * May a string literal follow the word just read (`word`, or a variable
945
- * reference) with nothing between? Only after a keyword taking an operand
946
- * (`typeof'x'`, `case"a"`), `of` in a `for` header, or a modifier before a
947
- * property name (`static'x'`, `get"y"() {}`); elsewhere the quote is an
948
- * apostrophe (`don't`), and the text is no JavaScript.
949
- */
950
- function stringMayFollowWord(): boolean {
951
- return (
952
- OPERAND_KEYWORDS.has(word) ||
953
- (word === 'of' && top().header) ||
954
- (keyNext && MODIFIERS.has(word))
955
- );
956
- }
957
-
958
- /** Close the frames from index `k` on. */
959
- function truncate(k: number) {
960
- if (lowest) {
961
- for (let j = frames.length - 1; j >= k; j--) foldLowest(j);
962
- lowest.length = k;
963
- }
964
- if (startsAt) startsAt.length = k;
965
- frames.length = k;
966
- if (stackIds) stackIds.length = k;
967
- for (const list of Object.values(closers)) {
968
- while (list.length && list[list.length - 1]! >= k) list.pop();
969
- }
970
- // A function or class body can't follow once its level is closed
971
- if (pendingBody && pendingBody.depth > k) pendingBody = undefined;
972
- }
973
-
974
- const code = (ch: string, index: number) =>
975
- handlers.code?.(ch, index, nesting);
976
- const literal = (text: string, index: number) =>
977
- handlers.literal?.(text, index, nesting);
978
- /** Literal text from `from` to `to`, sliced only for a handler. */
979
- const literalSpan = (from: number, to: number) =>
980
- handlers.literal?.(src.slice(from, to), from, nesting);
981
-
982
- /** Track the word just read (`i` is the index just past it). */
983
- function endWord() {
984
- if (!word) return;
985
- // A property name is never a keyword: `a.return`, `{ in: 1 }`
986
- const keyword = afterDot || keyNext ? '' : word;
987
- const inOperandPosition = operandNext && !stmtNext;
988
- operandNext =
989
- OPERAND_KEYWORDS.has(keyword) ||
990
- // `for (x of …)`, but `of` is an identifier where an operand goes
991
- (keyword === 'of' && top().header && !operandNext);
992
- afterHeaderKeyword = HEADER_KEYWORDS.has(keyword);
993
- if (keyword === 'function' || keyword === 'class') {
994
- // An expression where an operand is expected, else a declaration: at
995
- // a statement start, or after an operand and a line break (ASI)
996
- const body = frame(keyword === 'class' ? 'class' : 'block', '}');
997
- body.operand = inOperandPosition;
998
- pendingBody = { depth: frames.length, frame: body };
999
- bodyStarts++;
1000
- }
1001
- // `get name()`, `static _x = 1`: the property name is still to come
1002
- keyNext =
1003
- keyNext &&
1004
- MODIFIERS.has(word) &&
1005
- KEY_START_RE.test(src.charAt(skipTrivia(src, i)));
1006
- stmtNext =
1007
- keyword === 'else' ||
1008
- keyword === 'do' ||
1009
- (keyword === 'async' && stmtNext);
1010
- arrowBody = false;
1011
- afterDot = false;
1012
- lastPunct = '';
1013
- lastKeyword = keyword;
1014
- lineBreak = false;
1015
- word = '';
1016
- }
1017
-
1018
- /** A string, template or regex literal, or a variable reference, ended. */
1019
- function endOperand() {
1020
- endWord();
1021
- operandNext = false;
1022
- stmtNext = false;
1023
- arrowBody = false;
1024
- keyNext = false;
1025
- afterHeaderKeyword = false;
1026
- afterDot = false;
1027
- lastPunct = '';
1028
- lastKeyword = '';
1029
- lineBreak = false;
1030
- }
1031
-
1032
- /** Track an operator token: `operandNext` tells what may follow it. */
1033
- function endPunct(punct: string, nextIsOperand: boolean) {
1034
- operandNext = nextIsOperand;
1035
- stmtNext = false;
1036
- arrowBody = false;
1037
- keyNext = false;
1038
- afterDot = punct === '.' && !nextIsOperand;
1039
- afterHeaderKeyword = false;
1040
- lastPunct = punct;
1041
- lastKeyword = '';
1042
- lineBreak = false;
1043
- }
1044
-
1045
- /**
1046
- * A line break between tokens. After `return`, `break` or `continue` it
1047
- * ends the statement (ASI): `return⏎function f() {}` declares `f`.
1048
- */
1049
- function lineBreakSeen() {
1050
- lineBreak = true;
1051
- if (RESTRICTED_KEYWORDS.has(lastKeyword)) {
1052
- operandNext = true;
1053
- stmtNext = true;
1054
- lastKeyword = '';
1055
- }
1056
- }
1057
-
1058
- function open(f: Frame) {
1059
- push(f);
1060
- stmtNext = f.kind === 'block';
1061
- keyNext = f.kind === 'object' || f.kind === 'class';
1062
- }
1063
-
1064
- /**
1065
- * Go past frame `f` if an earlier scan lexed it: to just after it, or to
1066
- * the end of the source when it never closes. Whether it did, or
1067
- * MALFORMED when that scan found the code malformed.
1068
- */
1069
- function skipLexed(f: Frame): boolean | typeof MALFORMED {
1070
- const end = known(f);
1071
- if (end === undefined) return false;
1072
- if (end === MALFORMED) return MALFORMED;
1073
- i = end === UNCLOSED ? src.length : resume(end) + 1;
1074
- return true;
1075
- }
1076
-
1077
- function openBrace() {
1078
- let f: Frame;
1079
- if (pendingBody?.depth === frames.length) {
1080
- f = pendingBody.frame;
1081
- pendingBody = undefined;
1082
- } else if (!operandNext || stmtNext || arrowBody) {
1083
- f = frame('block', '}');
1084
- } else {
1085
- f = frame('object', '}');
1086
- f.operand = true;
1087
- }
1088
- f.open = i;
1089
- endPunct('{', true);
1090
- const end = known(f);
1091
- if (end === undefined) open(f);
1092
- else skipTo = { frame: f, end };
1093
- }
1094
-
1095
- /** State after the `}` closing `closed`, a block or literal. */
1096
- function afterBrace(closed: Frame) {
1097
- if (closed.operand) {
1098
- endPunct('}', false);
1099
- } else {
1100
- // A block: a statement (or the next class member) may follow
1101
- endPunct('}', true);
1102
- stmtNext = top().kind === 'block';
1103
- keyNext = top().kind === 'class';
1104
- }
1105
- }
1106
-
1107
- /**
1108
- * Close the innermost frame that `c` closes; a stray closer, with no such
1109
- * frame within the innermost braces (a block, an object literal, a class
1110
- * body or an interpolation), is ignored. So a stray `)` or `]` never
1111
- * closes the braces around it, and how a `{…}` ends depends only on the
1112
- * code inside it.
1113
- */
1114
- function close(c: (typeof CLOSERS)[number]) {
1115
- const k = Math.max(innermost(c), innermost('}'));
1116
- lookedFor(c, k);
1117
- if (k === 0 || frames[k]!.closer !== c) {
1118
- endPunct(c, c === '}');
1119
- return;
1120
- }
1121
- const closed = frames[k]!;
1122
- const result = closedAt(k);
1123
- if (lowest && c !== '}') {
1124
- // Its entry complete, with those of the frames still open in it
1125
- for (let j = frames.length - 1; j > k; j--) foldLowest(j);
1126
- recordParen(k, result);
1127
- }
1128
- truncate(k);
1129
- if (c === ']' && !ctx.lookahead) ctx.brackets.set(closed.open, i);
1130
- if (c === ')') {
1131
- // `if (…) %x = 1` vs `($n)%3`
1132
- endPunct(c, closed.header);
1133
- stmtNext = closed.header;
1134
- } else if (c === ']') {
1135
- endPunct(c, false);
1136
- } else {
1137
- record(closed, result);
1138
- afterBrace(closed);
1139
- }
1140
- }
1141
-
1142
- function trackCode(c: string) {
1143
- if (WORD_CHAR_RE.test(c)) {
1144
- word += c;
1145
- return;
1146
- }
1147
- endWord();
1148
- // A line break alone never changes operand/operator position:
1149
- // `$x = 5\n%n` continues the expression, as in JavaScript.
1150
- if (LINE_TERMINATOR_RE.test(c)) lineBreakSeen();
1151
- if (SPACE_RE.test(c)) return;
1152
- const t = top();
1153
- switch (c) {
1154
- case '(':
1155
- case '[': {
1156
- const f = frame('expr', c === '(' ? ')' : ']', i);
1157
- f.header = c === '(' && afterHeaderKeyword;
1158
- endPunct(c, true);
1159
- const result = knownParen(f);
1160
- if (result === undefined) open(f);
1161
- else parenTo = { frame: f, result };
1162
- break;
1163
- }
1164
- case '{':
1165
- openBrace();
1166
- break;
1167
- case ')':
1168
- case ']':
1169
- case '}':
1170
- close(c);
1171
- break;
1172
- case '.':
1173
- // Property access, unless it is the spread `...`
1174
- dots = lastPunct === '.' ? dots + 1 : 1;
1175
- endPunct(c, dots === 3);
1176
- break;
1177
- case ';':
1178
- endPunct(c, true);
1179
- stmtNext = t.kind === 'block';
1180
- keyNext = t.kind === 'class';
1181
- break;
1182
- case ',':
1183
- endPunct(c, true);
1184
- keyNext = t.kind === 'object';
1185
- break;
1186
- case '?':
1187
- endPunct(c, true);
1188
- t.ternary++;
1189
- break;
1190
- case ':':
1191
- endPunct(c, true);
1192
- // Not a conditional's `:`: an object literal value, or a statement
1193
- // after a `case`, `default` or label
1194
- if (t.ternary > 0) t.ternary--;
1195
- else stmtNext = t.kind === 'block';
1196
- break;
1197
- case '#':
1198
- // A private name: `this.#_x`
1199
- endPunct(c, false);
1200
- afterDot = true;
1201
- break;
1202
- case '*': {
1203
- // A generator method: `{ *_gen() {} }`
1204
- const key = keyNext;
1205
- endPunct(c, true);
1206
- keyNext = key;
1207
- break;
1208
- }
1209
- case '>': {
1210
- // `=>`: the body may be a block
1211
- const arrow = lastPunct === '=';
1212
- endPunct(c, true);
1213
- arrowBody = arrow;
1214
- break;
1215
- }
1216
- default:
1217
- endPunct(c, true);
986
+ const code = (a: number, b: number) => {
987
+ for (let i = a; i < b; i++) handlers.code?.(src.charAt(i), i, nesting);
988
+ };
989
+ const literal = (a: number, b: number) => {
990
+ if (b > a) handlers.literal?.(src.slice(a, b), a, nesting);
991
+ };
992
+ /** Spaces and comments between tokens. */
993
+ const gap = (a: number, b: number) => {
994
+ for (const [s, e] of comments) {
995
+ if (e <= a || s >= b) continue;
996
+ code(a, s);
997
+ literal(s, e);
998
+ a = e;
1218
999
  }
1219
- }
1220
-
1221
- while (i < src.length) {
1222
- const ch = src.charAt(i);
1223
-
1224
- // At the top level just past a `}`: go on as a scan that passed here in
1225
- // the same state did
1226
- const checkpoints = strict?.checkpoints;
1227
- const key = checkpoints && checkpointKey();
1228
- if (key !== undefined) {
1229
- const known = checkpoints!.get(key);
1230
- if (known === undefined) {
1231
- strict!.passed.push(key);
1232
- } else if (frames.length > 1 && known < 0) {
1233
- // The frames open here may close before the error or the end of the
1234
- // source, so their results are not known: none is recorded
1235
- strict!.malformed = known === MALFORMED;
1000
+ comments.length = 0;
1001
+ code(a, b);
1002
+ };
1003
+ /** Open brackets; `${` stands for a template interpolation. */
1004
+ const stack: string[] = [];
1005
+ let templates = 0;
1006
+ let gen = tokens(src, from, goal, { onComment });
1007
+ let pos = from;
1008
+ let prev: TokenType | undefined;
1009
+ for (;;) {
1010
+ const tok = gen.next().value!;
1011
+ if ('error' in tok) {
1012
+ const at = Math.max(pos, tok.pos);
1013
+ gap(pos, at);
1014
+ if (at >= src.length) return src.length;
1015
+ if (/^Unterminated/.test(tok.error.message)) {
1016
+ literal(at, src.length);
1236
1017
  return src.length;
1237
- } else if (known === MALFORMED) {
1238
- return malformed();
1239
- } else if (known === UNCLOSED) {
1240
- i = src.length;
1241
- break;
1242
- } else {
1243
- strict!.stopped = true;
1244
- return known;
1245
- }
1246
- }
1247
-
1248
- // Template literal text: escapes, the closing backtick, interpolations
1249
- if (top().kind === 'template') {
1250
- if (ch === '\\') {
1251
- literal(src.slice(i, i + 2), i);
1252
- i += 2;
1253
- } else if (ch === '`') {
1254
- literal(ch, i);
1255
- if (frames.length === 1) return i + 1; // the end of `lexTemplate`
1256
- record(top(), closedAt(frames.length - 1));
1257
- i++;
1258
- truncate(frames.length - 1);
1259
- endOperand();
1260
- } else if (ch === '$' && src.charAt(i + 1) === '{') {
1261
- const f = frame('expr', '}', i);
1262
- f.interpolation = true;
1263
- // An interpolation an earlier scan lexed: on with the text after it
1264
- const skipped = skipLexed(f);
1265
- if (skipped === MALFORMED) return malformed();
1266
- if (skipped) continue;
1267
- literal('${', i);
1268
- i += 2;
1269
- nesting++;
1270
- endPunct('{', true);
1271
- open(f);
1272
- } else {
1273
- literal(ch, i);
1274
- i++;
1275
1018
  }
1019
+ // A character no JavaScript has: code, and on after it
1020
+ code(at, at + 1);
1021
+ pos = at + 1;
1022
+ gen = tokens(src, pos, goal, { onComment });
1023
+ prev = undefined;
1276
1024
  continue;
1277
1025
  }
1278
-
1279
- // String literal — skip entirely
1280
- if (ch === '"' || ch === "'") {
1281
- if (strict && i > start && WORD_CHAR_RE.test(src.charAt(i - 1))) {
1282
- if (!stringMayFollowWord()) return malformed();
1283
- }
1284
- endWord();
1285
- const { end, closed } = scanStringLiteral(src, i);
1286
- if (strict && !closed) return malformed();
1287
- literalSpan(i, end);
1288
- i = end;
1289
- endOperand();
1290
- continue;
1026
+ const t = tok.type;
1027
+ if (t === tt.eof) {
1028
+ gap(pos, src.length);
1029
+ return src.length;
1291
1030
  }
1292
-
1293
- if (ch === '`') {
1294
- endWord();
1295
- const f = frame('template', '`', i);
1296
- // A template literal an earlier scan lexed: on after it
1297
- const skipped = skipLexed(f);
1298
- if (skipped === MALFORMED) return malformed();
1299
- if (skipped) {
1300
- endOperand();
1301
- continue;
1302
- }
1303
- literal(ch, i);
1304
- push(f);
1305
- i++;
1306
- continue;
1307
- }
1308
-
1309
- if (ch === '{' && strict?.stop?.(i)) {
1310
- recordOpenParens(STOPPED - i);
1311
- strict.stopped = true;
1312
- return i;
1313
- }
1314
-
1315
- // End of a template literal interpolation: the innermost `}` closer
1316
- if (ch === '}') {
1317
- const k = innermost('}');
1318
- lookedFor('}', k);
1319
- if (frames[k]!.interpolation) {
1320
- endWord();
1321
- record(frames[k]!, closedAt(k));
1322
- truncate(k);
1323
- nesting--;
1324
- literal(ch, i);
1325
- i++;
1326
- continue;
1327
- }
1328
- }
1329
-
1330
- if (ch === '/') {
1331
- endWord();
1332
- const next = src.charAt(i + 1);
1333
- // Comment — skip entirely; it is not a token
1334
- if (next === '/' || next === '*') {
1335
- let end = findCommentEnd(src, i, strict?.cache);
1336
- if (end < 0) {
1337
- if (strict) return malformed();
1338
- end = src.length;
1339
- }
1340
- literalSpan(i, end);
1341
- if (next === '*' && lineBreakIn(src, i, end, strict?.cache)) {
1342
- lineBreakSeen();
1343
- }
1344
- i = end;
1345
- continue;
1346
- }
1347
- // Regex literal — only where an operand is expected
1348
- if (operandNext) {
1349
- const { end, closed } = scanRegex(src, i, strict?.cache);
1350
- if (strict && !closed) return malformed();
1351
- literalSpan(i, end);
1352
- i = end;
1353
- endOperand();
1354
- continue;
1355
- }
1356
- }
1357
-
1358
- // In a class body, a line break after a complete member starts the next
1359
- // one (ASI): `_x = 1⏎_y = 2`, but `_x = a⏎instanceof B` continues it.
1360
- if (!word && lineBreak && !operandNext && top().kind === 'class') {
1361
- IDENT_RE.lastIndex = i;
1362
- const next = IDENT_RE.exec(src)?.[0];
1363
- if (next && next !== 'in' && next !== 'instanceof') keyNext = true;
1364
- }
1365
-
1366
- // `$name`, `_name` or `@name` reference where an identifier starts (`@`
1367
- // is no identifier character, so `typeof@x` holds one). The whole
1368
- // identifier must be the sigil and a name: `$a$b` and `$café` are
1369
- // identifiers of their own.
1370
- if (ch === '@' || ((ch === '$' || ch === '_') && !word)) {
1371
- endWord();
1372
- IDENT_RE.lastIndex = i + 1;
1373
- const name = IDENT_RE.exec(src)?.[0] ?? '';
1374
- if (VAR_NAME_RE.test(name) && !isPropertyName(ch, i + 1 + name.length)) {
1375
- handlers.variable?.(ch, name, i, nesting);
1376
- i += 1 + name.length;
1377
- endOperand();
1378
- continue;
1031
+ gap(pos, tok.start);
1032
+ if (t === tt.name) {
1033
+ const sigil = sigilAt(src, tok.start, tok.value as string);
1034
+ const property =
1035
+ prev === tt.dot ||
1036
+ prev === tt.questionDot ||
1037
+ ((sigil === '$' || sigil === '_') && keyPosition(src, tok.end, prev));
1038
+ if (sigil && !property) {
1039
+ handlers.variable?.(
1040
+ sigil,
1041
+ src.slice(tok.start + 1, tok.end),
1042
+ tok.start,
1043
+ nesting,
1044
+ );
1045
+ } else code(tok.start, tok.end);
1046
+ } else if (
1047
+ t === tt.string ||
1048
+ t === tt.regexp ||
1049
+ t === tt.template ||
1050
+ t === tt.invalidTemplate
1051
+ ) {
1052
+ literal(tok.start, tok.end);
1053
+ } else if (t === tt.backQuote) {
1054
+ literal(tok.start, tok.end);
1055
+ // acorn reads a (maybe empty) text piece before each closing backtick
1056
+ if (prev === tt.template || prev === tt.invalidTemplate) {
1057
+ templates--;
1058
+ if (oneTemplate && templates === 0) return tok.end;
1059
+ } else {
1060
+ templates++;
1379
1061
  }
1380
- }
1381
-
1382
- // Transient reference — where an operand is expected, or as the target
1383
- // of an assignment starting a line: `$x = 5\n%a = 1` would otherwise be
1384
- // the invalid assignment `5 % a = 1`.
1385
- if (ch === '%') {
1386
- endWord();
1387
- TRANS_NAME_RE.lastIndex = i + 1;
1388
- const name = TRANS_NAME_RE.exec(src)?.[0];
1389
- if (name) {
1390
- const end = i + 1 + name.length;
1391
- if (
1392
- operandNext ||
1393
- (lineBreak && ctx.lookahead && assignmentFollows(src, end, ctx))
1394
- ) {
1395
- handlers.variable?.('%', name, i, nesting);
1396
- i = end;
1397
- endOperand();
1398
- continue;
1399
- }
1062
+ } else if (t === tt.dollarBraceL) {
1063
+ literal(tok.start, tok.end);
1064
+ stack.push('${');
1065
+ nesting++;
1066
+ } else if (t === tt.braceR && last(stack) === '${') {
1067
+ stack.pop();
1068
+ nesting--;
1069
+ literal(tok.start, tok.end);
1070
+ } else {
1071
+ if (t === tt.braceL || t === tt.parenL || t === tt.bracketL) {
1072
+ stack.push(src.charAt(tok.start));
1073
+ } else if (t === tt.braceR || t === tt.parenR || t === tt.bracketR) {
1074
+ if (stack.length && last(stack) !== '${') stack.pop();
1400
1075
  }
1076
+ code(tok.start, tok.end);
1401
1077
  }
1078
+ prev = t;
1079
+ pos = tok.end;
1080
+ }
1081
+ }
1402
1082
 
1403
- // Increment/decrement: postfix after an operand on the same line (no line
1404
- // break may precede postfix `++`), prefix otherwise.
1405
- if ((ch === '+' || ch === '-') && src.charAt(i + 1) === ch) {
1406
- endWord();
1407
- const postfix = !operandNext && !lineBreak;
1408
- code(ch, i);
1409
- code(ch, i + 1);
1410
- i += 2;
1411
- endPunct(ch, !postfix);
1412
- continue;
1413
- }
1414
-
1415
- // `??` and optional chaining `?.` (but `a?.5:1` is a conditional)
1416
- if (ch === '?') {
1417
- const next = src.charAt(i + 1);
1418
- if (next === '?' || (next === '.' && !/\d/.test(src.charAt(i + 2)))) {
1419
- endWord();
1420
- code(ch, i);
1421
- code(next, i + 1);
1422
- i += 2;
1423
- dots = 1;
1424
- endPunct(next, next === '?');
1425
- continue;
1426
- }
1427
- }
1083
+ /** The sigil of the name token `value` at `start`, if it is a reference. */
1084
+ function sigilAt(src: string, start: number, value: string): Sigil | null {
1085
+ const c = src.charAt(start);
1086
+ if ((c === '@' || c === '%') && value.charAt(0) === c) return c;
1087
+ if ((c === '$' || c === '_') && SIGIL_IDENT_RE.test(value)) return c;
1088
+ return null;
1089
+ }
1428
1090
 
1429
- // End of an `outer` square bracket: a `]` that closes no `[` within the
1430
- // innermost braces closes it, over any `(` left open in it, as it closes
1431
- // a nested `[`. Look-ahead scans record where the `[`s nested in theirs
1432
- // end (`ctx.brackets`), so one starting at a `[` must find the same.
1433
- if (
1434
- ch === outer.closer &&
1435
- (frames.length === 1 ||
1436
- (ch === ']' && innermost(']') === 0 && innermost('}') === 0))
1437
- ) {
1438
- break;
1439
- }
1091
+ /** A key in an object literal: after `{` or `,`, before `:` or `(`. */
1092
+ function keyPosition(
1093
+ src: string,
1094
+ end: number,
1095
+ prev: TokenType | undefined,
1096
+ ): boolean {
1097
+ if (prev !== tt.braceL && prev !== tt.comma) return false;
1098
+ const next = /\S/g;
1099
+ next.lastIndex = end;
1100
+ const c = next.exec(src)?.[0];
1101
+ return c === ':' || c === '(';
1102
+ }
1440
1103
 
1441
- // Regular code character
1442
- trackCode(ch);
1443
- if (skipTo) {
1444
- // A `{…}` frame an earlier scan lexed: continue after it
1445
- const { frame: skipped, end } = skipTo;
1446
- skipTo = undefined;
1447
- if (end === MALFORMED) return malformed();
1448
- if (end === UNCLOSED) {
1449
- i = src.length;
1450
- break;
1451
- }
1452
- i = resume(end);
1453
- afterBrace(skipped);
1454
- i++;
1455
- continue;
1456
- }
1457
- if (parenTo) {
1458
- // A `(…)` or `[…]` frame an earlier scan lexed: continue after it
1459
- const { frame: skipped, result } = parenTo;
1460
- parenTo = undefined;
1461
- if (result === MALFORMED) return malformed();
1462
- if (result === UNCLOSED) {
1463
- i = src.length;
1464
- break;
1465
- }
1466
- if (result <= STOPPED) {
1467
- recordOpenParens(result);
1468
- strict!.stopped = true;
1469
- return STOPPED - result;
1470
- }
1471
- i = resume(result);
1472
- if (skipped.closer === ')') {
1473
- endPunct(')', skipped.header);
1474
- stmtNext = skipped.header;
1475
- } else {
1476
- endPunct(']', false);
1477
- }
1478
- i++;
1479
- continue;
1104
+ /**
1105
+ * Index just past the regex literal (and its flags, whatever they are)
1106
+ * opening at `start`, or -1 when no `/` on its line closes it.
1107
+ */
1108
+ function regexEnd(src: string, start: number): number {
1109
+ let inClass = false;
1110
+ for (let i = start + 1; i < src.length; i++) {
1111
+ const c = src.charAt(i);
1112
+ if (LINE_BREAK_RE.test(c)) return -1;
1113
+ if (c === '\\') i++;
1114
+ else if (c === '[') inClass = true;
1115
+ else if (c === ']') inClass = false;
1116
+ else if (c === '/' && !inClass) {
1117
+ FLAGS_RE.lastIndex = i + 1;
1118
+ FLAGS_RE.test(src);
1119
+ return FLAGS_RE.lastIndex;
1480
1120
  }
1481
- code(ch, i);
1482
- i++;
1483
1121
  }
1484
- if (i >= src.length) {
1485
- recordOpen(UNCLOSED);
1486
- recordOpenParens(UNCLOSED);
1487
- }
1488
- if (!ctx.lookahead) {
1489
- // Brackets left open here close at `i`: the end of `outer` or of `src`
1490
- for (const f of frames) if (f.closer === ']') ctx.brackets.set(f.open, i);
1491
- }
1492
- return Math.min(i, src.length);
1122
+ return -1;
1123
+ }
1493
1124
 
1494
- /**
1495
- * Is the `$`/`_` word ending at `end` a property name: after `.`, a key
1496
- * in an object literal (`{ _id: 1 }`, `{ _m() {} }`) or a member name in
1497
- * a class body?
1498
- */
1499
- function isPropertyName(sigil: string, end: number): boolean {
1500
- if (afterDot) return true;
1501
- if (!keyNext || sigil === '@') return false;
1502
- if (top().kind === 'class') return true;
1503
- const next = src.charAt(skipTrivia(src, end));
1504
- return next === ':' || next === '(';
1125
+ /**
1126
+ * Scan the `"…"` or `'…'` string literal opening at `start`. `end` is the
1127
+ * index just past its closing quote, or `src.length` when it is unterminated
1128
+ * (`closed` false). A backslash escapes the character after it, so a quote
1129
+ * after an even run of backslashes closes the string and one after an odd
1130
+ * run does not. Line breaks don't end it: quoted macro labels may span lines.
1131
+ */
1132
+ export function scanStringLiteral(
1133
+ src: string,
1134
+ start: number,
1135
+ ): { end: number; closed: boolean } {
1136
+ const quote = src.charAt(start);
1137
+ let i = start + 1;
1138
+ while (i < src.length) {
1139
+ const c = src.charAt(i);
1140
+ if (c === '\\') i += 2;
1141
+ else if (c === quote) return { end: i + 1, closed: true };
1142
+ else i++;
1505
1143
  }
1144
+ return { end: src.length, closed: false };
1506
1145
  }