@rohal12/spindle 0.51.4 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/dist/pkg/format.js +1 -1
  2. package/dist/pkg/headless.js +4313 -1640
  3. package/dist/pkg/macro-registry.json +7 -7
  4. package/dist/pkg/story-variables.js +1636 -177
  5. package/package.json +5 -2
  6. package/src/automation/runner.ts +2 -1
  7. package/src/class-registry.ts +214 -103
  8. package/src/components/Passage.tsx +2 -2
  9. package/src/components/PassageDialog.tsx +2 -5
  10. package/src/components/StoryInterface.tsx +2 -4
  11. package/src/components/macros/Button.tsx +5 -31
  12. package/src/components/macros/Checkbox.tsx +7 -4
  13. package/src/components/macros/Computed.tsx +19 -13
  14. package/src/components/macros/For.tsx +29 -3
  15. package/src/components/macros/If.tsx +8 -0
  16. package/src/components/macros/Include.tsx +7 -6
  17. package/src/components/macros/MacroError.tsx +2 -1
  18. package/src/components/macros/MacroLink.tsx +12 -45
  19. package/src/components/macros/Meter.tsx +11 -3
  20. package/src/components/macros/Nobr.tsx +1 -0
  21. package/src/components/macros/PassageDisplay.tsx +3 -0
  22. package/src/components/macros/Print.tsx +4 -0
  23. package/src/components/macros/Radiobutton.tsx +5 -2
  24. package/src/components/macros/SaveManager.tsx +25 -8
  25. package/src/components/macros/Span.tsx +1 -0
  26. package/src/components/macros/StoryTitle.tsx +1 -0
  27. package/src/components/macros/Switch.tsx +13 -0
  28. package/src/components/macros/Unset.tsx +30 -10
  29. package/src/components/macros/VarDisplay.tsx +21 -4
  30. package/src/components/macros/Widget.tsx +20 -1
  31. package/src/components/macros/WidgetInvocation.tsx +17 -75
  32. package/src/components/macros/arg-utils.ts +107 -1
  33. package/src/components/macros/detached-body.tsx +68 -0
  34. package/src/components/macros/option-utils.ts +3 -2
  35. package/src/define-macro.ts +32 -5
  36. package/src/execute-mutation.ts +270 -67
  37. package/src/expression.ts +86 -55
  38. package/src/hooks/use-action.ts +18 -3
  39. package/src/hooks/use-interpolate.ts +36 -5
  40. package/src/index.tsx +10 -1
  41. package/src/interpolation.ts +394 -96
  42. package/src/js-lexer.ts +1231 -97
  43. package/src/markup/code-attributes.ts +64 -0
  44. package/src/markup/markdown.ts +188 -9
  45. package/src/markup/render.tsx +430 -49
  46. package/src/markup/tokenizer.ts +578 -110
  47. package/src/prng.ts +8 -8
  48. package/src/registry.ts +35 -0
  49. package/src/saves/save-manager.ts +339 -158
  50. package/src/saves/storage.ts +16 -7
  51. package/src/saves/types.ts +2 -1
  52. package/src/store.ts +521 -154
  53. package/src/story-api.ts +31 -67
  54. package/src/story-init.ts +1 -1
  55. package/src/story-variables.ts +98 -102
  56. package/src/triggers.ts +6 -5
  57. package/src/utils/error-message.ts +12 -0
  58. package/src/utils/live-locals.ts +10 -3
  59. package/src/utils/namespace.ts +71 -0
  60. package/src/utils/object-path.ts +99 -14
  61. package/src/utils/stable-key.ts +14 -9
  62. package/src/widgets/widget-registry.ts +9 -0
  63. package/types/index.d.ts +43 -7
  64. package/types/tooling.d.ts +1 -0
package/src/js-lexer.ts CHANGED
@@ -5,15 +5,27 @@
5
5
  * and comments begin and end. The expression transformer (`expression.ts`)
6
6
  * and the macro argument splitters (`components/macros/arg-utils.ts`) both
7
7
  * walk source text through `lexJs` and differ only in what they do with the
8
- * pieces it reports.
8
+ * pieces it reports. The passage tokenizer (`markup/tokenizer.ts`) and
9
+ * attribute interpolation (`interpolation.ts`) find where the code in a
10
+ * `{…}` ends with `findCodeEnd`.
11
+ *
12
+ * It also finds the sigil variable references in code: `$name`, `_name` and
13
+ * `@name` where an identifier starts — not inside `a$b` or `ñ_x`, and not
14
+ * where a property name stands: after `.`, as an object literal key or a
15
+ * class member name — and `%name`.
9
16
  *
10
17
  * Besides literals, the scanner tracks whether the next token is an operand
11
18
  * or an operator, which decides two ambiguities: `/` opens a regex in operand
12
19
  * position and divides otherwise, and `%name` is a transient reference in
13
20
  * operand position while `%` after an operand — `($n)%3`, `$a[i] %2`,
14
- * `_i++ %n` — is the modulo operator.
21
+ * `_i++ %n` — is the modulo operator. For that it tells blocks from object
22
+ * literals: `}` closing a block may be followed by a statement, `}` closing
23
+ * an object literal by an operator.
15
24
  */
16
25
 
26
+ /** The sigil of a variable reference: story, temporary, local, transient. */
27
+ export type Sigil = '$' | '_' | '@' | '%';
28
+
17
29
  export interface JsLexHandlers {
18
30
  /**
19
31
  * One character of code, outside literals and comments. `nesting` is the
@@ -28,42 +40,67 @@ export interface JsLexHandlers {
28
40
  * reported through `code`).
29
41
  */
30
42
  literal?(text: string, index: number, nesting: number): void;
31
- /** A `%name` transient variable reference (`name` without the `%`). */
32
- transient?(name: string, index: number, nesting: number): void;
43
+ /**
44
+ * A sigil variable reference — `$name`, `_name`, `@name` or `%name` —
45
+ * covering the sigil and `name`.
46
+ */
47
+ variable?(sigil: Sigil, name: string, index: number, nesting: number): void;
33
48
  }
34
49
 
35
- /** Transient name after `%`: an identifier, so `%3` is never a reference. */
36
- const TRANS_NAME_RE = /[A-Za-z_]\w*/y;
37
50
  /**
38
- * The rest of an assignment target and its operator after a transient name:
39
- * ` = 1`, `.a.b += 2`, but not `== 1` or `=> 1`.
51
+ * What the source is: one expression (a macro argument, an `evaluate`d
52
+ * expression) or a list of statements (the body of `execute`).
40
53
  */
41
- const ASSIGNMENT_RE =
42
- /(?:\s*\.\s*[A-Za-z_$][\w$]*)*\s*(?:\*\*|<<|>>>?|&&|\|\||\?\?|[-+*/%&|^])?=(?![=>])/y;
54
+ export type JsGoal = 'expression' | 'statements';
55
+
56
+ /** Transient name after `%`: an identifier, so `%3` is never a reference. */
57
+ const TRANS_NAME_RE = /[A-Za-z_]\w*/y;
58
+ /** An assignment operator: `=`, `+=`, `??=`, … but not `==` or `=>`. */
59
+ const ASSIGN_OP_RE = /(?:\*\*|<<|>>>?|&&|\|\||\?\?|[-+*/%&|^])?=(?![=>])/y;
43
60
  /** Flags after the closing `/` of a regex literal. */
44
61
  const REGEX_FLAGS_RE = /\w*/y;
45
- /** Characters of identifiers, numbers and sigil variable references. */
46
- const WORD_CHAR_RE = /[\w$@]/;
62
+ /**
63
+ * Characters of identifiers (with the `\u…` escapes they may contain) and
64
+ * numbers. Surrogates stand for the astral identifier characters they encode.
65
+ */
66
+ const WORD_CHAR_RE = /[\p{ID_Continue}$\u200c\u200d\\\ud800-\udfff]/u;
67
+ /** An identifier, or the rest of one after a `$`, `_` or `@` sigil. */
68
+ const IDENT_RE = /[\p{ID_Continue}$\u200c\u200d]*/uy;
69
+ /** A sigil variable name: the whole identifier after the sigil. */
70
+ const VAR_NAME_RE = /^\w+$/;
71
+ /** What may start a property name after a `get`, `set`, … modifier. */
72
+ const KEY_START_RE = /[\p{ID_Continue}$\\"'[*#]/u;
47
73
  const SPACE_RE = /\s/;
48
- /** Keywords followed by an operand rather than an operator. */
74
+ const LINE_TERMINATOR_RE = /[\n\r\u2028\u2029]/;
75
+ const LINE_TERMINATOR_G = /[\n\r\u2028\u2029]/g;
76
+ /**
77
+ * Keywords followed by an operand rather than an operator (a declaration's
78
+ * binding counts as one: `const of of xs`, `let { _a: x } = o`).
79
+ */
49
80
  const OPERAND_KEYWORDS = new Set([
50
81
  'await',
51
82
  'case',
83
+ 'const',
52
84
  'delete',
53
85
  'do',
54
86
  'else',
55
87
  'in',
56
88
  'instanceof',
89
+ 'let',
57
90
  'new',
58
- 'of',
59
91
  'return',
60
92
  'throw',
61
93
  'typeof',
94
+ 'var',
62
95
  'void',
63
96
  'yield',
64
97
  ]);
65
98
  /** Keywords whose parenthesised header is followed by a statement. */
66
99
  const HEADER_KEYWORDS = new Set(['for', 'if', 'while', 'with']);
100
+ /** Keywords that a line break ends the statement after. */
101
+ const RESTRICTED_KEYWORDS = new Set(['break', 'continue', 'return']);
102
+ /** Words that may precede a property name in an object literal or class. */
103
+ const MODIFIERS = new Set(['async', 'get', 'set', 'static']);
67
104
 
68
105
  /**
69
106
  * Scan the `"…"` or `'…'` string literal opening at `start`. `end` is the
@@ -87,44 +124,179 @@ export function scanStringLiteral(
87
124
  return { end: src.length, closed: false };
88
125
  }
89
126
 
90
- /** Index just past the regex literal (with flags) opening at `start`. */
91
- function skipRegex(src: string, start: number): number {
92
- let inClass = false; // inside `[…]`, where `/` does not close
127
+ /**
128
+ * Scan the regex literal (with flags) opening at `start`. `end` is the index
129
+ * just past it; an unterminated one (`closed` false) ends at the line break
130
+ * (escaped or not) or the end of the source.
131
+ */
132
+ function scanRegex(
133
+ src: string,
134
+ start: number,
135
+ cache?: JsScanCache,
136
+ ): { end: number; closed: boolean } {
137
+ // The rest of a regex reads the same from just past a class, however the
138
+ // scan got there: record how it ends there, and use what is recorded
139
+ const pending: number[] = [];
140
+ const done = (end: number, closed: boolean) => {
141
+ if (cache) {
142
+ for (const at of pending) cache.regexes.set(at, end * 2 + +closed);
143
+ }
144
+ return { end, closed };
145
+ };
93
146
  let i = start + 1;
94
147
  while (i < src.length) {
95
148
  const c = src.charAt(i);
96
149
  if (c === '\\') {
150
+ // An escaped line break ends the line, and the regex, too
151
+ if (LINE_TERMINATOR_RE.test(src.charAt(i + 1))) return done(i + 1, false);
97
152
  i += 2;
98
153
  continue;
99
154
  }
100
- if (c === '\n') return i; // unterminated: leave the rest to the parser
101
- if (inClass) {
102
- if (c === ']') inClass = false;
103
- } else if (c === '[') {
104
- inClass = true;
105
- } else if (c === '/') {
155
+ // Unterminated at the end of the line: leave the rest to the parser
156
+ if (LINE_TERMINATOR_RE.test(c)) return done(i, false);
157
+ if (c === '[') {
158
+ // A class, where `/` does not close
159
+ i = scanRegexClass(src, i, cache);
160
+ if (src.charAt(i) !== ']') return done(i, false);
161
+ i++;
162
+ const known = cache?.regexes.get(i);
163
+ if (known !== undefined)
164
+ return done(Math.floor(known / 2), known % 2 === 1);
165
+ pending.push(i);
166
+ continue;
167
+ }
168
+ if (c === '/') {
106
169
  REGEX_FLAGS_RE.lastIndex = i + 1;
107
- return i + 1 + (REGEX_FLAGS_RE.exec(src)?.[0].length ?? 0);
170
+ const flags = REGEX_FLAGS_RE.exec(src)?.[0].length ?? 0;
171
+ return done(i + 1 + flags, true);
108
172
  }
109
173
  i++;
110
174
  }
111
- return src.length;
175
+ return done(src.length, false);
112
176
  }
113
177
 
114
- /** Index just past the comment opening at `start` (`//` or `/*`). */
115
- function skipComment(src: string, start: number): number {
178
+ /**
179
+ * Index of the `]` closing the regex character class opening at `open`, or
180
+ * of the line break (or the end of the source) that leaves it unterminated.
181
+ * Every `[` in a class reads the rest of it the same way, so the result is
182
+ * recorded for them too: scans of `/[/[/[…` from each `/` share one pass.
183
+ */
184
+ function scanRegexClass(
185
+ src: string,
186
+ open: number,
187
+ cache?: JsScanCache,
188
+ ): number {
189
+ const known = cache?.classes.get(open);
190
+ if (known !== undefined) return known;
191
+ const opens = [open];
192
+ let i = open + 1;
193
+ while (i < src.length) {
194
+ const c = src.charAt(i);
195
+ if (c === '\\') {
196
+ if (LINE_TERMINATOR_RE.test(src.charAt(i + 1))) {
197
+ i++;
198
+ break;
199
+ }
200
+ i += 2;
201
+ continue;
202
+ }
203
+ if (c === ']' || LINE_TERMINATOR_RE.test(c)) break;
204
+ if (c === '[' && cache) opens.push(i);
205
+ i++;
206
+ }
207
+ i = Math.min(i, src.length);
208
+ if (cache) for (const o of opens) cache.classes.set(o, i);
209
+ return i;
210
+ }
211
+
212
+ /**
213
+ * The next index from `from` on where `find` matches (-1 for none), for
214
+ * searches from increasing positions: a search from within the stretch the
215
+ * last one covered has the same answer.
216
+ */
217
+ function nextMatch(
218
+ memo: NextMatch | undefined,
219
+ from: number,
220
+ find: (from: number) => number,
221
+ ): number {
222
+ if (memo && from >= memo.from && (memo.at < 0 || from <= memo.at)) {
223
+ return memo.at;
224
+ }
225
+ const at = find(from);
226
+ if (memo) {
227
+ memo.from = from;
228
+ memo.at = at;
229
+ }
230
+ return at;
231
+ }
232
+
233
+ /** A search memo for `nextMatch`. */
234
+ interface NextMatch {
235
+ from: number;
236
+ at: number;
237
+ }
238
+
239
+ /**
240
+ * Index just past the comment opening at `start` (`//` or `/*`), or -1 for
241
+ * an unterminated `/*` comment.
242
+ */
243
+ function findCommentEnd(
244
+ src: string,
245
+ start: number,
246
+ cache?: JsScanCache,
247
+ ): number {
116
248
  if (src.charAt(start + 1) === '/') {
117
- const end = src.indexOf('\n', start);
249
+ const end = nextLineBreak(src, start, cache);
118
250
  return end < 0 ? src.length : end;
119
251
  }
120
- const end = src.indexOf('*/', start + 2);
121
- return end < 0 ? src.length : end + 2;
252
+ const end = nextMatch(cache?.commentClose, start + 2, (from) =>
253
+ src.indexOf('*/', from),
254
+ );
255
+ return end < 0 ? -1 : end + 2;
256
+ }
257
+
258
+ /** Next line break from `from` on (-1 for none). */
259
+ function nextLineBreak(src: string, from: number, cache?: JsScanCache): number {
260
+ return nextMatch(cache?.lineEnd, from, (at) => {
261
+ LINE_TERMINATOR_G.lastIndex = at;
262
+ return LINE_TERMINATOR_G.exec(src)?.index ?? -1;
263
+ });
264
+ }
265
+
266
+ /** Is there a line break between `from` and `to`? */
267
+ function lineBreakIn(
268
+ src: string,
269
+ from: number,
270
+ to: number,
271
+ cache?: JsScanCache,
272
+ ): boolean {
273
+ if (!cache) return LINE_TERMINATOR_RE.test(src.slice(from, to));
274
+ const at = nextLineBreak(src, from, cache);
275
+ return at >= 0 && at < to;
276
+ }
277
+
278
+ /** Index just past the comment opening at `start` (`//` or `/*`). */
279
+ function skipComment(src: string, start: number): number {
280
+ const end = findCommentEnd(src, start);
281
+ return end < 0 ? src.length : end;
282
+ }
283
+
284
+ /** Index of the first character from `i` on that is no space or comment. */
285
+ function skipTrivia(src: string, i: number): number {
286
+ while (i < src.length) {
287
+ const c = src.charAt(i);
288
+ if (SPACE_RE.test(c)) i++;
289
+ else if (c === '/' && '/*'.includes(src.charAt(i + 1)))
290
+ i = skipComment(src, i);
291
+ else break;
292
+ }
293
+ return i;
122
294
  }
123
295
 
124
296
  /**
125
297
  * Lex the template literal opening at `start` (a backtick): its backticks,
126
298
  * text and escapes are reported as literal text and the code of its `${…}`
127
- * interpolations through `lexJs` one nesting level deeper. Returns the index
299
+ * interpolations as `lexJs` does, one nesting level deeper. Returns the index
128
300
  * just past the closing backtick, or `src.length` if it is unterminated.
129
301
  */
130
302
  export function lexTemplate(
@@ -133,92 +305,790 @@ export function lexTemplate(
133
305
  handlers: JsLexHandlers = {},
134
306
  nesting = 0,
135
307
  ): number {
136
- const literal = (text: string, index: number) =>
137
- handlers.literal?.(text, index, nesting);
138
- literal('`', start);
139
- let i = start + 1;
140
- while (i < src.length) {
308
+ handlers.literal?.('`', start, nesting);
309
+ const outer = frame('template', '`', start);
310
+ return scan(src, handlers, start + 1, nesting, outer, newContext());
311
+ }
312
+
313
+ /**
314
+ * State shared by the scans of one source: the main scan and the look-ahead
315
+ * scans that find where a bracketed assignment target ends.
316
+ */
317
+ interface ScanContext {
318
+ /**
319
+ * Look ahead after a `%name` starting a line for an assignment. Off in
320
+ * look-ahead scans, which only need to match brackets — and whose own
321
+ * look-ahead could rescan the same text over and over.
322
+ */
323
+ lookahead: boolean;
324
+ /**
325
+ * Index of the `]` matching the `[` at an index (`src.length` if there is
326
+ * none), as found by look-ahead scans: each text is scanned ahead once.
327
+ */
328
+ brackets: Map<number, number>;
329
+ /** Set for `findCodeEnd`: the scan stops at the first lexical error. */
330
+ strict?: StrictScan;
331
+ }
332
+
333
+ interface StrictScan {
334
+ /** Stop at a `{` in code for which this holds. */
335
+ stop?: (index: number) => boolean;
336
+ /** The scan ran into a lexical error. */
337
+ malformed: boolean;
338
+ /** The scan stopped where `stop` held. */
339
+ stopped: boolean;
340
+ /** Results shared with other scans of the source. */
341
+ cache?: JsScanCache;
342
+ /** `cache.braces`, unless the scan has a `stop`. */
343
+ braces?: Map<number, number>;
344
+ /** `cache.checkpoints` for this kind of scan. */
345
+ checkpoints?: Map<CheckpointKey, number>;
346
+ /** `cache.parens` for this kind of scan. */
347
+ parens?: Map<number, number>;
348
+ /** Checkpoints this scan passed, to record its result at. */
349
+ passed: CheckpointKey[];
350
+ }
351
+
352
+ /** Characters that a scan checkpoint follows (`checkpointKey`). */
353
+ const CHECKPOINT_AFTER = new Set([
354
+ '}',
355
+ '"',
356
+ "'",
357
+ '`',
358
+ '/',
359
+ '\n',
360
+ '\r',
361
+ '\u2028',
362
+ '\u2029',
363
+ ' ',
364
+ '\t',
365
+ ]);
366
+
367
+ /**
368
+ * How far into a scan its results start to be shared within brackets
369
+ * (`checkpointKey`, `knownParen`). Most scans end sooner, so they don't pay
370
+ * for recording results no other scan will use; a long one pays this much
371
+ * before it can use what earlier scans recorded, which keeps scans from
372
+ * many starts about linear.
373
+ */
374
+ const SHARE_AFTER = 256;
375
+
376
+ /**
377
+ * A scan checkpoint: a position and the scan state there (`checkpointKey`).
378
+ * A number at the top level, a string within brackets.
379
+ */
380
+ type CheckpointKey = number | string;
381
+
382
+ /** A frame result: it is still open at the end of the source. */
383
+ const UNCLOSED = -1;
384
+ /** A frame result: a lexical error inside it ends the scan. */
385
+ const MALFORMED = -2;
386
+ /** A frame result: the scan stops at index `s` inside it (`STOPPED - s`). */
387
+ const STOPPED = -3;
388
+
389
+ /** The closers whose search for their frame a frame result may depend on. */
390
+ const CLOSERS = [')', ']', '}'] as const;
391
+ const FRAME_KINDS: readonly Frame['kind'][] = [
392
+ 'block',
393
+ 'object',
394
+ 'class',
395
+ 'template',
396
+ ];
397
+
398
+ /**
399
+ * Key of a frame result, for the frames whose code lexes the same whatever
400
+ * surrounds them: braces (a block, an object literal, a class body), a
401
+ * template literal and its interpolations. Undefined for parentheses and
402
+ * square brackets, which a stray closer inside may close.
403
+ */
404
+ function frameKey(f: Frame): number | undefined {
405
+ const kind = f.interpolation ? 4 : FRAME_KINDS.indexOf(f.kind);
406
+ return kind < 0 ? undefined : f.open * 5 + kind;
407
+ }
408
+
409
+ /**
410
+ * Results that `findCodeEnd` scans of one source share, so that scanning it
411
+ * from many starts doesn't lex the same code over and over.
412
+ */
413
+ export interface JsScanCache {
414
+ /**
415
+ * Where each `{…}` frame, template literal or `${…}` interpolation closes:
416
+ * the index of its `}` or closing backtick, `UNCLOSED` or `MALFORMED`. The
417
+ * code inside such a frame lexes the same whatever surrounds it, given
418
+ * where it opens and its kind — a stray `)` or `]` inside never closes it
419
+ * — so a later scan entering the same frame skips to its end.
420
+ */
421
+ braces: Map<number, number>;
422
+ /** Look-ahead bracket matches (`ScanContext.brackets`). */
423
+ brackets: Map<number, number>;
424
+ /** Regex character class ends (`scanRegexClass`). */
425
+ classes: Map<number, number>;
426
+ /** How a regex goes on from just past a class: `end * 2 + closed`. */
427
+ regexes: Map<number, number>;
428
+ /** The last search for a line break ending a `//` comment. */
429
+ lineEnd: NextMatch;
430
+ /** The last search for the end of a block comment. */
431
+ commentClose: NextMatch;
432
+ /**
433
+ * How scans go on from points, by the kind of scan (its goal and how it
434
+ * ends) and then by the point and the scan state there, the brackets open
435
+ * around it included: the index the scan ends at, `UNCLOSED` or
436
+ * `MALFORMED`. The points are where no word is being read (see
437
+ * `checkpointKey`). Scans from different starts soon pass such points in
438
+ * the same state, and from there on go the same way.
439
+ */
440
+ checkpoints: Map<string, Map<CheckpointKey, number>>;
441
+ /**
442
+ * Ids of the stacks of open brackets that checkpoints have seen (see
443
+ * `scan`), by the id of the stack below the innermost bracket, the state
444
+ * of that one and the innermost bracket.
445
+ */
446
+ stacks: Map<string, number>;
447
+ /**
448
+ * How `(…)` and `[…]` frames end, by the kind of scan and then by the frame
449
+ * (`parenKey`): `end * 2 + 1` if it closes at `end` and the code inside
450
+ * started a function or class body (which replaced the one to come, and
451
+ * can't follow once the frame is closed), `end * 2` otherwise, or
452
+ * `UNCLOSED`, `MALFORMED` or `STOPPED - s`. Unlike braces, a stray closer inside may close a frame
453
+ * around them, so the code inside lexes the same only around frames for
454
+ * which the closers it met find nothing to close; the key says which
455
+ * closers met none.
456
+ */
457
+ parens: Map<string, Map<number, number>>;
458
+ /**
459
+ * How far into a scan its results start to be shared within brackets
460
+ * (`SHARE_AFTER`; tests set 0 to share them all).
461
+ */
462
+ shareAfter: number;
463
+ }
464
+
465
+ export function createJsScanCache(): JsScanCache {
466
+ return {
467
+ checkpoints: new Map(),
468
+ stacks: new Map(),
469
+ parens: new Map(),
470
+ shareAfter: SHARE_AFTER,
471
+ braces: new Map(),
472
+ brackets: new Map(),
473
+ classes: new Map(),
474
+ regexes: new Map(),
475
+ lineEnd: { from: Infinity, at: -1 },
476
+ commentClose: { from: Infinity, at: -1 },
477
+ };
478
+ }
479
+
480
+ const newContext = (): ScanContext => ({
481
+ lookahead: true,
482
+ brackets: new Map(),
483
+ });
484
+
485
+ /** Index of the `]` matching the `[` at `open`, or `src.length`. */
486
+ function matchBracket(src: string, open: number, ctx: ScanContext): number {
487
+ let close = ctx.brackets.get(open);
488
+ if (close === undefined) {
489
+ const ahead = { lookahead: false, brackets: ctx.brackets };
490
+ close = scan(src, {}, open + 1, 0, frame('expr', ']', open), ahead);
491
+ ctx.brackets.set(open, close);
492
+ }
493
+ return close;
494
+ }
495
+
496
+ /**
497
+ * Is the code at `i`, just past a `%name`, the rest of an assignment target
498
+ * and its operator: ` = 1`, `.a[b] += 2`, but not `== 1`, `=> 1` or `% 2`?
499
+ */
500
+ function assignmentFollows(src: string, i: number, ctx: ScanContext): boolean {
501
+ for (;;) {
502
+ i = skipTrivia(src, i);
141
503
  const c = src.charAt(i);
142
- if (c === '\\') {
143
- literal(src.slice(i, i + 2), i);
144
- i += 2;
145
- } else if (c === '`') {
146
- literal(c, i);
147
- return i + 1;
148
- } else if (c === '$' && src.charAt(i + 1) === '{') {
149
- literal('${', i);
150
- i = lexJs(src, handlers, i + 2, nesting + 1);
151
- if (i < src.length) {
152
- literal('}', i);
153
- i++;
154
- }
155
- } else {
156
- literal(c, i);
504
+ if (c === '.' && src.charAt(i + 1) !== '.') {
505
+ IDENT_RE.lastIndex = skipTrivia(src, i + 1);
506
+ const name = IDENT_RE.exec(src)?.[0];
507
+ if (!name) return false;
508
+ i = IDENT_RE.lastIndex;
509
+ } else if (c === '[') {
510
+ i = matchBracket(src, i, ctx);
511
+ if (i >= src.length) return false;
157
512
  i++;
513
+ } else {
514
+ break;
158
515
  }
159
516
  }
160
- return Math.min(i, src.length);
517
+ ASSIGN_OP_RE.lastIndex = i;
518
+ return ASSIGN_OP_RE.test(src);
519
+ }
520
+
521
+ /**
522
+ * The code between a pair of brackets, or the whole source.
523
+ *
524
+ * - `block`: statements (a block, a function body, the whole source as
525
+ * statements),
526
+ * - `object`: the property definitions of an object literal,
527
+ * - `class`: the member definitions of a class body,
528
+ * - `expr`: an expression (parentheses, square brackets, a template literal
529
+ * interpolation, the whole source as an expression),
530
+ * - `template`: the text of a template literal.
531
+ */
532
+ interface Frame {
533
+ kind: 'block' | 'object' | 'class' | 'expr' | 'template';
534
+ /** The character closing it, or '' for the whole source. */
535
+ closer: string;
536
+ /** Index of its opening bracket. */
537
+ open: number;
538
+ /** A parenthesised `if`/`for`/`while`/`with` header: a statement follows. */
539
+ header: boolean;
540
+ /** Its closing `}` ends an operand (object literal, function expression). */
541
+ operand: boolean;
542
+ /** Conditional-expression `?`s awaiting their `:`. */
543
+ ternary: number;
544
+ /** A template literal's `${…}` interpolation. */
545
+ interpolation: boolean;
546
+ }
547
+
548
+ function frame(kind: Frame['kind'], closer: string, open = -1): Frame {
549
+ return {
550
+ kind,
551
+ closer,
552
+ open,
553
+ header: false,
554
+ operand: false,
555
+ ternary: 0,
556
+ interpolation: false,
557
+ };
161
558
  }
162
559
 
163
560
  /**
164
- * Walk `src` from `start`, reporting code characters, literal text and
165
- * transient references to `handlers` in source order. Every character of the
166
- * scanned range is reported exactly once (a transient reference covers its
167
- * `%` and name). Returns the index where scanning stopped: `src.length`, or
168
- * — for an interpolation (`nesting > 0`) — the index of the `}` closing it.
561
+ * Walk `src`, reporting code characters, literal text and variable
562
+ * references to `handlers` in source order. Every character is reported
563
+ * exactly once (a variable reference covers its sigil and name). Returns
564
+ * `src.length`.
169
565
  */
170
566
  export function lexJs(
171
567
  src: string,
172
568
  handlers: JsLexHandlers,
173
- start = 0,
174
- nesting = 0,
569
+ goal: JsGoal = 'expression',
175
570
  ): number {
176
- const interpolation = nesting > 0;
571
+ const outer = frame(goal === 'statements' ? 'block' : 'expr', '');
572
+ return scan(src, handlers, 0, 0, outer, newContext());
573
+ }
574
+
575
+ export interface FindCodeEndOptions {
576
+ /** What the code is (default `expression`). */
577
+ goal?: JsGoal;
578
+ /** End the code at a `{` in code, at any depth, for which this holds. */
579
+ stop?: (index: number) => boolean;
580
+ /**
581
+ * Names `stop` for `cache`: scans with the same key share results (scans
582
+ * with a `stop` but no key don't use `cache.checkpoints`).
583
+ */
584
+ stopKey?: string;
585
+ /**
586
+ * Results to share with other scans of the same source. Scanning a source
587
+ * from n starts then takes about linear time instead of n scans of the
588
+ * rest of it.
589
+ */
590
+ cache?: JsScanCache;
591
+ }
592
+
593
+ /**
594
+ * Find where the code starting at `start` ends, lexing it as JavaScript:
595
+ * braces, quotes and backticks inside string, template and regex literals and
596
+ * comments don't count.
597
+ *
598
+ * Without `stop`, the code ends at the first `}` in code outside the
599
+ * brackets it opened (the `}` closing a `{…}` around it); with `stop`, at
600
+ * the first `{` in code, at any depth, for which `stop` holds. Returns the
601
+ * index of that `}` or `{`.
602
+ *
603
+ * Returns -1 when there is no such end, or when the code before it is not
604
+ * well-formed JavaScript as far as a lexer can tell: an unterminated string,
605
+ * regex literal or block comment, or a quote directly after an identifier or
606
+ * number (`don't`, but not `typeof'x'`). Callers fall back to a more lenient
607
+ * reading there, so text that only looks like code is not swallowed by an
608
+ * apostrophe or a stray quote.
609
+ */
610
+ export function findCodeEnd(
611
+ src: string,
612
+ start: number,
613
+ { goal = 'expression', stop, stopKey, cache }: FindCodeEndOptions = {},
614
+ ): number {
615
+ const strict: StrictScan = {
616
+ stop,
617
+ malformed: false,
618
+ stopped: false,
619
+ passed: [],
620
+ };
621
+ strict.cache = cache;
622
+ // Where a `{…}` frame ends depends on `stop`
623
+ if (cache && !stop) strict.braces = cache.braces;
624
+ if (cache && (!stop || stopKey !== undefined)) {
625
+ const kindKey = `${goal} ${stop ? `stop ${stopKey}` : '}'}`;
626
+ let checkpoints = cache.checkpoints.get(kindKey);
627
+ if (!checkpoints) cache.checkpoints.set(kindKey, (checkpoints = new Map()));
628
+ strict.checkpoints = checkpoints;
629
+ let parens = cache.parens.get(kindKey);
630
+ if (!parens) cache.parens.set(kindKey, (parens = new Map()));
631
+ strict.parens = parens;
632
+ }
633
+ const ctx: ScanContext = {
634
+ lookahead: true,
635
+ brackets: cache?.brackets ?? new Map(),
636
+ strict,
637
+ };
638
+ const kind = goal === 'statements' ? 'block' : 'expr';
639
+ const outer = frame(kind, stop ? '' : '}', start - 1);
640
+ const end = scan(src, {}, start, 0, outer, ctx);
641
+ if (strict.checkpoints) {
642
+ const ended = stop ? strict.stopped : end < src.length;
643
+ const result = strict.malformed ? MALFORMED : ended ? end : UNCLOSED;
644
+ for (const at of strict.passed) strict.checkpoints.set(at, result);
645
+ }
646
+ if (strict.malformed) return -1;
647
+ if (stop) return strict.stopped ? end : -1;
648
+ return end < src.length ? end : -1;
649
+ }
650
+
651
+ /**
652
+ * Scan from `start` within `outer`. Returns `src.length`, or the index of
653
+ * the `]` closing an `outer` square bracket, or the index just past the
654
+ * backtick closing an `outer` template literal.
655
+ *
656
+ * Nested brackets, template literals and their interpolations are frames
657
+ * on a stack, not recursive calls: however deep the nesting, the scan
658
+ * cannot overflow the call stack.
659
+ */
660
+ function scan(
661
+ src: string,
662
+ handlers: JsLexHandlers,
663
+ start: number,
664
+ nesting: number,
665
+ outer: Frame,
666
+ ctx: ScanContext,
667
+ ): number {
668
+ const frames: Frame[] = [outer];
669
+ const top = () => frames[frames.length - 1]!;
670
+ /**
671
+ * Indices of the open frames closed by `)`, `]` and `}` (above `outer`),
672
+ * innermost last: the frame a closer closes is found without walking the
673
+ * stack.
674
+ */
675
+ const closers: Record<string, number[]> = { ')': [], ']': [], '}': [] };
676
+ const innermost = (c: string) => {
677
+ const list = closers[c]!;
678
+ return list.length ? list[list.length - 1]! : 0;
679
+ };
177
680
  let i = start;
178
681
 
179
682
  let operandNext = true; // an operand (not an operator) comes next
683
+ let stmtNext = outer.kind === 'block'; // a statement starts here
684
+ let arrowBody = false; // an arrow function body starts here
685
+ let keyNext = false; // a property name may come next
180
686
  let word = ''; // identifier/number currently being read
181
- let afterDot = false; // `word` is a property name, never a keyword
687
+ let afterDot = false; // the next word is a property name, never a keyword
688
+ let dots = 0; // length of the current run of `.` tokens
182
689
  let afterHeaderKeyword = false; // last token was if/while/for/with
183
690
  let lastPunct = '';
691
+ let lastKeyword = ''; // the last token, if it was a keyword
184
692
  let lineBreak = false; // a line break since the last token
185
- let braceDepth = 0;
186
- const parenIsHeader: boolean[] = []; // per open `(`: closes a header?
693
+ /** The next `{` at this depth opens a function or class body. */
694
+ let pendingBody: { depth: number; frame: Frame } | undefined;
695
+
696
+ const strict = ctx.strict;
697
+ const braces = strict?.braces;
698
+ const stacks = strict?.checkpoints && strict.cache?.stacks;
699
+ /**
700
+ * With checkpoints, per open frame the id of the stack up to it, found
701
+ * when a checkpoint needs it: the kind and flags of each frame, and the
702
+ * conditional-expression count of each but the innermost (which changes
703
+ * only while it is innermost, and is part of the checkpoint state). How a
704
+ * scan goes on depends on these, not on where the frames opened.
705
+ */
706
+ const stackIds: (number | undefined)[] | undefined = stacks ? [0] : undefined;
707
+
708
+ /** The id of the stack of open frames (`stackIds`). */
709
+ function stackId(): number {
710
+ let k = frames.length - 1;
711
+ while (stackIds![k] === undefined) k--;
712
+ for (k++; k < frames.length; k++) {
713
+ const f = frames[k]!;
714
+ const key =
715
+ `${stackIds![k - 1]} ${frames[k - 1]!.ternary} ${f.kind} ${f.closer}` +
716
+ ` ${+f.header}${+f.operand}${+f.interpolation}`;
717
+ let id = stacks!.get(key);
718
+ if (id === undefined) stacks!.set(key, (id = stacks!.size + 1));
719
+ stackIds![k] = id;
720
+ }
721
+ return stackIds![frames.length - 1]!;
722
+ }
723
+ /** A `{…}` frame result found in `braces`, to skip to. */
724
+ let skipTo: { frame: Frame; end: number } | undefined;
725
+
726
+ const parens = strict?.parens;
727
+ const shareAfter = strict?.cache?.shareAfter ?? SHARE_AFTER;
728
+ /**
729
+ * With `parens`, per open frame and per closer in `CLOSERS`: the lowest
730
+ * frame index the closer looked for its frame at, while this frame or one
731
+ * opened in it was innermost. Below a frame's own index, the code in it
732
+ * depended on what is around it. A frame's entry takes in those of the
733
+ * frames opened in it as they close.
734
+ */
735
+ const lowest: number[][] | undefined = parens
736
+ ? [[Infinity, Infinity, Infinity]]
737
+ : undefined;
738
+ /**
739
+ * How many function or class bodies to come the scan has started, and with
740
+ * `parens`, that count when each open frame opened. Code in a `(…)` or
741
+ * `[…]` that starts one replaces the one to come, and the new one can't
742
+ * follow once the frame is closed: none is to come then.
743
+ */
744
+ let bodyStarts = 0;
745
+ const startsAt: number[] | undefined = parens ? [0] : undefined;
746
+ /** A `(…)` or `[…]` frame result found in `parens`, to skip to. */
747
+ let parenTo: { frame: Frame; result: number } | undefined;
748
+
749
+ /** The closer `c` looked for its frame and found index `k`. */
750
+ function lookedFor(c: (typeof CLOSERS)[number], k: number) {
751
+ if (!lowest) return;
752
+ const entry = lowest[lowest.length - 1]!;
753
+ const n = CLOSERS.indexOf(c);
754
+ if (k < entry[n]!) entry[n] = k;
755
+ }
756
+
757
+ /** Take the entry of the frame at index `k` into the one below it. */
758
+ function foldLowest(k: number) {
759
+ const from = lowest![k]!;
760
+ const into = lowest![k - 1]!;
761
+ for (let n = 0; n < 3; n++) if (from[n]! < into[n]!) into[n] = from[n]!;
762
+ }
763
+
764
+ /** Key of a `(…)` or `[…]` frame result, but for the closers it met. */
765
+ const parenKey = (f: Frame) =>
766
+ (f.open * 3 + (f.closer === ']' ? 2 : +f.header)) * 8;
767
+
768
+ /**
769
+ * Record how the frame at index `k` ends, if it is a `(…)` or `[…]`, its
770
+ * `lowest` entry complete: under the closers that looked below it, which
771
+ * found nothing to close (else the frame was closed with them).
772
+ */
773
+ function recordParen(k: number, result: number) {
774
+ const f = frames[k]!;
775
+ if (k === 0 || (f.closer !== ')' && f.closer !== ']')) return;
776
+ if (f.open - start < shareAfter) return;
777
+ let met = 0;
778
+ for (let n = 0; n < 3; n++) if (lowest![k]![n]! < k) met |= 1 << n;
779
+ parens!.set(parenKey(f) + met, result);
780
+ }
781
+
782
+ /** Record how the open `(…)` and `[…]` frames end: the scan ends in them. */
783
+ function recordOpenParens(result: number) {
784
+ if (!lowest) return;
785
+ for (let k = frames.length - 1; k > 0; k--) {
786
+ recordParen(k, result);
787
+ foldLowest(k);
788
+ }
789
+ }
790
+
791
+ /**
792
+ * How an earlier scan found a `(…)` or `[…]` frame opening here to end,
793
+ * if it did: a result recorded under closers that find nothing to close
794
+ * around the frame here too. The frame is not open yet.
795
+ */
796
+ function knownParen(f: Frame): number | undefined {
797
+ if (!parens || f.open - start < shareAfter) return undefined;
798
+ // Where each closer would look for its frame, and whether it would find
799
+ // none to close (it is stray)
800
+ const at = [
801
+ Math.max(innermost(')'), innermost('}')),
802
+ Math.max(innermost(']'), innermost('}')),
803
+ innermost('}'),
804
+ ];
805
+ let stray = 0;
806
+ for (let n = 0; n < 3; n++) {
807
+ const k = at[n]!;
808
+ if (k === 0 || (n < 2 && frames[k]!.closer !== CLOSERS[n])) {
809
+ stray |= 1 << n;
810
+ }
811
+ }
812
+ const key = parenKey(f);
813
+ for (let met = stray; ; met = (met - 1) & stray) {
814
+ const result = parens.get(key + met);
815
+ if (result !== undefined) {
816
+ // The closers it met look here too
817
+ for (let n = 0; n < 3; n++) {
818
+ if (met & (1 << n)) lookedFor(CLOSERS[n]!, at[n]!);
819
+ }
820
+ return result;
821
+ }
822
+ if (met === 0) return undefined;
823
+ }
824
+ }
825
+
826
+ /** Record how the open frames end, when the scan ends in them. */
827
+ function recordOpen(result: number) {
828
+ if (!braces) return;
829
+ for (let k = 1; k < frames.length; k++) record(frames[k]!, result);
830
+ }
831
+
832
+ /** Record how a frame ends. */
833
+ function record(f: Frame, result: number) {
834
+ const key = braces && frameKey(f);
835
+ if (key !== undefined) braces!.set(key, result);
836
+ }
837
+
838
+ /** How an earlier scan found the frame `f` to end, if it did. */
839
+ const known = (f: Frame) => {
840
+ const key = braces && frameKey(f);
841
+ return key === undefined ? undefined : braces!.get(key);
842
+ };
843
+
844
+ /** End a strict scan at a lexical error. */
845
+ const malformed = () => {
846
+ recordOpen(MALFORMED);
847
+ recordOpenParens(MALFORMED);
848
+ strict!.malformed = true;
849
+ return src.length;
850
+ };
851
+
852
+ function push(f: Frame) {
853
+ stackIds?.push(undefined);
854
+ closers[f.closer]?.push(frames.length);
855
+ frames.push(f);
856
+ lowest?.push([Infinity, Infinity, Infinity]);
857
+ startsAt?.push(bodyStarts);
858
+ }
859
+
860
+ /**
861
+ * Just past a token or a space, line break or comment, but not within a
862
+ * word: the key of the position and the whole scan state there, the open
863
+ * frames included, which decide how the scan goes on (or undefined
864
+ * elsewhere). No word is being read there, and the character before is no
865
+ * word character, so a quote after it starts a string in any scan.
866
+ *
867
+ * Within brackets too, once the scan is long (`SHARE_AFTER`). Scans that
868
+ * never passed such a point in the same state each ran on to the end of
869
+ * the source, so many of them took quadratic time: inside an unclosed `(`
870
+ * in `{(}{(}{(…` (there was no point within brackets), or after the regex
871
+ * literals of `{a'</a </p>…` (there was no point after a space).
872
+ */
873
+ function checkpointKey(): CheckpointKey | undefined {
874
+ const ternary = top().ternary;
875
+ if (
876
+ ternary > 3 ||
877
+ !CHECKPOINT_AFTER.has(src.charAt(i - 1)) ||
878
+ (frames.length > 1 && i - start < shareAfter)
879
+ ) {
880
+ return undefined;
881
+ }
882
+ let state = ternary;
883
+ if (operandNext) state |= 1 << 2;
884
+ if (stmtNext) state |= 1 << 3;
885
+ if (keyNext) state |= 1 << 4;
886
+ if (pendingBody) {
887
+ state |= pendingBody.frame.kind === 'class' ? 1 << 5 : 1 << 6;
888
+ if (pendingBody.frame.operand) state |= 1 << 7;
889
+ }
890
+ if (arrowBody) state |= 1 << 8;
891
+ if (afterDot) state |= 1 << 9;
892
+ if (afterHeaderKeyword) state |= 1 << 10;
893
+ if (lineBreak) state |= 1 << 11;
894
+ if (RESTRICTED_KEYWORDS.has(lastKeyword)) state |= 1 << 12;
895
+ if (lastPunct === '=') state |= 1 << 13;
896
+ // A run of dots: one or two (a spread may follow), three, or more
897
+ if (lastPunct === '.') state |= Math.min(dots, 4) << 14;
898
+ const at = i * (1 << 17) + state;
899
+ if (frames.length === 1) return at;
900
+ // A function or class body to come may open at an outer level
901
+ const body = pendingBody ? frames.length - pendingBody.depth : '';
902
+ return `${at} ${stackId()} ${body}`;
903
+ }
904
+
905
+ /**
906
+ * May a string literal follow the word just read (`word`, or a variable
907
+ * reference) with nothing between? Only after a keyword taking an operand
908
+ * (`typeof'x'`, `case"a"`), `of` in a `for` header, or a modifier before a
909
+ * property name (`static'x'`, `get"y"() {}`); elsewhere the quote is an
910
+ * apostrophe (`don't`), and the text is no JavaScript.
911
+ */
912
+ function stringMayFollowWord(): boolean {
913
+ return (
914
+ OPERAND_KEYWORDS.has(word) ||
915
+ (word === 'of' && top().header) ||
916
+ (keyNext && MODIFIERS.has(word))
917
+ );
918
+ }
919
+
920
+ /** Close the frames from index `k` on. */
921
+ function truncate(k: number) {
922
+ if (lowest) {
923
+ for (let j = frames.length - 1; j >= k; j--) foldLowest(j);
924
+ lowest.length = k;
925
+ startsAt!.length = k;
926
+ }
927
+ frames.length = k;
928
+ if (stackIds) stackIds.length = k;
929
+ for (const list of Object.values(closers)) {
930
+ while (list.length && list[list.length - 1]! >= k) list.pop();
931
+ }
932
+ // A function or class body can't follow once its level is closed
933
+ if (pendingBody && pendingBody.depth > k) pendingBody = undefined;
934
+ }
187
935
 
188
936
  const code = (ch: string, index: number) =>
189
937
  handlers.code?.(ch, index, nesting);
190
938
  const literal = (text: string, index: number) =>
191
939
  handlers.literal?.(text, index, nesting);
940
+ /** Literal text from `from` to `to`, sliced only for a handler. */
941
+ const literalSpan = (from: number, to: number) =>
942
+ handlers.literal?.(src.slice(from, to), from, nesting);
192
943
 
944
+ /** Track the word just read (`i` is the index just past it). */
193
945
  function endWord() {
194
946
  if (!word) return;
195
- operandNext = !afterDot && OPERAND_KEYWORDS.has(word);
196
- afterHeaderKeyword = !afterDot && HEADER_KEYWORDS.has(word);
947
+ // A property name is never a keyword: `a.return`, `{ in: 1 }`
948
+ const keyword = afterDot || keyNext ? '' : word;
949
+ const inOperandPosition = operandNext && !stmtNext;
950
+ operandNext =
951
+ OPERAND_KEYWORDS.has(keyword) ||
952
+ // `for (x of …)`, but `of` is an identifier where an operand goes
953
+ (keyword === 'of' && top().header && !operandNext);
954
+ afterHeaderKeyword = HEADER_KEYWORDS.has(keyword);
955
+ if (keyword === 'function' || keyword === 'class') {
956
+ // An expression where an operand is expected, else a declaration: at
957
+ // a statement start, or after an operand and a line break (ASI)
958
+ const body = frame(keyword === 'class' ? 'class' : 'block', '}');
959
+ body.operand = inOperandPosition;
960
+ pendingBody = { depth: frames.length, frame: body };
961
+ bodyStarts++;
962
+ }
963
+ // `get name()`, `static _x = 1`: the property name is still to come
964
+ keyNext =
965
+ keyNext &&
966
+ MODIFIERS.has(word) &&
967
+ KEY_START_RE.test(src.charAt(skipTrivia(src, i)));
968
+ stmtNext =
969
+ keyword === 'else' ||
970
+ keyword === 'do' ||
971
+ (keyword === 'async' && stmtNext);
972
+ arrowBody = false;
197
973
  afterDot = false;
198
974
  lastPunct = '';
975
+ lastKeyword = keyword;
199
976
  lineBreak = false;
200
977
  word = '';
201
978
  }
202
979
 
203
- /** A string, template or regex literal, or a transient reference, ended. */
980
+ /** A string, template or regex literal, or a variable reference, ended. */
204
981
  function endOperand() {
205
982
  endWord();
206
983
  operandNext = false;
984
+ stmtNext = false;
985
+ arrowBody = false;
986
+ keyNext = false;
207
987
  afterHeaderKeyword = false;
208
988
  afterDot = false;
209
989
  lastPunct = '';
990
+ lastKeyword = '';
210
991
  lineBreak = false;
211
992
  }
212
993
 
213
994
  /** Track an operator token: `operandNext` tells what may follow it. */
214
995
  function endPunct(punct: string, nextIsOperand: boolean) {
215
996
  operandNext = nextIsOperand;
216
- afterDot = punct === '.';
997
+ stmtNext = false;
998
+ arrowBody = false;
999
+ keyNext = false;
1000
+ afterDot = punct === '.' && !nextIsOperand;
217
1001
  afterHeaderKeyword = false;
218
1002
  lastPunct = punct;
1003
+ lastKeyword = '';
219
1004
  lineBreak = false;
220
1005
  }
221
1006
 
1007
+ /**
1008
+ * A line break between tokens. After `return`, `break` or `continue` it
1009
+ * ends the statement (ASI): `return⏎function f() {}` declares `f`.
1010
+ */
1011
+ function lineBreakSeen() {
1012
+ lineBreak = true;
1013
+ if (RESTRICTED_KEYWORDS.has(lastKeyword)) {
1014
+ operandNext = true;
1015
+ stmtNext = true;
1016
+ lastKeyword = '';
1017
+ }
1018
+ }
1019
+
1020
+ function open(f: Frame) {
1021
+ push(f);
1022
+ stmtNext = f.kind === 'block';
1023
+ keyNext = f.kind === 'object' || f.kind === 'class';
1024
+ }
1025
+
1026
+ function openBrace() {
1027
+ let f: Frame;
1028
+ if (pendingBody?.depth === frames.length) {
1029
+ f = pendingBody.frame;
1030
+ pendingBody = undefined;
1031
+ } else if (!operandNext || stmtNext || arrowBody) {
1032
+ f = frame('block', '}');
1033
+ } else {
1034
+ f = frame('object', '}');
1035
+ f.operand = true;
1036
+ }
1037
+ f.open = i;
1038
+ endPunct('{', true);
1039
+ const end = known(f);
1040
+ if (end === undefined) open(f);
1041
+ else skipTo = { frame: f, end };
1042
+ }
1043
+
1044
+ /** State after the `}` closing `closed`, a block or literal. */
1045
+ function afterBrace(closed: Frame) {
1046
+ if (closed.operand) {
1047
+ endPunct('}', false);
1048
+ } else {
1049
+ // A block: a statement (or the next class member) may follow
1050
+ endPunct('}', true);
1051
+ stmtNext = top().kind === 'block';
1052
+ keyNext = top().kind === 'class';
1053
+ }
1054
+ }
1055
+
1056
+ /**
1057
+ * Close the innermost frame that `c` closes; a stray closer, with no such
1058
+ * frame within the innermost braces (a block, an object literal, a class
1059
+ * body or an interpolation), is ignored. So a stray `)` or `]` never
1060
+ * closes the braces around it, and how a `{…}` ends depends only on the
1061
+ * code inside it.
1062
+ */
1063
+ function close(c: (typeof CLOSERS)[number]) {
1064
+ const k = Math.max(innermost(c), innermost('}'));
1065
+ lookedFor(c, k);
1066
+ if (k === 0 || frames[k]!.closer !== c) {
1067
+ endPunct(c, c === '}');
1068
+ return;
1069
+ }
1070
+ const closed = frames[k]!;
1071
+ if (lowest && c !== '}') {
1072
+ // Its entry complete, with those of the frames still open in it
1073
+ for (let j = frames.length - 1; j > k; j--) foldLowest(j);
1074
+ // Whether code in it started a function or class body: none is to come
1075
+ const bodyStarted = bodyStarts !== startsAt![k];
1076
+ recordParen(k, i * 2 + +bodyStarted);
1077
+ }
1078
+ truncate(k);
1079
+ if (c === ']' && !ctx.lookahead) ctx.brackets.set(closed.open, i);
1080
+ if (c === ')') {
1081
+ // `if (…) %x = 1` vs `($n)%3`
1082
+ endPunct(c, closed.header);
1083
+ stmtNext = closed.header;
1084
+ } else if (c === ']') {
1085
+ endPunct(c, false);
1086
+ } else {
1087
+ record(closed, i);
1088
+ afterBrace(closed);
1089
+ }
1090
+ }
1091
+
222
1092
  function trackCode(c: string) {
223
1093
  if (WORD_CHAR_RE.test(c)) {
224
1094
  word += c;
@@ -227,63 +1097,250 @@ export function lexJs(
227
1097
  endWord();
228
1098
  // A line break alone never changes operand/operator position:
229
1099
  // `$x = 5\n%n` continues the expression, as in JavaScript.
230
- if (c === '\n') lineBreak = true;
1100
+ if (LINE_TERMINATOR_RE.test(c)) lineBreakSeen();
231
1101
  if (SPACE_RE.test(c)) return;
232
- if (c === '(') parenIsHeader.push(afterHeaderKeyword);
233
- if (c === '{') braceDepth++;
234
- if (c === '}') braceDepth--;
235
- if (c === ')') {
236
- // `if (…) %x = 1` vs `($n)%3`
237
- endPunct(c, parenIsHeader.pop() ?? false);
238
- } else if (c === '.') {
239
- // Property access, unless it is the spread `...`
240
- endPunct(c, lastPunct === '.');
241
- } else {
242
- // `]` ends an operand; `}` closes a block, so a statement may follow.
243
- endPunct(c, c !== ']');
1102
+ const t = top();
1103
+ switch (c) {
1104
+ case '(':
1105
+ case '[': {
1106
+ const f = frame('expr', c === '(' ? ')' : ']', i);
1107
+ f.header = c === '(' && afterHeaderKeyword;
1108
+ endPunct(c, true);
1109
+ const result = knownParen(f);
1110
+ if (result === undefined) open(f);
1111
+ else parenTo = { frame: f, result };
1112
+ break;
1113
+ }
1114
+ case '{':
1115
+ openBrace();
1116
+ break;
1117
+ case ')':
1118
+ case ']':
1119
+ case '}':
1120
+ close(c);
1121
+ break;
1122
+ case '.':
1123
+ // Property access, unless it is the spread `...`
1124
+ dots = lastPunct === '.' ? dots + 1 : 1;
1125
+ endPunct(c, dots === 3);
1126
+ break;
1127
+ case ';':
1128
+ endPunct(c, true);
1129
+ stmtNext = t.kind === 'block';
1130
+ keyNext = t.kind === 'class';
1131
+ break;
1132
+ case ',':
1133
+ endPunct(c, true);
1134
+ keyNext = t.kind === 'object';
1135
+ break;
1136
+ case '?':
1137
+ endPunct(c, true);
1138
+ t.ternary++;
1139
+ break;
1140
+ case ':':
1141
+ endPunct(c, true);
1142
+ // Not a conditional's `:`: an object literal value, or a statement
1143
+ // after a `case`, `default` or label
1144
+ if (t.ternary > 0) t.ternary--;
1145
+ else stmtNext = t.kind === 'block';
1146
+ break;
1147
+ case '#':
1148
+ // A private name: `this.#_x`
1149
+ endPunct(c, false);
1150
+ afterDot = true;
1151
+ break;
1152
+ case '*': {
1153
+ // A generator method: `{ *_gen() {} }`
1154
+ const key = keyNext;
1155
+ endPunct(c, true);
1156
+ keyNext = key;
1157
+ break;
1158
+ }
1159
+ case '>': {
1160
+ // `=>`: the body may be a block
1161
+ const arrow = lastPunct === '=';
1162
+ endPunct(c, true);
1163
+ arrowBody = arrow;
1164
+ break;
1165
+ }
1166
+ default:
1167
+ endPunct(c, true);
244
1168
  }
245
1169
  }
246
1170
 
247
1171
  while (i < src.length) {
248
1172
  const ch = src.charAt(i);
249
1173
 
1174
+ // At the top level just past a `}`: go on as a scan that passed here in
1175
+ // the same state did
1176
+ const checkpoints = strict?.checkpoints;
1177
+ const key = checkpoints && checkpointKey();
1178
+ if (key !== undefined) {
1179
+ const known = checkpoints!.get(key);
1180
+ if (known === undefined) {
1181
+ strict!.passed.push(key);
1182
+ } else if (frames.length > 1 && known < 0) {
1183
+ // The frames open here may close before the error or the end of the
1184
+ // source, so their results are not known: none is recorded
1185
+ strict!.malformed = known === MALFORMED;
1186
+ return src.length;
1187
+ } else if (known === MALFORMED) {
1188
+ return malformed();
1189
+ } else if (known === UNCLOSED) {
1190
+ i = src.length;
1191
+ break;
1192
+ } else {
1193
+ strict!.stopped = true;
1194
+ return known;
1195
+ }
1196
+ }
1197
+
1198
+ // Template literal text: escapes, the closing backtick, interpolations
1199
+ if (top().kind === 'template') {
1200
+ if (ch === '\\') {
1201
+ literal(src.slice(i, i + 2), i);
1202
+ i += 2;
1203
+ } else if (ch === '`') {
1204
+ literal(ch, i);
1205
+ if (frames.length === 1) return i + 1; // the end of `lexTemplate`
1206
+ record(top(), i);
1207
+ i++;
1208
+ truncate(frames.length - 1);
1209
+ endOperand();
1210
+ } else if (ch === '$' && src.charAt(i + 1) === '{') {
1211
+ const f = frame('expr', '}', i);
1212
+ f.interpolation = true;
1213
+ const end = known(f);
1214
+ if (end === MALFORMED) return malformed();
1215
+ if (end === UNCLOSED) {
1216
+ i = src.length;
1217
+ break;
1218
+ }
1219
+ if (end !== undefined) {
1220
+ // An interpolation an earlier scan lexed: on with the text after it
1221
+ i = end + 1;
1222
+ continue;
1223
+ }
1224
+ literal('${', i);
1225
+ i += 2;
1226
+ nesting++;
1227
+ endPunct('{', true);
1228
+ open(f);
1229
+ } else {
1230
+ literal(ch, i);
1231
+ i++;
1232
+ }
1233
+ continue;
1234
+ }
1235
+
250
1236
  // String literal — skip entirely
251
1237
  if (ch === '"' || ch === "'") {
252
- const { end } = scanStringLiteral(src, i);
253
- literal(src.slice(i, end), i);
1238
+ if (strict && i > start && WORD_CHAR_RE.test(src.charAt(i - 1))) {
1239
+ if (!stringMayFollowWord()) return malformed();
1240
+ }
1241
+ endWord();
1242
+ const { end, closed } = scanStringLiteral(src, i);
1243
+ if (strict && !closed) return malformed();
1244
+ literalSpan(i, end);
254
1245
  i = end;
255
1246
  endOperand();
256
1247
  continue;
257
1248
  }
258
1249
 
259
1250
  if (ch === '`') {
260
- i = lexTemplate(src, i, handlers, nesting);
261
- endOperand();
1251
+ endWord();
1252
+ const f = frame('template', '`', i);
1253
+ const end = known(f);
1254
+ if (end === MALFORMED) return malformed();
1255
+ if (end === UNCLOSED) {
1256
+ i = src.length;
1257
+ break;
1258
+ }
1259
+ if (end !== undefined) {
1260
+ // A template literal an earlier scan lexed: on after it
1261
+ i = end + 1;
1262
+ endOperand();
1263
+ continue;
1264
+ }
1265
+ literal(ch, i);
1266
+ push(f);
1267
+ i++;
262
1268
  continue;
263
1269
  }
264
1270
 
1271
+ if (ch === '{' && strict?.stop?.(i)) {
1272
+ recordOpenParens(STOPPED - i);
1273
+ strict.stopped = true;
1274
+ return i;
1275
+ }
1276
+
1277
+ // End of a template literal interpolation: the innermost `}` closer
1278
+ if (ch === '}') {
1279
+ const k = innermost('}');
1280
+ lookedFor('}', k);
1281
+ if (frames[k]!.interpolation) {
1282
+ endWord();
1283
+ record(frames[k]!, i);
1284
+ truncate(k);
1285
+ nesting--;
1286
+ literal(ch, i);
1287
+ i++;
1288
+ continue;
1289
+ }
1290
+ }
1291
+
265
1292
  if (ch === '/') {
266
1293
  endWord();
267
1294
  const next = src.charAt(i + 1);
268
1295
  // Comment — skip entirely; it is not a token
269
1296
  if (next === '/' || next === '*') {
270
- const end = skipComment(src, i);
271
- const comment = src.slice(i, end);
272
- literal(comment, i);
273
- if (comment.includes('\n')) lineBreak = true;
1297
+ let end = findCommentEnd(src, i, strict?.cache);
1298
+ if (end < 0) {
1299
+ if (strict) return malformed();
1300
+ end = src.length;
1301
+ }
1302
+ literalSpan(i, end);
1303
+ if (next === '*' && lineBreakIn(src, i, end, strict?.cache)) {
1304
+ lineBreakSeen();
1305
+ }
274
1306
  i = end;
275
1307
  continue;
276
1308
  }
277
1309
  // Regex literal — only where an operand is expected
278
1310
  if (operandNext) {
279
- const end = skipRegex(src, i);
280
- literal(src.slice(i, end), i);
1311
+ const { end, closed } = scanRegex(src, i, strict?.cache);
1312
+ if (strict && !closed) return malformed();
1313
+ literalSpan(i, end);
281
1314
  i = end;
282
1315
  endOperand();
283
1316
  continue;
284
1317
  }
285
1318
  }
286
1319
 
1320
+ // In a class body, a line break after a complete member starts the next
1321
+ // one (ASI): `_x = 1⏎_y = 2`, but `_x = a⏎instanceof B` continues it.
1322
+ if (!word && lineBreak && !operandNext && top().kind === 'class') {
1323
+ IDENT_RE.lastIndex = i;
1324
+ const next = IDENT_RE.exec(src)?.[0];
1325
+ if (next && next !== 'in' && next !== 'instanceof') keyNext = true;
1326
+ }
1327
+
1328
+ // `$name`, `_name` or `@name` reference where an identifier starts (`@`
1329
+ // is no identifier character, so `typeof@x` holds one). The whole
1330
+ // identifier must be the sigil and a name: `$a$b` and `$café` are
1331
+ // identifiers of their own.
1332
+ if (ch === '@' || ((ch === '$' || ch === '_') && !word)) {
1333
+ endWord();
1334
+ IDENT_RE.lastIndex = i + 1;
1335
+ const name = IDENT_RE.exec(src)?.[0] ?? '';
1336
+ if (VAR_NAME_RE.test(name) && !isPropertyName(ch, i + 1 + name.length)) {
1337
+ handlers.variable?.(ch, name, i, nesting);
1338
+ i += 1 + name.length;
1339
+ endOperand();
1340
+ continue;
1341
+ }
1342
+ }
1343
+
287
1344
  // Transient reference — where an operand is expected, or as the target
288
1345
  // of an assignment starting a line: `$x = 5\n%a = 1` would otherwise be
289
1346
  // the invalid assignment `5 % a = 1`.
@@ -292,10 +1349,13 @@ export function lexJs(
292
1349
  TRANS_NAME_RE.lastIndex = i + 1;
293
1350
  const name = TRANS_NAME_RE.exec(src)?.[0];
294
1351
  if (name) {
295
- ASSIGNMENT_RE.lastIndex = i + 1 + name.length;
296
- if (operandNext || (lineBreak && ASSIGNMENT_RE.test(src))) {
297
- handlers.transient?.(name, i, nesting);
298
- i += 1 + name.length;
1352
+ const end = i + 1 + name.length;
1353
+ if (
1354
+ operandNext ||
1355
+ (lineBreak && ctx.lookahead && assignmentFollows(src, end, ctx))
1356
+ ) {
1357
+ handlers.variable?.('%', name, i, nesting);
1358
+ i = end;
299
1359
  endOperand();
300
1360
  continue;
301
1361
  }
@@ -314,13 +1374,87 @@ export function lexJs(
314
1374
  continue;
315
1375
  }
316
1376
 
317
- // End of a template interpolation
318
- if (ch === '}' && interpolation && braceDepth === 0) break;
1377
+ // `??` and optional chaining `?.` (but `a?.5:1` is a conditional)
1378
+ if (ch === '?') {
1379
+ const next = src.charAt(i + 1);
1380
+ if (next === '?' || (next === '.' && !/\d/.test(src.charAt(i + 2)))) {
1381
+ endWord();
1382
+ code(ch, i);
1383
+ code(next, i + 1);
1384
+ i += 2;
1385
+ dots = 1;
1386
+ endPunct(next, next === '?');
1387
+ continue;
1388
+ }
1389
+ }
1390
+
1391
+ // End of an `outer` square bracket
1392
+ if (ch === outer.closer && frames.length === 1) break;
319
1393
 
320
1394
  // Regular code character
321
1395
  trackCode(ch);
1396
+ if (skipTo) {
1397
+ // A `{…}` frame an earlier scan lexed: continue after it
1398
+ const { frame: skipped, end } = skipTo;
1399
+ skipTo = undefined;
1400
+ if (end === MALFORMED) return malformed();
1401
+ if (end === UNCLOSED) {
1402
+ i = src.length;
1403
+ break;
1404
+ }
1405
+ i = end;
1406
+ afterBrace(skipped);
1407
+ i++;
1408
+ continue;
1409
+ }
1410
+ if (parenTo) {
1411
+ // A `(…)` or `[…]` frame an earlier scan lexed: continue after it
1412
+ const { frame: skipped, result } = parenTo;
1413
+ parenTo = undefined;
1414
+ if (result === MALFORMED) return malformed();
1415
+ if (result === UNCLOSED) {
1416
+ i = src.length;
1417
+ break;
1418
+ }
1419
+ if (result <= STOPPED) {
1420
+ recordOpenParens(result);
1421
+ strict!.stopped = true;
1422
+ return STOPPED - result;
1423
+ }
1424
+ i = Math.floor(result / 2);
1425
+ if (result % 2 === 1) pendingBody = undefined;
1426
+ if (skipped.closer === ')') {
1427
+ endPunct(')', skipped.header);
1428
+ stmtNext = skipped.header;
1429
+ } else {
1430
+ endPunct(']', false);
1431
+ }
1432
+ i++;
1433
+ continue;
1434
+ }
322
1435
  code(ch, i);
323
1436
  i++;
324
1437
  }
325
- return i;
1438
+ if (i >= src.length) {
1439
+ recordOpen(UNCLOSED);
1440
+ recordOpenParens(UNCLOSED);
1441
+ }
1442
+ if (!ctx.lookahead) {
1443
+ // Brackets left open here close at `i`: the end of `outer` or of `src`
1444
+ for (const f of frames) if (f.closer === ']') ctx.brackets.set(f.open, i);
1445
+ }
1446
+ return Math.min(i, src.length);
1447
+
1448
+ /**
1449
+ * Is the `$`/`_` word ending at `end` a property name: after `.`, a key
1450
+ * in an object literal (`{ _id: 1 }`, `{ _m() {} }`) or a member name in
1451
+ * a class body?
1452
+ */
1453
+ function isPropertyName(sigil: string, end: number): boolean {
1454
+ if (afterDot) return true;
1455
+ if (!keyNext || sigil === '@') return false;
1456
+ if (top().kind === 'class') return true;
1457
+ const next = src.charAt(skipTrivia(src, end));
1458
+ return next === ':' || next === '(';
1459
+ }
326
1460
  }