@rohal12/spindle 0.51.4 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/dist/pkg/format.js +1 -1
  2. package/dist/pkg/headless.js +4313 -1640
  3. package/dist/pkg/macro-registry.json +7 -7
  4. package/dist/pkg/story-variables.js +1636 -177
  5. package/package.json +5 -2
  6. package/src/automation/runner.ts +2 -1
  7. package/src/class-registry.ts +214 -103
  8. package/src/components/Passage.tsx +2 -2
  9. package/src/components/PassageDialog.tsx +2 -5
  10. package/src/components/StoryInterface.tsx +2 -4
  11. package/src/components/macros/Button.tsx +5 -31
  12. package/src/components/macros/Checkbox.tsx +7 -4
  13. package/src/components/macros/Computed.tsx +19 -13
  14. package/src/components/macros/For.tsx +29 -3
  15. package/src/components/macros/If.tsx +8 -0
  16. package/src/components/macros/Include.tsx +7 -6
  17. package/src/components/macros/MacroError.tsx +2 -1
  18. package/src/components/macros/MacroLink.tsx +12 -45
  19. package/src/components/macros/Meter.tsx +11 -3
  20. package/src/components/macros/Nobr.tsx +1 -0
  21. package/src/components/macros/PassageDisplay.tsx +3 -0
  22. package/src/components/macros/Print.tsx +4 -0
  23. package/src/components/macros/Radiobutton.tsx +5 -2
  24. package/src/components/macros/SaveManager.tsx +25 -8
  25. package/src/components/macros/Span.tsx +1 -0
  26. package/src/components/macros/StoryTitle.tsx +1 -0
  27. package/src/components/macros/Switch.tsx +13 -0
  28. package/src/components/macros/Unset.tsx +30 -10
  29. package/src/components/macros/VarDisplay.tsx +21 -4
  30. package/src/components/macros/Widget.tsx +20 -1
  31. package/src/components/macros/WidgetInvocation.tsx +17 -75
  32. package/src/components/macros/arg-utils.ts +107 -1
  33. package/src/components/macros/detached-body.tsx +68 -0
  34. package/src/components/macros/option-utils.ts +3 -2
  35. package/src/define-macro.ts +32 -5
  36. package/src/execute-mutation.ts +270 -67
  37. package/src/expression.ts +86 -55
  38. package/src/hooks/use-action.ts +18 -3
  39. package/src/hooks/use-interpolate.ts +36 -5
  40. package/src/index.tsx +10 -1
  41. package/src/interpolation.ts +394 -96
  42. package/src/js-lexer.ts +1231 -97
  43. package/src/markup/code-attributes.ts +64 -0
  44. package/src/markup/markdown.ts +188 -9
  45. package/src/markup/render.tsx +430 -49
  46. package/src/markup/tokenizer.ts +578 -110
  47. package/src/prng.ts +8 -8
  48. package/src/registry.ts +35 -0
  49. package/src/saves/save-manager.ts +339 -158
  50. package/src/saves/storage.ts +16 -7
  51. package/src/saves/types.ts +2 -1
  52. package/src/store.ts +521 -154
  53. package/src/story-api.ts +31 -67
  54. package/src/story-init.ts +1 -1
  55. package/src/story-variables.ts +98 -102
  56. package/src/triggers.ts +6 -5
  57. package/src/utils/error-message.ts +12 -0
  58. package/src/utils/live-locals.ts +10 -3
  59. package/src/utils/namespace.ts +71 -0
  60. package/src/utils/object-path.ts +99 -14
  61. package/src/utils/stable-key.ts +14 -9
  62. package/src/widgets/widget-registry.ts +9 -0
  63. package/types/index.d.ts +43 -7
  64. package/types/tooling.d.ts +1 -0
@@ -1,3 +1,6 @@
1
+ import { createJsScanCache, findCodeEnd, type JsScanCache } from '../js-lexer';
2
+ import { isCodeAttribute } from './code-attributes';
3
+
1
4
  export interface TextToken {
2
5
  type: 'text';
3
6
  value: string;
@@ -84,6 +87,17 @@ const HTML_VOID_TAGS = new Set([
84
87
  'wbr',
85
88
  ]);
86
89
 
90
+ /** Variable sigils: story ($), temporary (_), local (@), transient (%). */
91
+ const SIGIL_CHARS = new Set(['$', '_', '@', '%']);
92
+
93
+ /**
94
+ * Characters other than a sigil that open an expression after `{`:
95
+ * `{(Math.max($a, 0))}`, `{!$done}`. Other characters that can start an
96
+ * expression (quotes, digits, `[`, `-`) are left out, because braces around
97
+ * them are common as literal text (JSON, regex quantifiers, `{[[link]]}`).
98
+ */
99
+ const EXPRESSION_START = new Set(['(', '!']);
100
+
87
101
  /** Macros whose body is JavaScript source, kept verbatim instead of tokenized. */
88
102
  const RAW_BODY_MACROS = new Set(['do']);
89
103
 
@@ -172,11 +186,8 @@ function parseSelectors(
172
186
  if (/[a-zA-Z0-9_-]/.test(input[i]!)) {
173
187
  name += input[i];
174
188
  i++;
175
- } else if (
176
- input[i] === '{' &&
177
- (input[i + 1] === '$' || input[i + 1] === '_' || input[i + 1] === '@')
178
- ) {
179
- // Consume interpolation: {$var}, {_var}, {@var} (with optional dot paths)
189
+ } else if (input[i] === '{' && SIGIL_CHARS.has(input[i + 1]!)) {
190
+ // Consume interpolation: {$var}, {_var}, {@var}, {%var} (with optional dot paths)
180
191
  const braceStart = i;
181
192
  i += 2; // skip { and prefix
182
193
  while (i < input.length && /[\w.]/.test(input[i]!)) i++;
@@ -211,12 +222,69 @@ function parseSelectors(
211
222
  function parseHtmlAttributes(
212
223
  input: string,
213
224
  j: number,
225
+ memo: ScanMemo,
214
226
  ): { attributes: Record<string, string>; endIdx: number } {
215
227
  const attributes: Record<string, string> = {};
228
+ // As in HTML, the first of attributes with the same (case-insensitive)
229
+ // name wins. Defined as own properties so `__proto__` is kept too.
230
+ const seen = new Set<string>();
231
+ const endIdx = scanAttributes(input, j, memo, (name, value) => {
232
+ const lower = name.toLowerCase();
233
+ if (seen.has(lower)) return;
234
+ seen.add(lower);
235
+ Object.defineProperty(attributes, name, {
236
+ value,
237
+ enumerable: true,
238
+ writable: true,
239
+ configurable: true,
240
+ });
241
+ });
242
+ return { attributes, endIdx };
243
+ }
216
244
 
245
+ /**
246
+ * Index just past the attributes of a tag that start at `j` (after its
247
+ * name), where its `>` or `/>` goes. Pass the same memo to repeated scans of
248
+ * one input to share their work.
249
+ */
250
+ export function scanTagAttributes(
251
+ input: string,
252
+ j: number,
253
+ memo: ScanMemo = createScanMemo(),
254
+ ): number {
255
+ return scanAttributes(input, j, memo);
256
+ }
257
+
258
+ /**
259
+ * Scan the attributes of a tag from position j, handing each to `add` when
260
+ * given. Returns the position after the last attribute.
261
+ *
262
+ * Where the attributes end depends only on where an attribute starts, so
263
+ * without `add` the result is recorded for every attribute start passed (in
264
+ * `memo.tag`) and a recorded one is used instead of scanning on. An unquoted
265
+ * value takes in a `<`, so in `<a x=<a x=<a x=…` each opener's scan would
266
+ * otherwise run over all the openers after it, taking quadratic time. A tag
267
+ * whose attributes end in its `>` is consumed, so collecting the attributes
268
+ * of those (with `add`, unrecorded) reads each character once.
269
+ */
270
+ function scanAttributes(
271
+ input: string,
272
+ j: number,
273
+ memo: ScanMemo,
274
+ add?: (name: string, value: string) => void,
275
+ ): number {
276
+ const passed: number[] = [];
217
277
  while (j < input.length) {
218
278
  // Skip whitespace
219
279
  while (j < input.length && /\s/.test(input[j]!)) j++;
280
+ if (!add) {
281
+ const known = memo.tag.get(j);
282
+ if (known !== undefined) {
283
+ j = known;
284
+ break;
285
+ }
286
+ passed.push(j);
287
+ }
220
288
  // End of tag?
221
289
  if (
222
290
  j >= input.length ||
@@ -242,32 +310,256 @@ function parseHtmlAttributes(
242
310
  const quote = input[j]!;
243
311
  j++; // skip opening quote
244
312
  const valStart = j;
245
- while (j < input.length) {
246
- if (input[j] === '{') {
247
- // Skip a whole {…} interpolation so quotes inside it don't end the value
248
- const closeIdx = scanBalancedBrace(input, j + 1);
249
- if (closeIdx !== -1) {
250
- j = closeIdx + 1;
251
- continue;
252
- }
253
- } else if (input[j] === quote) break;
254
- j++;
255
- }
256
- attributes[attrName] = input.slice(valStart, j);
313
+ j = scanQuotedValue(input, j, quote, isCodeAttribute(attrName), memo);
314
+ add?.(attrName, input.slice(valStart, j));
257
315
  if (j < input.length) j++; // skip closing quote
258
316
  } else {
259
- // Unquoted value
317
+ // Unquoted value, up to whitespace or `>`. It takes in a `<`, so in
318
+ // `<a:=<a:=<a:=…` it holds every tag after it: the last run of such
319
+ // characters found is kept, as each tag's value ends where it does.
260
320
  const valStart = j;
261
- while (j < input.length && /[^\s>]/.test(input[j]!)) j++;
262
- attributes[attrName] = input.slice(valStart, j);
321
+ const run = memo.unquoted;
322
+ if (j < run.from || j > run.to) {
323
+ run.from = j;
324
+ while (j < input.length && /[^\s>]/.test(input[j]!)) j++;
325
+ run.to = j;
326
+ }
327
+ j = run.to;
328
+ add?.(attrName, input.slice(valStart, j));
263
329
  }
264
330
  } else {
265
331
  // Boolean attribute
266
- attributes[attrName] = '';
332
+ add?.(attrName, '');
267
333
  }
268
334
  }
335
+ for (const at of passed) memo.tag.set(at, j);
336
+ return j;
337
+ }
269
338
 
270
- return { attributes, endIdx: j };
339
+ /**
340
+ * Index of the quote ending the attribute value that starts at `j`, or the
341
+ * end of the input. A `{…}` interpolation in it is skipped whole, so quotes
342
+ * inside it don't end the value. A code attribute's value is no markup
343
+ * (`isCodeAttribute`): only `{` and a sigil open a reference, other braces
344
+ * and backslashes are text, as `splitSigilTemplate` reads them.
345
+ *
346
+ * The value reads the same from just past an interpolation, however the scan
347
+ * got there, so the end is recorded there (in `memo.value`), and a recorded
348
+ * one is used. An unclosed value that skips interpolations to the end of
349
+ * the input (`<a x="}<a x={<a x="}…`) is then read once, not once per tag.
350
+ */
351
+ function scanQuotedValue(
352
+ input: string,
353
+ j: number,
354
+ quote: string,
355
+ code: boolean,
356
+ memo: ScanMemo,
357
+ ): number {
358
+ const variant = (quote === '"' ? 0 : 2) + (code ? 1 : 0);
359
+ const passed: number[] = [];
360
+ let checkpoint = true;
361
+ while (j < input.length) {
362
+ if (checkpoint) {
363
+ const known = memo.value.get(j * 4 + variant);
364
+ if (known !== undefined) {
365
+ j = known;
366
+ break;
367
+ }
368
+ passed.push(j * 4 + variant);
369
+ checkpoint = false;
370
+ }
371
+ if (!code && input[j] === '\\') {
372
+ // A brace after an odd backslash run is escaped (`\{`) and opens
373
+ // no interpolation, as in passage text.
374
+ let k = j + 1;
375
+ while (input[k] === '\\') k++;
376
+ const brace = input[k] === '{' || input[k] === '}';
377
+ j = brace && (k - j) % 2 === 1 ? k + 1 : k;
378
+ continue;
379
+ }
380
+ if (input[j] === '{') {
381
+ const closeIdx = !code
382
+ ? scanBlockClose(input, j, memo)
383
+ : SIGIL_CHARS.has(input[j + 1]!)
384
+ ? scanBalancedBrace(input, j + 1, memo)
385
+ : -1;
386
+ if (closeIdx !== -1) {
387
+ j = closeIdx + 1;
388
+ checkpoint = true;
389
+ continue;
390
+ }
391
+ } else if (input[j] === quote) break;
392
+ j++;
393
+ }
394
+ for (const at of passed) memo.value.set(at, j);
395
+ return j;
396
+ }
397
+
398
+ /**
399
+ * Results of the brace and template scans of one input, shared by repeated
400
+ * scans of it. A scan's result depends only on the input and where it
401
+ * starts, so caching it is exact. Without the cache, each unclosed template
402
+ * literal was scanned once as a template and again as plain text, so nested
403
+ * unclosed ones (`` {$a`${$a`${… ``) took time exponential in their depth;
404
+ * and each unclosed `{` was scanned to the end of the input, so many of them
405
+ * took quadratic time. A memo must only be reused for scans of the same input
406
+ * string.
407
+ */
408
+ export interface ScanMemo {
409
+ /** `scanBalancedBrace` results, by lenient start. */
410
+ code: Map<number, number>;
411
+ /** JavaScript lexer results. */
412
+ js: JsScanCache;
413
+ /** Lenient brace scan results. */
414
+ brace: Map<number, number>;
415
+ /** Lenient template literal scan results. */
416
+ template: Map<number, number>;
417
+ /** Lenient template literal scan results, by a point in its text. */
418
+ templateText: Map<number, number>;
419
+ /** The last macro name run found: no whitespace or } in [from, to). */
420
+ name: { from: number; to: number };
421
+ /** The last unquoted attribute value run: no whitespace or > in [from, to). */
422
+ unquoted: { from: number; to: number };
423
+ /** Link scan results (`scanLinkClose`). */
424
+ link: Map<number, number>;
425
+ /** Where the attributes of a tag end, by attribute start (`scanAttributes`). */
426
+ tag: Map<number, number>;
427
+ /** Where quoted attribute values end, by checkpoint (`scanQuotedValue`). */
428
+ value: Map<number, number>;
429
+ /**
430
+ * The last search for a raw-body closer, by macro name: the first one
431
+ * from `from` on is at `at` (-1 for none).
432
+ */
433
+ rawClose: Map<string, { from: number; at: number }>;
434
+ }
435
+
436
+ export function createScanMemo(): ScanMemo {
437
+ return {
438
+ code: new Map(),
439
+ js: createJsScanCache(),
440
+ brace: new Map(),
441
+ template: new Map(),
442
+ templateText: new Map(),
443
+ name: { from: 0, to: -1 },
444
+ unquoted: { from: 0, to: -1 },
445
+ link: new Map(),
446
+ tag: new Map(),
447
+ value: new Map(),
448
+ rawClose: new Map(),
449
+ };
450
+ }
451
+
452
+ /**
453
+ * Find the } closing the code that starts at position i: a `{$…}`
454
+ * expression from its sigil on, an attribute interpolation. The code is
455
+ * lexed as JavaScript, so only a } in code outside the brackets it opened
456
+ * counts — not one in a string, template or regex literal or a comment —
457
+ * and `/` after an operand divides.
458
+ *
459
+ * Text that is not well-formed JavaScript (an apostrophe, an unterminated
460
+ * literal or comment, unbalanced brackets) is scanned leniently instead, as
461
+ * prose-like macro arguments are: braces count except inside string and
462
+ * template literals, and a quote that can't start a string (apostrophe,
463
+ * escaped, not closed on its line) is text.
464
+ *
465
+ * Returns the index of the closing } or -1 if there is none. Pass the same
466
+ * memo to repeated scans of one input to share their work.
467
+ */
468
+ export function scanBalancedBrace(
469
+ input: string,
470
+ i: number,
471
+ memo: ScanMemo = createScanMemo(),
472
+ ): number {
473
+ return scanClose(input, i, i, memo);
474
+ }
475
+
476
+ /**
477
+ * Index of the } closing the macro whose content (name, then arguments)
478
+ * starts at `contentStart`, or -1. The name runs up to whitespace or the }
479
+ * (as `parseMacroContent` reads it); the arguments after it are code. The
480
+ * lenient scan covers the whole content, as it always has.
481
+ */
482
+ function scanMacroClose(
483
+ input: string,
484
+ contentStart: number,
485
+ memo: ScanMemo,
486
+ ): number {
487
+ const run = memo.name;
488
+ if (contentStart < run.from || contentStart > run.to) {
489
+ let k = contentStart;
490
+ while (k < input.length && input[k] !== '}' && !/\s/.test(input[k]!)) k++;
491
+ run.from = contentStart;
492
+ run.to = k;
493
+ }
494
+ return scanClose(input, run.to, contentStart, memo);
495
+ }
496
+
497
+ /**
498
+ * Index of the } closing the `{…}` block opened at `open`, read as passage
499
+ * text reads it: a macro (`{name args}`, `{/name}`) has its arguments lexed
500
+ * after its name, an expression (`{$…}`, `{(…)}`, `{!…}`) from its first
501
+ * character, after any `.class#id` selectors. Any other block is scanned as
502
+ * code from just past the {. Returns -1 if it is unclosed.
503
+ */
504
+ function scanBlockClose(input: string, open: number, memo: ScanMemo): number {
505
+ let at = open + 1;
506
+ const c = input[at];
507
+ if (c === '.' || c === '#') {
508
+ at = parseSelectors(input, at).endIdx;
509
+ if (input[at] === ' ') at++;
510
+ }
511
+ const first = input[at];
512
+ if (first !== undefined && (first === '/' || /[a-zA-Z]/.test(first))) {
513
+ return scanMacroClose(input, at, memo);
514
+ }
515
+ return scanBalancedBrace(input, at, memo);
516
+ }
517
+
518
+ /** Lex the code from `codeStart`, else scan leniently from `lenientStart`. */
519
+ function scanClose(
520
+ input: string,
521
+ codeStart: number,
522
+ lenientStart: number,
523
+ memo: ScanMemo,
524
+ ): number {
525
+ let end = memo.code.get(lenientStart);
526
+ if (end === undefined) {
527
+ end = findCodeEnd(input, codeStart, { cache: memo.js });
528
+ if (end === -1) end = scanBraceLenient(input, lenientStart, memo);
529
+ memo.code.set(lenientStart, end);
530
+ }
531
+ return end;
532
+ }
533
+
534
+ /**
535
+ * Index of the ]] closing the link whose text starts at `from`, or -1. A
536
+ * [[ inside opens a nested pair. Each nested [[ starts a scan that reads the
537
+ * rest the same way, so its result is recorded too: unclosed [[s don't each
538
+ * scan to the end of the input.
539
+ */
540
+ function scanLinkClose(input: string, from: number, memo: ScanMemo): number {
541
+ const known = memo.link.get(from);
542
+ if (known !== undefined) return known;
543
+ const opens = [from];
544
+ let i = from;
545
+ while (i < input.length) {
546
+ if (input[i] === '[' && input[i + 1] === '[') {
547
+ i += 2;
548
+ const inner = memo.link.get(i);
549
+ if (inner === undefined) opens.push(i);
550
+ else if (inner === -1)
551
+ break; // nor does this one close
552
+ else i = inner + 2;
553
+ } else if (input[i] === ']' && input[i + 1] === ']') {
554
+ memo.link.set(opens.pop()!, i);
555
+ if (!opens.length) return i;
556
+ i += 2;
557
+ } else {
558
+ i++;
559
+ }
560
+ }
561
+ for (const open of opens) memo.link.set(open, -1);
562
+ return -1;
271
563
  }
272
564
 
273
565
  /**
@@ -288,76 +580,188 @@ function skipQuoted(input: string, i: number): number {
288
580
  return -1;
289
581
  }
290
582
 
291
- /**
292
- * Skip a `…` template literal opening at i, including ${…} parts.
293
- * Returns the index just past the closing backtick, or -1 if unclosed.
294
- */
295
- function skipTemplate(input: string, i: number): number {
296
- let j = i + 1;
297
- while (j < input.length) {
298
- const c = input[j];
299
- if (c === '\\') {
300
- j += 2;
301
- } else if (c === '`') {
302
- return j + 1;
303
- } else if (c === '$' && input[j + 1] === '{') {
304
- const closeIdx = scanBalancedBrace(input, j + 2);
305
- if (closeIdx === -1) return -1;
306
- j = closeIdx + 1;
307
- } else {
308
- j++;
309
- }
310
- }
311
- return -1;
312
- }
313
-
314
583
  /**
315
584
  * A quote directly after a letter/digit is an apostrophe (don't), not a
316
585
  * string; after a backslash it is an escaped attribute delimiter (\").
317
586
  */
318
587
  const NON_STRING_QUOTE_PREFIX = /[\p{L}\p{N}_\\]/u;
319
588
 
589
+ /**
590
+ * A brace or template literal scan in progress: braces from `start` (just
591
+ * past a {) to their closing }, or a template literal from its backtick at
592
+ * `start` to the closing one.
593
+ */
594
+ interface LenientScan {
595
+ template: boolean;
596
+ start: number;
597
+ /**
598
+ * For braces, per brace depth from the outermost: the checkpoints at that
599
+ * depth whose scans end where the depth does.
600
+ */
601
+ levels: number[][];
602
+ /**
603
+ * For a template literal, the points in its text it passed (just past its
604
+ * backtick, an escape or an interpolation), whose scans end where it does.
605
+ */
606
+ passed?: number[];
607
+ }
608
+
320
609
  /**
321
610
  * Scan for the balanced closing } starting at position i (just past the {).
322
611
  * Braces inside string and template literals are ignored. A quote that
323
- * can't start a string (apostrophe, escaped, unterminated) counts as text.
612
+ * can't start a string (apostrophe, escaped, unterminated) counts as text,
613
+ * and so does a backtick without a closing one.
324
614
  * Returns the index of the closing } or -1 if unbalanced.
615
+ *
616
+ * Template literals and their ${…} parts are scans on a stack, not
617
+ * recursive calls, so deep nesting can't overflow the call stack. How the
618
+ * text reads depends only on where a scan is, so a scan passing a point —
619
+ * just past a {, a string or a template literal — goes on as a scan
620
+ * starting there would: its result there is recorded too, and a recorded
621
+ * result is used instead of scanning on. Scans from many starts in one
622
+ * input then take about linear time.
325
623
  */
326
- export function scanBalancedBrace(input: string, i: number): number {
327
- let depth = 1;
328
- while (i < input.length) {
329
- const c = input[i]!;
330
- if (c === '{') {
331
- depth++;
332
- } else if (c === '}') {
333
- if (--depth === 0) return i;
334
- } else if (
335
- (c === '"' || c === "'") &&
336
- !(i > 0 && NON_STRING_QUOTE_PREFIX.test(input[i - 1]!))
337
- ) {
338
- const end = skipQuoted(input, i);
339
- if (end !== -1) {
340
- i = end;
624
+ function scanBraceLenient(input: string, i: number, memo: ScanMemo): number {
625
+ const known = memo.brace.get(i);
626
+ if (known !== undefined) return known;
627
+ const stack: LenientScan[] = [{ template: false, start: i, levels: [[]] }];
628
+ for (;;) {
629
+ const scan = stack[stack.length - 1]!;
630
+ let end: number | undefined; // set when `scan` is done
631
+ if (scan.template) {
632
+ // Template text reads the same from such a point, however the scan got
633
+ // there: in `` `\`\`\`… `` each backtick a scan from an earlier one
634
+ // reads as escaped starts a template that reads the same rest.
635
+ let point = true;
636
+ while (end === undefined && i < input.length) {
637
+ if (point) {
638
+ const known = memo.templateText.get(i);
639
+ if (known !== undefined) {
640
+ end = known;
641
+ break;
642
+ }
643
+ scan.passed!.push(i);
644
+ point = false;
645
+ }
646
+ const c = input[i];
647
+ if (c === '\\') {
648
+ i += 2;
649
+ point = true;
650
+ } else if (c === '`') {
651
+ end = i + 1;
652
+ } else if (c === '$' && input[i + 1] === '{') {
653
+ const inner = memo.brace.get(i + 2);
654
+ if (inner === undefined) break; // scan the ${…} first
655
+ if (inner === -1) end = -1;
656
+ else {
657
+ i = inner + 1;
658
+ point = true;
659
+ }
660
+ } else {
661
+ i++;
662
+ }
663
+ }
664
+ if (end === undefined && i < input.length) {
665
+ stack.push({ template: false, start: i + 2, levels: [[]] });
666
+ i += 2;
341
667
  continue;
342
668
  }
343
- } else if (c === '`') {
344
- const end = skipTemplate(input, i);
345
- if (end !== -1) {
346
- i = end;
669
+ end ??= -1;
670
+ } else {
671
+ const { levels } = scan;
672
+ /** The } at `close` ends the innermost level. */
673
+ const closeLevel = (close: number) => {
674
+ for (const at of levels.pop()!) memo.brace.set(at, close);
675
+ if (!levels.length) end = close;
676
+ };
677
+ while (end === undefined && i < input.length) {
678
+ const c = input[i]!;
679
+ let at = -1; // a checkpoint, if one starts here
680
+ if (c === '{') {
681
+ levels.push([]);
682
+ at = ++i;
683
+ } else if (c === '}') {
684
+ closeLevel(i++);
685
+ } else if (
686
+ (c === '"' || c === "'") &&
687
+ !(i > 0 && NON_STRING_QUOTE_PREFIX.test(input[i - 1]!))
688
+ ) {
689
+ const close = skipQuoted(input, i);
690
+ if (close === -1) i++;
691
+ else at = i = close;
692
+ } else if (c === '`') {
693
+ const close = memo.template.get(i);
694
+ if (close === undefined) break; // scan the template first
695
+ if (close === -1) i++;
696
+ else at = i = close;
697
+ } else {
698
+ i++;
699
+ }
700
+ if (at === -1) continue;
701
+ const known = memo.brace.get(at);
702
+ if (known === undefined) {
703
+ levels[levels.length - 1]!.push(at);
704
+ } else if (known === -1) {
705
+ // The innermost level never closes, so neither do the others
706
+ i = input.length;
707
+ } else {
708
+ closeLevel(known);
709
+ i = known + 1;
710
+ }
711
+ }
712
+ if (end === undefined && i < input.length) {
713
+ stack.push({ template: true, start: i, levels: [], passed: [] });
714
+ i++;
347
715
  continue;
348
716
  }
717
+ if (end === undefined) {
718
+ end = -1;
719
+ for (const level of levels) {
720
+ for (const at of level) memo.brace.set(at, -1);
721
+ }
722
+ }
723
+ }
724
+ // `scan` is done: hand its result to the scan that started it
725
+ stack.pop();
726
+ (scan.template ? memo.template : memo.brace).set(scan.start, end);
727
+ for (const at of scan.passed ?? []) memo.templateText.set(at, end);
728
+ const parent = stack[stack.length - 1];
729
+ if (!parent) return end;
730
+ if (parent.template) {
731
+ // An unclosed ${…} leaves the template unclosed: go on to its end
732
+ i = end === -1 ? input.length : end + 1;
733
+ } else {
734
+ // An unclosed template's backtick is text
735
+ i = end === -1 ? scan.start + 1 : end;
349
736
  }
350
- i++;
351
737
  }
352
- return -1;
738
+ }
739
+
740
+ /**
741
+ * Options for {@link tokenize}.
742
+ */
743
+ export interface TokenizeOptions {
744
+ /**
745
+ * Text mode, for markup that becomes a string (HTML attribute values,
746
+ * macro labels): only `{…}` markup and brace escapes are recognized, while
747
+ * `[[` and `<` are text. With no markdown to pair up the backslashes of a
748
+ * run before a brace, they are paired up here: `\\{` is one backslash
749
+ * before a live brace, `\\\{` one before a literal one.
750
+ */
751
+ text?: boolean;
353
752
  }
354
753
 
355
754
  /**
356
755
  * Single-pass tokenizer for Twine passage content.
357
756
  * Recognizes: [[links]], {$variable}, {_temporary}, {macroName args}
358
757
  */
359
- export function tokenize(input: string): Token[] {
758
+ export function tokenize(
759
+ input: string,
760
+ options: TokenizeOptions = {},
761
+ ): Token[] {
762
+ const textMode = options.text === true;
360
763
  const tokens: Token[] = [];
764
+ const memo = createScanMemo();
361
765
  let i = 0;
362
766
  let textStart = 0;
363
767
 
@@ -373,20 +777,43 @@ export function tokenize(input: string): Token[] {
373
777
  }
374
778
 
375
779
  /**
376
- * After an opening raw-body macro ({do}), emit everything up to the
377
- * first {/name} as a single text token so JavaScript source is not
378
- * parsed as markup. A literal "{/do}" inside the code would end it early.
379
- * Without a closer, nothing is consumed and the AST builder reports it.
780
+ * After an opening raw-body macro ({do}), emit everything up to its
781
+ * {/name} as a single text token so JavaScript source is not parsed as
782
+ * markup. The body is lexed as JavaScript statements: a {/do} in code ends
783
+ * it, one inside a string, template or regex literal or a comment does
784
+ * not. If the body is not well-formed JavaScript up to a {/do} in code,
785
+ * the first {/do} ends it. Without a closer, nothing is consumed and the
786
+ * AST builder reports it.
380
787
  */
381
788
  function consumeRawBody(name: string, isClose: boolean) {
382
789
  const lower = name.toLowerCase();
383
790
  if (isClose || !RAW_BODY_MACROS.has(lower)) return;
384
- const closeRe = new RegExp(`\\{/${lower}\\s*\\}`, 'gi');
385
- closeRe.lastIndex = i;
386
- const m = closeRe.exec(input);
387
- if (!m) return;
388
- const closeStart = m.index;
389
- const closeEnd = closeStart + m[0].length;
791
+ const closer = `\\{/${lower}\\s*\\}`;
792
+ // The first closer from here on. Without one from `from` on there is
793
+ // none later either, so many unclosed {do}s don't each search the rest.
794
+ let last = memo.rawClose.get(lower);
795
+ if (!last || i < last.from || (last.at >= 0 && i > last.at)) {
796
+ const firstRe = new RegExp(closer, 'gi');
797
+ firstRe.lastIndex = i;
798
+ last = { from: i, at: firstRe.exec(input)?.index ?? -1 };
799
+ memo.rawClose.set(lower, last);
800
+ }
801
+ const first = last.at;
802
+ if (first === -1) return;
803
+ const atRe = new RegExp(closer, 'iy');
804
+ const closerAt = (k: number) => {
805
+ atRe.lastIndex = k;
806
+ return atRe.test(input);
807
+ };
808
+ let closeStart = findCodeEnd(input, i, {
809
+ goal: 'statements',
810
+ stop: closerAt,
811
+ stopKey: lower,
812
+ cache: memo.js,
813
+ });
814
+ if (closeStart === -1) closeStart = first;
815
+ atRe.lastIndex = closeStart;
816
+ const closeEnd = closeStart + atRe.exec(input)![0].length;
390
817
  flushText(closeStart);
391
818
  tokens.push({
392
819
  type: 'macro',
@@ -398,6 +825,34 @@ export function tokenize(input: string): Token[] {
398
825
  textStart = closeEnd;
399
826
  }
400
827
 
828
+ /**
829
+ * Push an expression token for the `{…}` block opened at `start` whose
830
+ * expression starts at `exprStart`, flushing the text before it first
831
+ * when `flush`. Returns false, consuming nothing, if the block is unclosed.
832
+ */
833
+ function pushExpression(
834
+ exprStart: number,
835
+ start: number,
836
+ flush: boolean,
837
+ className?: string,
838
+ id?: string,
839
+ ): boolean {
840
+ const closeIdx = scanBalancedBrace(input, exprStart, memo);
841
+ if (closeIdx === -1) return false;
842
+ if (flush) flushText(start);
843
+ const token: ExpressionToken = {
844
+ type: 'expression',
845
+ expression: input.slice(exprStart, closeIdx),
846
+ start,
847
+ end: closeIdx + 1,
848
+ };
849
+ if (className) token.className = className;
850
+ if (id) token.id = id;
851
+ tokens.push(token);
852
+ i = textStart = closeIdx + 1;
853
+ return true;
854
+ }
855
+
401
856
  while (i < input.length) {
402
857
  // Escaped braces: \{ and \}. Count the whole backslash run so \\{ is a
403
858
  // backslash pair before a live brace. In an odd run the last backslash
@@ -407,6 +862,15 @@ export function tokenize(input: string): Token[] {
407
862
  let k = i + 1;
408
863
  while (input[k] === '\\') k++;
409
864
  const next = input[k];
865
+ if (textMode && (next === '{' || next === '}')) {
866
+ const escaped = (k - i) % 2 === 1;
867
+ const end = escaped ? k + 1 : k;
868
+ const value = '\\'.repeat((k - i) >> 1) + (escaped ? next : '');
869
+ flushText(i);
870
+ if (value) tokens.push({ type: 'text', value, start: i, end });
871
+ i = textStart = end;
872
+ continue;
873
+ }
410
874
  if ((next === '{' || next === '}') && (k - i) % 2 === 1) {
411
875
  flushText(k - 1);
412
876
  tokens.push({ type: 'text', value: next, start: k - 1, end: k + 1 });
@@ -419,7 +883,7 @@ export function tokenize(input: string): Token[] {
419
883
  }
420
884
 
421
885
  // Check for [[ link
422
- if (input[i] === '[' && input[i + 1] === '[') {
886
+ if (!textMode && input[i] === '[' && input[i + 1] === '[') {
423
887
  flushText(i);
424
888
  const start = i;
425
889
  i += 2;
@@ -437,30 +901,18 @@ export function tokenize(input: string): Token[] {
437
901
  }
438
902
 
439
903
  // Find closing ]]
440
- let depth = 1;
441
904
  const innerStart = i;
442
- while (i < input.length && depth > 0) {
443
- if (input[i] === '[' && input[i + 1] === '[') {
444
- depth++;
445
- i += 2;
446
- } else if (input[i] === ']' && input[i + 1] === ']') {
447
- depth--;
448
- if (depth === 0) break;
449
- i += 2;
450
- } else {
451
- i++;
452
- }
453
- }
905
+ const closeIdx = scanLinkClose(input, innerStart, memo);
454
906
 
455
- if (depth !== 0) {
907
+ if (closeIdx === -1) {
456
908
  // Unclosed link — treat as text
457
909
  i = start + 2;
458
910
  textStart = start;
459
911
  continue;
460
912
  }
461
913
 
462
- const inner = input.slice(innerStart, i);
463
- i += 2; // skip ]]
914
+ const inner = input.slice(innerStart, closeIdx);
915
+ i = closeIdx + 2; // skip ]]
464
916
 
465
917
  const { display, target } = parseLink(inner);
466
918
  const linkToken: LinkToken = {
@@ -518,7 +970,7 @@ export function tokenize(input: string): Token[] {
518
970
  continue;
519
971
  }
520
972
  // Complex expression — scan for balanced closing }
521
- const closeIdx$ = scanBalancedBrace(input, nameStart);
973
+ const closeIdx$ = scanBalancedBrace(input, nameStart - 1, memo);
522
974
  if (closeIdx$ !== -1) {
523
975
  const expression = input.slice(afterSelectors, closeIdx$);
524
976
  i = closeIdx$ + 1;
@@ -563,7 +1015,7 @@ export function tokenize(input: string): Token[] {
563
1015
  continue;
564
1016
  }
565
1017
  // Complex expression — scan for balanced closing }
566
- const closeIdx_ = scanBalancedBrace(input, nameStart);
1018
+ const closeIdx_ = scanBalancedBrace(input, nameStart - 1, memo);
567
1019
  if (closeIdx_ !== -1) {
568
1020
  const expression = input.slice(afterSelectors, closeIdx_);
569
1021
  i = closeIdx_ + 1;
@@ -608,7 +1060,7 @@ export function tokenize(input: string): Token[] {
608
1060
  continue;
609
1061
  }
610
1062
  // Complex expression — scan for balanced closing }
611
- const closeIdx_at = scanBalancedBrace(input, nameStart);
1063
+ const closeIdx_at = scanBalancedBrace(input, nameStart - 1, memo);
612
1064
  if (closeIdx_at !== -1) {
613
1065
  const expression = input.slice(afterSelectors, closeIdx_at);
614
1066
  i = closeIdx_at + 1;
@@ -653,7 +1105,7 @@ export function tokenize(input: string): Token[] {
653
1105
  continue;
654
1106
  }
655
1107
  // Complex expression — scan for balanced closing }
656
- const closeIdx_pct = scanBalancedBrace(input, nameStart);
1108
+ const closeIdx_pct = scanBalancedBrace(input, nameStart - 1, memo);
657
1109
  if (closeIdx_pct !== -1) {
658
1110
  const expression = input.slice(afterSelectors, closeIdx_pct);
659
1111
  i = closeIdx_pct + 1;
@@ -675,11 +1127,18 @@ export function tokenize(input: string): Token[] {
675
1127
  continue;
676
1128
  }
677
1129
 
1130
+ if (
1131
+ EXPRESSION_START.has(charAfter!) &&
1132
+ pushExpression(afterSelectors, start, false, className, id)
1133
+ ) {
1134
+ continue;
1135
+ }
1136
+
678
1137
  if (charAfter !== undefined && /[a-zA-Z]/.test(charAfter)) {
679
1138
  // {.class#id macroName args}
680
1139
  // Scan to closing }, tracking brace nesting and string literals
681
1140
  const contentStart = afterSelectors;
682
- const closeIdx = scanBalancedBrace(input, contentStart);
1141
+ const closeIdx = scanMacroClose(input, contentStart, memo);
683
1142
 
684
1143
  if (closeIdx === -1) {
685
1144
  i = start + 1;
@@ -734,7 +1193,7 @@ export function tokenize(input: string): Token[] {
734
1193
  continue;
735
1194
  }
736
1195
  // Complex expression — scan for balanced closing }
737
- const closeIdx = scanBalancedBrace(input, nameStart);
1196
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
738
1197
  if (closeIdx !== -1) {
739
1198
  const expression = input.slice(start + 1, closeIdx);
740
1199
  i = closeIdx + 1;
@@ -774,7 +1233,7 @@ export function tokenize(input: string): Token[] {
774
1233
  continue;
775
1234
  }
776
1235
  // Complex expression — scan for balanced closing }
777
- const closeIdx = scanBalancedBrace(input, nameStart);
1236
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
778
1237
  if (closeIdx !== -1) {
779
1238
  const expression = input.slice(start + 1, closeIdx);
780
1239
  i = closeIdx + 1;
@@ -814,7 +1273,7 @@ export function tokenize(input: string): Token[] {
814
1273
  continue;
815
1274
  }
816
1275
  // Complex expression — scan for balanced closing }
817
- const closeIdx = scanBalancedBrace(input, nameStart);
1276
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
818
1277
  if (closeIdx !== -1) {
819
1278
  const expression = input.slice(start + 1, closeIdx);
820
1279
  i = closeIdx + 1;
@@ -854,7 +1313,7 @@ export function tokenize(input: string): Token[] {
854
1313
  continue;
855
1314
  }
856
1315
  // Complex expression — scan for balanced closing }
857
- const closeIdx = scanBalancedBrace(input, nameStart);
1316
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
858
1317
  if (closeIdx !== -1) {
859
1318
  const expression = input.slice(start + 1, closeIdx);
860
1319
  i = closeIdx + 1;
@@ -873,6 +1332,13 @@ export function tokenize(input: string): Token[] {
873
1332
  continue;
874
1333
  }
875
1334
 
1335
+ // {(expr)} or {!expr}: an expression that doesn't start with a variable
1336
+ if (EXPRESSION_START.has(nextChar!)) {
1337
+ if (pushExpression(i + 1, start, true)) continue;
1338
+ i++;
1339
+ continue;
1340
+ }
1341
+
876
1342
  // {macro ...} or {/macro} — but not bare { that's just text
877
1343
  // Must start with a letter or /
878
1344
  if (
@@ -884,7 +1350,7 @@ export function tokenize(input: string): Token[] {
884
1350
  // Scan to closing }, tracking brace nesting (object literals)
885
1351
  // and string literals
886
1352
  const contentStart = i + 1;
887
- const closeIdx = scanBalancedBrace(input, contentStart);
1353
+ const closeIdx = scanMacroClose(input, contentStart, memo);
888
1354
 
889
1355
  if (closeIdx === -1) {
890
1356
  // Unclosed macro — treat as text
@@ -916,7 +1382,7 @@ export function tokenize(input: string): Token[] {
916
1382
  }
917
1383
 
918
1384
  // Check for < — HTML tag
919
- if (input[i] === '<') {
1385
+ if (!textMode && input[i] === '<') {
920
1386
  const start = i;
921
1387
  let j = i + 1;
922
1388
 
@@ -958,9 +1424,10 @@ export function tokenize(input: string): Token[] {
958
1424
  continue;
959
1425
  }
960
1426
  } else {
961
- // Opening or self-closing tag: parse attributes
962
- const parsed = parseHtmlAttributes(input, j);
963
- j = parsed.endIdx;
1427
+ // Opening or self-closing tag: find where its attributes end, and
1428
+ // read them only if the tag closes there
1429
+ const attrsStart = j;
1430
+ j = scanAttributes(input, attrsStart, memo);
964
1431
 
965
1432
  let isSelfClose = HTML_VOID_TAGS.has(tagLower);
966
1433
  if (input[j] === '/') {
@@ -974,7 +1441,8 @@ export function tokenize(input: string): Token[] {
974
1441
  tokens.push({
975
1442
  type: 'html',
976
1443
  tag,
977
- attributes: parsed.attributes,
1444
+ attributes: parseHtmlAttributes(input, attrsStart, memo)
1445
+ .attributes,
978
1446
  isClose: false,
979
1447
  isSelfClose,
980
1448
  start,