@rohal12/spindle 0.51.3 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/dist/pkg/format.js +1 -1
  2. package/dist/pkg/headless.js +4833 -1603
  3. package/dist/pkg/macro-registry.json +7 -7
  4. package/dist/pkg/story-variables.js +1658 -189
  5. package/package.json +5 -2
  6. package/src/automation/runner.ts +2 -1
  7. package/src/class-registry.ts +277 -90
  8. package/src/components/Passage.tsx +2 -2
  9. package/src/components/PassageDialog.tsx +2 -5
  10. package/src/components/StoryInterface.tsx +2 -4
  11. package/src/components/macros/Button.tsx +9 -32
  12. package/src/components/macros/Checkbox.tsx +10 -4
  13. package/src/components/macros/Computed.tsx +19 -13
  14. package/src/components/macros/Dialog.tsx +4 -1
  15. package/src/components/macros/For.tsx +33 -4
  16. package/src/components/macros/If.tsx +8 -0
  17. package/src/components/macros/Include.tsx +15 -24
  18. package/src/components/macros/MacroError.tsx +2 -1
  19. package/src/components/macros/MacroLink.tsx +14 -46
  20. package/src/components/macros/Meter.tsx +26 -40
  21. package/src/components/macros/Nobr.tsx +1 -0
  22. package/src/components/macros/PassageDisplay.tsx +3 -0
  23. package/src/components/macros/Print.tsx +4 -0
  24. package/src/components/macros/Radiobutton.tsx +27 -2
  25. package/src/components/macros/SaveManager.tsx +39 -14
  26. package/src/components/macros/Span.tsx +1 -0
  27. package/src/components/macros/StoryTitle.tsx +1 -0
  28. package/src/components/macros/Switch.tsx +13 -0
  29. package/src/components/macros/Unset.tsx +30 -10
  30. package/src/components/macros/VarDisplay.tsx +21 -4
  31. package/src/components/macros/Watch.tsx +88 -40
  32. package/src/components/macros/Widget.tsx +20 -1
  33. package/src/components/macros/WidgetInvocation.tsx +20 -158
  34. package/src/components/macros/arg-utils.ts +226 -0
  35. package/src/components/macros/detached-body.tsx +68 -0
  36. package/src/components/macros/option-utils.ts +10 -5
  37. package/src/define-macro.ts +44 -38
  38. package/src/execute-mutation.ts +499 -28
  39. package/src/expression.ts +88 -272
  40. package/src/hooks/use-action.ts +18 -3
  41. package/src/hooks/use-interpolate.ts +36 -5
  42. package/src/index.tsx +10 -1
  43. package/src/interpolation.ts +394 -96
  44. package/src/js-lexer.ts +1460 -0
  45. package/src/markup/code-attributes.ts +64 -0
  46. package/src/markup/markdown.ts +188 -9
  47. package/src/markup/render.tsx +552 -113
  48. package/src/markup/tokenizer.ts +601 -119
  49. package/src/prng.ts +8 -8
  50. package/src/registry.ts +35 -0
  51. package/src/saves/save-manager.ts +368 -153
  52. package/src/saves/storage.ts +24 -7
  53. package/src/saves/types.ts +20 -6
  54. package/src/store.ts +549 -137
  55. package/src/story-api.ts +46 -81
  56. package/src/story-init.ts +1 -1
  57. package/src/story-variables.ts +98 -102
  58. package/src/triggers.ts +6 -2
  59. package/src/utils/error-message.ts +12 -0
  60. package/src/utils/live-locals.ts +10 -3
  61. package/src/utils/namespace.ts +71 -0
  62. package/src/utils/object-path.ts +194 -0
  63. package/src/utils/stable-key.ts +82 -0
  64. package/src/widgets/widget-registry.ts +9 -0
  65. package/types/index.d.ts +43 -7
  66. package/types/tooling.d.ts +1 -0
@@ -1,3 +1,6 @@
1
+ import { createJsScanCache, findCodeEnd, type JsScanCache } from '../js-lexer';
2
+ import { isCodeAttribute } from './code-attributes';
3
+
1
4
  export interface TextToken {
2
5
  type: 'text';
3
6
  value: string;
@@ -84,6 +87,17 @@ const HTML_VOID_TAGS = new Set([
84
87
  'wbr',
85
88
  ]);
86
89
 
90
+ /** Variable sigils: story ($), temporary (_), local (@), transient (%). */
91
+ const SIGIL_CHARS = new Set(['$', '_', '@', '%']);
92
+
93
+ /**
94
+ * Characters other than a sigil that open an expression after `{`:
95
+ * `{(Math.max($a, 0))}`, `{!$done}`. Other characters that can start an
96
+ * expression (quotes, digits, `[`, `-`) are left out, because braces around
97
+ * them are common as literal text (JSON, regex quantifiers, `{[[link]]}`).
98
+ */
99
+ const EXPRESSION_START = new Set(['(', '!']);
100
+
87
101
  /** Macros whose body is JavaScript source, kept verbatim instead of tokenized. */
88
102
  const RAW_BODY_MACROS = new Set(['do']);
89
103
 
@@ -172,11 +186,8 @@ function parseSelectors(
172
186
  if (/[a-zA-Z0-9_-]/.test(input[i]!)) {
173
187
  name += input[i];
174
188
  i++;
175
- } else if (
176
- input[i] === '{' &&
177
- (input[i + 1] === '$' || input[i + 1] === '_' || input[i + 1] === '@')
178
- ) {
179
- // Consume interpolation: {$var}, {_var}, {@var} (with optional dot paths)
189
+ } else if (input[i] === '{' && SIGIL_CHARS.has(input[i + 1]!)) {
190
+ // Consume interpolation: {$var}, {_var}, {@var}, {%var} (with optional dot paths)
180
191
  const braceStart = i;
181
192
  i += 2; // skip { and prefix
182
193
  while (i < input.length && /[\w.]/.test(input[i]!)) i++;
@@ -211,12 +222,69 @@ function parseSelectors(
211
222
  function parseHtmlAttributes(
212
223
  input: string,
213
224
  j: number,
225
+ memo: ScanMemo,
214
226
  ): { attributes: Record<string, string>; endIdx: number } {
215
227
  const attributes: Record<string, string> = {};
228
+ // As in HTML, the first of attributes with the same (case-insensitive)
229
+ // name wins. Defined as own properties so `__proto__` is kept too.
230
+ const seen = new Set<string>();
231
+ const endIdx = scanAttributes(input, j, memo, (name, value) => {
232
+ const lower = name.toLowerCase();
233
+ if (seen.has(lower)) return;
234
+ seen.add(lower);
235
+ Object.defineProperty(attributes, name, {
236
+ value,
237
+ enumerable: true,
238
+ writable: true,
239
+ configurable: true,
240
+ });
241
+ });
242
+ return { attributes, endIdx };
243
+ }
216
244
 
245
+ /**
246
+ * Index just past the attributes of a tag that start at `j` (after its
247
+ * name), where its `>` or `/>` goes. Pass the same memo to repeated scans of
248
+ * one input to share their work.
249
+ */
250
+ export function scanTagAttributes(
251
+ input: string,
252
+ j: number,
253
+ memo: ScanMemo = createScanMemo(),
254
+ ): number {
255
+ return scanAttributes(input, j, memo);
256
+ }
257
+
258
+ /**
259
+ * Scan the attributes of a tag from position j, handing each to `add` when
260
+ * given. Returns the position after the last attribute.
261
+ *
262
+ * Where the attributes end depends only on where an attribute starts, so
263
+ * without `add` the result is recorded for every attribute start passed (in
264
+ * `memo.tag`) and a recorded one is used instead of scanning on. An unquoted
265
+ * value takes in a `<`, so in `<a x=<a x=<a x=…` each opener's scan would
266
+ * otherwise run over all the openers after it, taking quadratic time. A tag
267
+ * whose attributes end in its `>` is consumed, so collecting the attributes
268
+ * of those (with `add`, unrecorded) reads each character once.
269
+ */
270
+ function scanAttributes(
271
+ input: string,
272
+ j: number,
273
+ memo: ScanMemo,
274
+ add?: (name: string, value: string) => void,
275
+ ): number {
276
+ const passed: number[] = [];
217
277
  while (j < input.length) {
218
278
  // Skip whitespace
219
279
  while (j < input.length && /\s/.test(input[j]!)) j++;
280
+ if (!add) {
281
+ const known = memo.tag.get(j);
282
+ if (known !== undefined) {
283
+ j = known;
284
+ break;
285
+ }
286
+ passed.push(j);
287
+ }
220
288
  // End of tag?
221
289
  if (
222
290
  j >= input.length ||
@@ -231,39 +299,267 @@ function parseHtmlAttributes(
231
299
  const attrName = input.slice(attrStart, j);
232
300
  if (!attrName) break;
233
301
 
234
- // Check for = value
235
- if (input[j] === '=') {
236
- j++; // skip =
302
+ // Check for = value. HTML allows whitespace on either side of the =
303
+ // (`id = "x"`); whitespace not followed by = ends a boolean attribute.
304
+ let eqIdx = j;
305
+ while (eqIdx < input.length && /\s/.test(input[eqIdx]!)) eqIdx++;
306
+ if (input[eqIdx] === '=') {
307
+ j = eqIdx + 1; // skip =
308
+ while (j < input.length && /\s/.test(input[j]!)) j++;
237
309
  if (input[j] === '"' || input[j] === "'") {
238
310
  const quote = input[j]!;
239
311
  j++; // skip opening quote
240
312
  const valStart = j;
241
- while (j < input.length) {
242
- if (input[j] === '{') {
243
- // Skip a whole {…} interpolation so quotes inside it don't end the value
244
- const closeIdx = scanBalancedBrace(input, j + 1);
245
- if (closeIdx !== -1) {
246
- j = closeIdx + 1;
247
- continue;
248
- }
249
- } else if (input[j] === quote) break;
250
- j++;
251
- }
252
- attributes[attrName] = input.slice(valStart, j);
313
+ j = scanQuotedValue(input, j, quote, isCodeAttribute(attrName), memo);
314
+ add?.(attrName, input.slice(valStart, j));
253
315
  if (j < input.length) j++; // skip closing quote
254
316
  } else {
255
- // Unquoted value
317
+ // Unquoted value, up to whitespace or `>`. It takes in a `<`, so in
318
+ // `<a:=<a:=<a:=…` it holds every tag after it: the last run of such
319
+ // characters found is kept, as each tag's value ends where it does.
256
320
  const valStart = j;
257
- while (j < input.length && /[^\s>]/.test(input[j]!)) j++;
258
- attributes[attrName] = input.slice(valStart, j);
321
+ const run = memo.unquoted;
322
+ if (j < run.from || j > run.to) {
323
+ run.from = j;
324
+ while (j < input.length && /[^\s>]/.test(input[j]!)) j++;
325
+ run.to = j;
326
+ }
327
+ j = run.to;
328
+ add?.(attrName, input.slice(valStart, j));
259
329
  }
260
330
  } else {
261
331
  // Boolean attribute
262
- attributes[attrName] = '';
332
+ add?.(attrName, '');
263
333
  }
264
334
  }
335
+ for (const at of passed) memo.tag.set(at, j);
336
+ return j;
337
+ }
338
+
339
+ /**
340
+ * Index of the quote ending the attribute value that starts at `j`, or the
341
+ * end of the input. A `{…}` interpolation in it is skipped whole, so quotes
342
+ * inside it don't end the value. A code attribute's value is no markup
343
+ * (`isCodeAttribute`): only `{` and a sigil open a reference, other braces
344
+ * and backslashes are text, as `splitSigilTemplate` reads them.
345
+ *
346
+ * The value reads the same from just past an interpolation, however the scan
347
+ * got there, so the end is recorded there (in `memo.value`), and a recorded
348
+ * one is used. An unclosed value that skips interpolations to the end of
349
+ * the input (`<a x="}<a x={<a x="}…`) is then read once, not once per tag.
350
+ */
351
+ function scanQuotedValue(
352
+ input: string,
353
+ j: number,
354
+ quote: string,
355
+ code: boolean,
356
+ memo: ScanMemo,
357
+ ): number {
358
+ const variant = (quote === '"' ? 0 : 2) + (code ? 1 : 0);
359
+ const passed: number[] = [];
360
+ let checkpoint = true;
361
+ while (j < input.length) {
362
+ if (checkpoint) {
363
+ const known = memo.value.get(j * 4 + variant);
364
+ if (known !== undefined) {
365
+ j = known;
366
+ break;
367
+ }
368
+ passed.push(j * 4 + variant);
369
+ checkpoint = false;
370
+ }
371
+ if (!code && input[j] === '\\') {
372
+ // A brace after an odd backslash run is escaped (`\{`) and opens
373
+ // no interpolation, as in passage text.
374
+ let k = j + 1;
375
+ while (input[k] === '\\') k++;
376
+ const brace = input[k] === '{' || input[k] === '}';
377
+ j = brace && (k - j) % 2 === 1 ? k + 1 : k;
378
+ continue;
379
+ }
380
+ if (input[j] === '{') {
381
+ const closeIdx = !code
382
+ ? scanBlockClose(input, j, memo)
383
+ : SIGIL_CHARS.has(input[j + 1]!)
384
+ ? scanBalancedBrace(input, j + 1, memo)
385
+ : -1;
386
+ if (closeIdx !== -1) {
387
+ j = closeIdx + 1;
388
+ checkpoint = true;
389
+ continue;
390
+ }
391
+ } else if (input[j] === quote) break;
392
+ j++;
393
+ }
394
+ for (const at of passed) memo.value.set(at, j);
395
+ return j;
396
+ }
397
+
398
+ /**
399
+ * Results of the brace and template scans of one input, shared by repeated
400
+ * scans of it. A scan's result depends only on the input and where it
401
+ * starts, so caching it is exact. Without the cache, each unclosed template
402
+ * literal was scanned once as a template and again as plain text, so nested
403
+ * unclosed ones (`` {$a`${$a`${… ``) took time exponential in their depth;
404
+ * and each unclosed `{` was scanned to the end of the input, so many of them
405
+ * took quadratic time. A memo must only be reused for scans of the same input
406
+ * string.
407
+ */
408
+ export interface ScanMemo {
409
+ /** `scanBalancedBrace` results, by lenient start. */
410
+ code: Map<number, number>;
411
+ /** JavaScript lexer results. */
412
+ js: JsScanCache;
413
+ /** Lenient brace scan results. */
414
+ brace: Map<number, number>;
415
+ /** Lenient template literal scan results. */
416
+ template: Map<number, number>;
417
+ /** Lenient template literal scan results, by a point in its text. */
418
+ templateText: Map<number, number>;
419
+ /** The last macro name run found: no whitespace or } in [from, to). */
420
+ name: { from: number; to: number };
421
+ /** The last unquoted attribute value run: no whitespace or > in [from, to). */
422
+ unquoted: { from: number; to: number };
423
+ /** Link scan results (`scanLinkClose`). */
424
+ link: Map<number, number>;
425
+ /** Where the attributes of a tag end, by attribute start (`scanAttributes`). */
426
+ tag: Map<number, number>;
427
+ /** Where quoted attribute values end, by checkpoint (`scanQuotedValue`). */
428
+ value: Map<number, number>;
429
+ /**
430
+ * The last search for a raw-body closer, by macro name: the first one
431
+ * from `from` on is at `at` (-1 for none).
432
+ */
433
+ rawClose: Map<string, { from: number; at: number }>;
434
+ }
435
+
436
+ export function createScanMemo(): ScanMemo {
437
+ return {
438
+ code: new Map(),
439
+ js: createJsScanCache(),
440
+ brace: new Map(),
441
+ template: new Map(),
442
+ templateText: new Map(),
443
+ name: { from: 0, to: -1 },
444
+ unquoted: { from: 0, to: -1 },
445
+ link: new Map(),
446
+ tag: new Map(),
447
+ value: new Map(),
448
+ rawClose: new Map(),
449
+ };
450
+ }
451
+
452
+ /**
453
+ * Find the } closing the code that starts at position i: a `{$…}`
454
+ * expression from its sigil on, an attribute interpolation. The code is
455
+ * lexed as JavaScript, so only a } in code outside the brackets it opened
456
+ * counts — not one in a string, template or regex literal or a comment —
457
+ * and `/` after an operand divides.
458
+ *
459
+ * Text that is not well-formed JavaScript (an apostrophe, an unterminated
460
+ * literal or comment, unbalanced brackets) is scanned leniently instead, as
461
+ * prose-like macro arguments are: braces count except inside string and
462
+ * template literals, and a quote that can't start a string (apostrophe,
463
+ * escaped, not closed on its line) is text.
464
+ *
465
+ * Returns the index of the closing } or -1 if there is none. Pass the same
466
+ * memo to repeated scans of one input to share their work.
467
+ */
468
+ export function scanBalancedBrace(
469
+ input: string,
470
+ i: number,
471
+ memo: ScanMemo = createScanMemo(),
472
+ ): number {
473
+ return scanClose(input, i, i, memo);
474
+ }
475
+
476
+ /**
477
+ * Index of the } closing the macro whose content (name, then arguments)
478
+ * starts at `contentStart`, or -1. The name runs up to whitespace or the }
479
+ * (as `parseMacroContent` reads it); the arguments after it are code. The
480
+ * lenient scan covers the whole content, as it always has.
481
+ */
482
+ function scanMacroClose(
483
+ input: string,
484
+ contentStart: number,
485
+ memo: ScanMemo,
486
+ ): number {
487
+ const run = memo.name;
488
+ if (contentStart < run.from || contentStart > run.to) {
489
+ let k = contentStart;
490
+ while (k < input.length && input[k] !== '}' && !/\s/.test(input[k]!)) k++;
491
+ run.from = contentStart;
492
+ run.to = k;
493
+ }
494
+ return scanClose(input, run.to, contentStart, memo);
495
+ }
496
+
497
+ /**
498
+ * Index of the } closing the `{…}` block opened at `open`, read as passage
499
+ * text reads it: a macro (`{name args}`, `{/name}`) has its arguments lexed
500
+ * after its name, an expression (`{$…}`, `{(…)}`, `{!…}`) from its first
501
+ * character, after any `.class#id` selectors. Any other block is scanned as
502
+ * code from just past the {. Returns -1 if it is unclosed.
503
+ */
504
+ function scanBlockClose(input: string, open: number, memo: ScanMemo): number {
505
+ let at = open + 1;
506
+ const c = input[at];
507
+ if (c === '.' || c === '#') {
508
+ at = parseSelectors(input, at).endIdx;
509
+ if (input[at] === ' ') at++;
510
+ }
511
+ const first = input[at];
512
+ if (first !== undefined && (first === '/' || /[a-zA-Z]/.test(first))) {
513
+ return scanMacroClose(input, at, memo);
514
+ }
515
+ return scanBalancedBrace(input, at, memo);
516
+ }
265
517
 
266
- return { attributes, endIdx: j };
518
+ /** Lex the code from `codeStart`, else scan leniently from `lenientStart`. */
519
+ function scanClose(
520
+ input: string,
521
+ codeStart: number,
522
+ lenientStart: number,
523
+ memo: ScanMemo,
524
+ ): number {
525
+ let end = memo.code.get(lenientStart);
526
+ if (end === undefined) {
527
+ end = findCodeEnd(input, codeStart, { cache: memo.js });
528
+ if (end === -1) end = scanBraceLenient(input, lenientStart, memo);
529
+ memo.code.set(lenientStart, end);
530
+ }
531
+ return end;
532
+ }
533
+
534
+ /**
535
+ * Index of the ]] closing the link whose text starts at `from`, or -1. A
536
+ * [[ inside opens a nested pair. Each nested [[ starts a scan that reads the
537
+ * rest the same way, so its result is recorded too: unclosed [[s don't each
538
+ * scan to the end of the input.
539
+ */
540
+ function scanLinkClose(input: string, from: number, memo: ScanMemo): number {
541
+ const known = memo.link.get(from);
542
+ if (known !== undefined) return known;
543
+ const opens = [from];
544
+ let i = from;
545
+ while (i < input.length) {
546
+ if (input[i] === '[' && input[i + 1] === '[') {
547
+ i += 2;
548
+ const inner = memo.link.get(i);
549
+ if (inner === undefined) opens.push(i);
550
+ else if (inner === -1)
551
+ break; // nor does this one close
552
+ else i = inner + 2;
553
+ } else if (input[i] === ']' && input[i + 1] === ']') {
554
+ memo.link.set(opens.pop()!, i);
555
+ if (!opens.length) return i;
556
+ i += 2;
557
+ } else {
558
+ i++;
559
+ }
560
+ }
561
+ for (const open of opens) memo.link.set(open, -1);
562
+ return -1;
267
563
  }
268
564
 
269
565
  /**
@@ -284,76 +580,188 @@ function skipQuoted(input: string, i: number): number {
284
580
  return -1;
285
581
  }
286
582
 
287
- /**
288
- * Skip a `…` template literal opening at i, including ${…} parts.
289
- * Returns the index just past the closing backtick, or -1 if unclosed.
290
- */
291
- function skipTemplate(input: string, i: number): number {
292
- let j = i + 1;
293
- while (j < input.length) {
294
- const c = input[j];
295
- if (c === '\\') {
296
- j += 2;
297
- } else if (c === '`') {
298
- return j + 1;
299
- } else if (c === '$' && input[j + 1] === '{') {
300
- const closeIdx = scanBalancedBrace(input, j + 2);
301
- if (closeIdx === -1) return -1;
302
- j = closeIdx + 1;
303
- } else {
304
- j++;
305
- }
306
- }
307
- return -1;
308
- }
309
-
310
583
  /**
311
584
  * A quote directly after a letter/digit is an apostrophe (don't), not a
312
585
  * string; after a backslash it is an escaped attribute delimiter (\").
313
586
  */
314
587
  const NON_STRING_QUOTE_PREFIX = /[\p{L}\p{N}_\\]/u;
315
588
 
589
+ /**
590
+ * A brace or template literal scan in progress: braces from `start` (just
591
+ * past a {) to their closing }, or a template literal from its backtick at
592
+ * `start` to the closing one.
593
+ */
594
+ interface LenientScan {
595
+ template: boolean;
596
+ start: number;
597
+ /**
598
+ * For braces, per brace depth from the outermost: the checkpoints at that
599
+ * depth whose scans end where the depth does.
600
+ */
601
+ levels: number[][];
602
+ /**
603
+ * For a template literal, the points in its text it passed (just past its
604
+ * backtick, an escape or an interpolation), whose scans end where it does.
605
+ */
606
+ passed?: number[];
607
+ }
608
+
316
609
  /**
317
610
  * Scan for the balanced closing } starting at position i (just past the {).
318
611
  * Braces inside string and template literals are ignored. A quote that
319
- * can't start a string (apostrophe, escaped, unterminated) counts as text.
612
+ * can't start a string (apostrophe, escaped, unterminated) counts as text,
613
+ * and so does a backtick without a closing one.
320
614
  * Returns the index of the closing } or -1 if unbalanced.
615
+ *
616
+ * Template literals and their ${…} parts are scans on a stack, not
617
+ * recursive calls, so deep nesting can't overflow the call stack. How the
618
+ * text reads depends only on where a scan is, so a scan passing a point —
619
+ * just past a {, a string or a template literal — goes on as a scan
620
+ * starting there would: its result there is recorded too, and a recorded
621
+ * result is used instead of scanning on. Scans from many starts in one
622
+ * input then take about linear time.
321
623
  */
322
- export function scanBalancedBrace(input: string, i: number): number {
323
- let depth = 1;
324
- while (i < input.length) {
325
- const c = input[i]!;
326
- if (c === '{') {
327
- depth++;
328
- } else if (c === '}') {
329
- if (--depth === 0) return i;
330
- } else if (
331
- (c === '"' || c === "'") &&
332
- !(i > 0 && NON_STRING_QUOTE_PREFIX.test(input[i - 1]!))
333
- ) {
334
- const end = skipQuoted(input, i);
335
- if (end !== -1) {
336
- i = end;
624
+ function scanBraceLenient(input: string, i: number, memo: ScanMemo): number {
625
+ const known = memo.brace.get(i);
626
+ if (known !== undefined) return known;
627
+ const stack: LenientScan[] = [{ template: false, start: i, levels: [[]] }];
628
+ for (;;) {
629
+ const scan = stack[stack.length - 1]!;
630
+ let end: number | undefined; // set when `scan` is done
631
+ if (scan.template) {
632
+ // Template text reads the same from such a point, however the scan got
633
+ // there: in `` `\`\`\`… `` each backtick a scan from an earlier one
634
+ // reads as escaped starts a template that reads the same rest.
635
+ let point = true;
636
+ while (end === undefined && i < input.length) {
637
+ if (point) {
638
+ const known = memo.templateText.get(i);
639
+ if (known !== undefined) {
640
+ end = known;
641
+ break;
642
+ }
643
+ scan.passed!.push(i);
644
+ point = false;
645
+ }
646
+ const c = input[i];
647
+ if (c === '\\') {
648
+ i += 2;
649
+ point = true;
650
+ } else if (c === '`') {
651
+ end = i + 1;
652
+ } else if (c === '$' && input[i + 1] === '{') {
653
+ const inner = memo.brace.get(i + 2);
654
+ if (inner === undefined) break; // scan the ${…} first
655
+ if (inner === -1) end = -1;
656
+ else {
657
+ i = inner + 1;
658
+ point = true;
659
+ }
660
+ } else {
661
+ i++;
662
+ }
663
+ }
664
+ if (end === undefined && i < input.length) {
665
+ stack.push({ template: false, start: i + 2, levels: [[]] });
666
+ i += 2;
337
667
  continue;
338
668
  }
339
- } else if (c === '`') {
340
- const end = skipTemplate(input, i);
341
- if (end !== -1) {
342
- i = end;
669
+ end ??= -1;
670
+ } else {
671
+ const { levels } = scan;
672
+ /** The } at `close` ends the innermost level. */
673
+ const closeLevel = (close: number) => {
674
+ for (const at of levels.pop()!) memo.brace.set(at, close);
675
+ if (!levels.length) end = close;
676
+ };
677
+ while (end === undefined && i < input.length) {
678
+ const c = input[i]!;
679
+ let at = -1; // a checkpoint, if one starts here
680
+ if (c === '{') {
681
+ levels.push([]);
682
+ at = ++i;
683
+ } else if (c === '}') {
684
+ closeLevel(i++);
685
+ } else if (
686
+ (c === '"' || c === "'") &&
687
+ !(i > 0 && NON_STRING_QUOTE_PREFIX.test(input[i - 1]!))
688
+ ) {
689
+ const close = skipQuoted(input, i);
690
+ if (close === -1) i++;
691
+ else at = i = close;
692
+ } else if (c === '`') {
693
+ const close = memo.template.get(i);
694
+ if (close === undefined) break; // scan the template first
695
+ if (close === -1) i++;
696
+ else at = i = close;
697
+ } else {
698
+ i++;
699
+ }
700
+ if (at === -1) continue;
701
+ const known = memo.brace.get(at);
702
+ if (known === undefined) {
703
+ levels[levels.length - 1]!.push(at);
704
+ } else if (known === -1) {
705
+ // The innermost level never closes, so neither do the others
706
+ i = input.length;
707
+ } else {
708
+ closeLevel(known);
709
+ i = known + 1;
710
+ }
711
+ }
712
+ if (end === undefined && i < input.length) {
713
+ stack.push({ template: true, start: i, levels: [], passed: [] });
714
+ i++;
343
715
  continue;
344
716
  }
717
+ if (end === undefined) {
718
+ end = -1;
719
+ for (const level of levels) {
720
+ for (const at of level) memo.brace.set(at, -1);
721
+ }
722
+ }
723
+ }
724
+ // `scan` is done: hand its result to the scan that started it
725
+ stack.pop();
726
+ (scan.template ? memo.template : memo.brace).set(scan.start, end);
727
+ for (const at of scan.passed ?? []) memo.templateText.set(at, end);
728
+ const parent = stack[stack.length - 1];
729
+ if (!parent) return end;
730
+ if (parent.template) {
731
+ // An unclosed ${…} leaves the template unclosed: go on to its end
732
+ i = end === -1 ? input.length : end + 1;
733
+ } else {
734
+ // An unclosed template's backtick is text
735
+ i = end === -1 ? scan.start + 1 : end;
345
736
  }
346
- i++;
347
737
  }
348
- return -1;
738
+ }
739
+
740
+ /**
741
+ * Options for {@link tokenize}.
742
+ */
743
+ export interface TokenizeOptions {
744
+ /**
745
+ * Text mode, for markup that becomes a string (HTML attribute values,
746
+ * macro labels): only `{…}` markup and brace escapes are recognized, while
747
+ * `[[` and `<` are text. With no markdown to pair up the backslashes of a
748
+ * run before a brace, they are paired up here: `\\{` is one backslash
749
+ * before a live brace, `\\\{` one before a literal one.
750
+ */
751
+ text?: boolean;
349
752
  }
350
753
 
351
754
  /**
352
755
  * Single-pass tokenizer for Twine passage content.
353
756
  * Recognizes: [[links]], {$variable}, {_temporary}, {macroName args}
354
757
  */
355
- export function tokenize(input: string): Token[] {
758
+ export function tokenize(
759
+ input: string,
760
+ options: TokenizeOptions = {},
761
+ ): Token[] {
762
+ const textMode = options.text === true;
356
763
  const tokens: Token[] = [];
764
+ const memo = createScanMemo();
357
765
  let i = 0;
358
766
  let textStart = 0;
359
767
 
@@ -369,20 +777,43 @@ export function tokenize(input: string): Token[] {
369
777
  }
370
778
 
371
779
  /**
372
- * After an opening raw-body macro ({do}), emit everything up to the
373
- * first {/name} as a single text token so JavaScript source is not
374
- * parsed as markup. A literal "{/do}" inside the code would end it early.
375
- * Without a closer, nothing is consumed and the AST builder reports it.
780
+ * After an opening raw-body macro ({do}), emit everything up to its
781
+ * {/name} as a single text token so JavaScript source is not parsed as
782
+ * markup. The body is lexed as JavaScript statements: a {/do} in code ends
783
+ * it, one inside a string, template or regex literal or a comment does
784
+ * not. If the body is not well-formed JavaScript up to a {/do} in code,
785
+ * the first {/do} ends it. Without a closer, nothing is consumed and the
786
+ * AST builder reports it.
376
787
  */
377
788
  function consumeRawBody(name: string, isClose: boolean) {
378
789
  const lower = name.toLowerCase();
379
790
  if (isClose || !RAW_BODY_MACROS.has(lower)) return;
380
- const closeRe = new RegExp(`\\{/${lower}\\s*\\}`, 'gi');
381
- closeRe.lastIndex = i;
382
- const m = closeRe.exec(input);
383
- if (!m) return;
384
- const closeStart = m.index;
385
- const closeEnd = closeStart + m[0].length;
791
+ const closer = `\\{/${lower}\\s*\\}`;
792
+ // The first closer from here on. Without one from `from` on there is
793
+ // none later either, so many unclosed {do}s don't each search the rest.
794
+ let last = memo.rawClose.get(lower);
795
+ if (!last || i < last.from || (last.at >= 0 && i > last.at)) {
796
+ const firstRe = new RegExp(closer, 'gi');
797
+ firstRe.lastIndex = i;
798
+ last = { from: i, at: firstRe.exec(input)?.index ?? -1 };
799
+ memo.rawClose.set(lower, last);
800
+ }
801
+ const first = last.at;
802
+ if (first === -1) return;
803
+ const atRe = new RegExp(closer, 'iy');
804
+ const closerAt = (k: number) => {
805
+ atRe.lastIndex = k;
806
+ return atRe.test(input);
807
+ };
808
+ let closeStart = findCodeEnd(input, i, {
809
+ goal: 'statements',
810
+ stop: closerAt,
811
+ stopKey: lower,
812
+ cache: memo.js,
813
+ });
814
+ if (closeStart === -1) closeStart = first;
815
+ atRe.lastIndex = closeStart;
816
+ const closeEnd = closeStart + atRe.exec(input)![0].length;
386
817
  flushText(closeStart);
387
818
  tokens.push({
388
819
  type: 'macro',
@@ -394,18 +825,65 @@ export function tokenize(input: string): Token[] {
394
825
  textStart = closeEnd;
395
826
  }
396
827
 
828
+ /**
829
+ * Push an expression token for the `{…}` block opened at `start` whose
830
+ * expression starts at `exprStart`, flushing the text before it first
831
+ * when `flush`. Returns false, consuming nothing, if the block is unclosed.
832
+ */
833
+ function pushExpression(
834
+ exprStart: number,
835
+ start: number,
836
+ flush: boolean,
837
+ className?: string,
838
+ id?: string,
839
+ ): boolean {
840
+ const closeIdx = scanBalancedBrace(input, exprStart, memo);
841
+ if (closeIdx === -1) return false;
842
+ if (flush) flushText(start);
843
+ const token: ExpressionToken = {
844
+ type: 'expression',
845
+ expression: input.slice(exprStart, closeIdx),
846
+ start,
847
+ end: closeIdx + 1,
848
+ };
849
+ if (className) token.className = className;
850
+ if (id) token.id = id;
851
+ tokens.push(token);
852
+ i = textStart = closeIdx + 1;
853
+ return true;
854
+ }
855
+
397
856
  while (i < input.length) {
398
- // Handle escaped braces: \{ and \}
399
- if (input[i] === '\\' && (input[i + 1] === '{' || input[i + 1] === '}')) {
400
- flushText(i);
401
- tokens.push({ type: 'text', value: input[i + 1]!, start: i, end: i + 2 });
402
- i += 2;
403
- textStart = i;
857
+ // Escaped braces: \{ and \}. Count the whole backslash run so \\{ is a
858
+ // backslash pair before a live brace. In an odd run the last backslash
859
+ // escapes the brace; the even rest stays text, which markdown collapses
860
+ // pair by pair like any other \\ in the passage.
861
+ if (input[i] === '\\') {
862
+ let k = i + 1;
863
+ while (input[k] === '\\') k++;
864
+ const next = input[k];
865
+ if (textMode && (next === '{' || next === '}')) {
866
+ const escaped = (k - i) % 2 === 1;
867
+ const end = escaped ? k + 1 : k;
868
+ const value = '\\'.repeat((k - i) >> 1) + (escaped ? next : '');
869
+ flushText(i);
870
+ if (value) tokens.push({ type: 'text', value, start: i, end });
871
+ i = textStart = end;
872
+ continue;
873
+ }
874
+ if ((next === '{' || next === '}') && (k - i) % 2 === 1) {
875
+ flushText(k - 1);
876
+ tokens.push({ type: 'text', value: next, start: k - 1, end: k + 1 });
877
+ i = k + 1;
878
+ textStart = i;
879
+ continue;
880
+ }
881
+ i = k;
404
882
  continue;
405
883
  }
406
884
 
407
885
  // Check for [[ link
408
- if (input[i] === '[' && input[i + 1] === '[') {
886
+ if (!textMode && input[i] === '[' && input[i + 1] === '[') {
409
887
  flushText(i);
410
888
  const start = i;
411
889
  i += 2;
@@ -423,30 +901,18 @@ export function tokenize(input: string): Token[] {
423
901
  }
424
902
 
425
903
  // Find closing ]]
426
- let depth = 1;
427
904
  const innerStart = i;
428
- while (i < input.length && depth > 0) {
429
- if (input[i] === '[' && input[i + 1] === '[') {
430
- depth++;
431
- i += 2;
432
- } else if (input[i] === ']' && input[i + 1] === ']') {
433
- depth--;
434
- if (depth === 0) break;
435
- i += 2;
436
- } else {
437
- i++;
438
- }
439
- }
905
+ const closeIdx = scanLinkClose(input, innerStart, memo);
440
906
 
441
- if (depth !== 0) {
907
+ if (closeIdx === -1) {
442
908
  // Unclosed link — treat as text
443
909
  i = start + 2;
444
910
  textStart = start;
445
911
  continue;
446
912
  }
447
913
 
448
- const inner = input.slice(innerStart, i);
449
- i += 2; // skip ]]
914
+ const inner = input.slice(innerStart, closeIdx);
915
+ i = closeIdx + 2; // skip ]]
450
916
 
451
917
  const { display, target } = parseLink(inner);
452
918
  const linkToken: LinkToken = {
@@ -504,7 +970,7 @@ export function tokenize(input: string): Token[] {
504
970
  continue;
505
971
  }
506
972
  // Complex expression — scan for balanced closing }
507
- const closeIdx$ = scanBalancedBrace(input, nameStart);
973
+ const closeIdx$ = scanBalancedBrace(input, nameStart - 1, memo);
508
974
  if (closeIdx$ !== -1) {
509
975
  const expression = input.slice(afterSelectors, closeIdx$);
510
976
  i = closeIdx$ + 1;
@@ -549,7 +1015,7 @@ export function tokenize(input: string): Token[] {
549
1015
  continue;
550
1016
  }
551
1017
  // Complex expression — scan for balanced closing }
552
- const closeIdx_ = scanBalancedBrace(input, nameStart);
1018
+ const closeIdx_ = scanBalancedBrace(input, nameStart - 1, memo);
553
1019
  if (closeIdx_ !== -1) {
554
1020
  const expression = input.slice(afterSelectors, closeIdx_);
555
1021
  i = closeIdx_ + 1;
@@ -594,7 +1060,7 @@ export function tokenize(input: string): Token[] {
594
1060
  continue;
595
1061
  }
596
1062
  // Complex expression — scan for balanced closing }
597
- const closeIdx_at = scanBalancedBrace(input, nameStart);
1063
+ const closeIdx_at = scanBalancedBrace(input, nameStart - 1, memo);
598
1064
  if (closeIdx_at !== -1) {
599
1065
  const expression = input.slice(afterSelectors, closeIdx_at);
600
1066
  i = closeIdx_at + 1;
@@ -639,7 +1105,7 @@ export function tokenize(input: string): Token[] {
639
1105
  continue;
640
1106
  }
641
1107
  // Complex expression — scan for balanced closing }
642
- const closeIdx_pct = scanBalancedBrace(input, nameStart);
1108
+ const closeIdx_pct = scanBalancedBrace(input, nameStart - 1, memo);
643
1109
  if (closeIdx_pct !== -1) {
644
1110
  const expression = input.slice(afterSelectors, closeIdx_pct);
645
1111
  i = closeIdx_pct + 1;
@@ -661,11 +1127,18 @@ export function tokenize(input: string): Token[] {
661
1127
  continue;
662
1128
  }
663
1129
 
1130
+ if (
1131
+ EXPRESSION_START.has(charAfter!) &&
1132
+ pushExpression(afterSelectors, start, false, className, id)
1133
+ ) {
1134
+ continue;
1135
+ }
1136
+
664
1137
  if (charAfter !== undefined && /[a-zA-Z]/.test(charAfter)) {
665
1138
  // {.class#id macroName args}
666
1139
  // Scan to closing }, tracking brace nesting and string literals
667
1140
  const contentStart = afterSelectors;
668
- const closeIdx = scanBalancedBrace(input, contentStart);
1141
+ const closeIdx = scanMacroClose(input, contentStart, memo);
669
1142
 
670
1143
  if (closeIdx === -1) {
671
1144
  i = start + 1;
@@ -720,7 +1193,7 @@ export function tokenize(input: string): Token[] {
720
1193
  continue;
721
1194
  }
722
1195
  // Complex expression — scan for balanced closing }
723
- const closeIdx = scanBalancedBrace(input, nameStart);
1196
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
724
1197
  if (closeIdx !== -1) {
725
1198
  const expression = input.slice(start + 1, closeIdx);
726
1199
  i = closeIdx + 1;
@@ -760,7 +1233,7 @@ export function tokenize(input: string): Token[] {
760
1233
  continue;
761
1234
  }
762
1235
  // Complex expression — scan for balanced closing }
763
- const closeIdx = scanBalancedBrace(input, nameStart);
1236
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
764
1237
  if (closeIdx !== -1) {
765
1238
  const expression = input.slice(start + 1, closeIdx);
766
1239
  i = closeIdx + 1;
@@ -800,7 +1273,7 @@ export function tokenize(input: string): Token[] {
800
1273
  continue;
801
1274
  }
802
1275
  // Complex expression — scan for balanced closing }
803
- const closeIdx = scanBalancedBrace(input, nameStart);
1276
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
804
1277
  if (closeIdx !== -1) {
805
1278
  const expression = input.slice(start + 1, closeIdx);
806
1279
  i = closeIdx + 1;
@@ -840,7 +1313,7 @@ export function tokenize(input: string): Token[] {
840
1313
  continue;
841
1314
  }
842
1315
  // Complex expression — scan for balanced closing }
843
- const closeIdx = scanBalancedBrace(input, nameStart);
1316
+ const closeIdx = scanBalancedBrace(input, nameStart - 1, memo);
844
1317
  if (closeIdx !== -1) {
845
1318
  const expression = input.slice(start + 1, closeIdx);
846
1319
  i = closeIdx + 1;
@@ -859,6 +1332,13 @@ export function tokenize(input: string): Token[] {
859
1332
  continue;
860
1333
  }
861
1334
 
1335
+ // {(expr)} or {!expr}: an expression that doesn't start with a variable
1336
+ if (EXPRESSION_START.has(nextChar!)) {
1337
+ if (pushExpression(i + 1, start, true)) continue;
1338
+ i++;
1339
+ continue;
1340
+ }
1341
+
862
1342
  // {macro ...} or {/macro} — but not bare { that's just text
863
1343
  // Must start with a letter or /
864
1344
  if (
@@ -870,7 +1350,7 @@ export function tokenize(input: string): Token[] {
870
1350
  // Scan to closing }, tracking brace nesting (object literals)
871
1351
  // and string literals
872
1352
  const contentStart = i + 1;
873
- const closeIdx = scanBalancedBrace(input, contentStart);
1353
+ const closeIdx = scanMacroClose(input, contentStart, memo);
874
1354
 
875
1355
  if (closeIdx === -1) {
876
1356
  // Unclosed macro — treat as text
@@ -902,7 +1382,7 @@ export function tokenize(input: string): Token[] {
902
1382
  }
903
1383
 
904
1384
  // Check for < — HTML tag
905
- if (input[i] === '<') {
1385
+ if (!textMode && input[i] === '<') {
906
1386
  const start = i;
907
1387
  let j = i + 1;
908
1388
 
@@ -944,9 +1424,10 @@ export function tokenize(input: string): Token[] {
944
1424
  continue;
945
1425
  }
946
1426
  } else {
947
- // Opening or self-closing tag: parse attributes
948
- const parsed = parseHtmlAttributes(input, j);
949
- j = parsed.endIdx;
1427
+ // Opening or self-closing tag: find where its attributes end, and
1428
+ // read them only if the tag closes there
1429
+ const attrsStart = j;
1430
+ j = scanAttributes(input, attrsStart, memo);
950
1431
 
951
1432
  let isSelfClose = HTML_VOID_TAGS.has(tagLower);
952
1433
  if (input[j] === '/') {
@@ -960,7 +1441,8 @@ export function tokenize(input: string): Token[] {
960
1441
  tokens.push({
961
1442
  type: 'html',
962
1443
  tag,
963
- attributes: parsed.attributes,
1444
+ attributes: parseHtmlAttributes(input, attrsStart, memo)
1445
+ .attributes,
964
1446
  isClose: false,
965
1447
  isSelfClose,
966
1448
  start,