@rohal12/spindle 0.53.0 → 0.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,1112 +0,0 @@
1
- import {
2
- createJsScanCache,
3
- findCodeEnd,
4
- type JsScanCache,
5
- type Sigil,
6
- } from '../js-lexer';
7
- import { isCodeAttribute } from './code-attributes';
8
-
9
- /** The namespace a variable reference reads. */
10
- export type VariableScope = 'variable' | 'temporary' | 'local' | 'transient';
11
-
12
- /** Variable sigils: story ($), temporary (_), local (@), transient (%). */
13
- export const SIGIL_SCOPES: Readonly<Record<Sigil, VariableScope>> = {
14
- $: 'variable',
15
- _: 'temporary',
16
- '@': 'local',
17
- '%': 'transient',
18
- };
19
-
20
- /** The sigil of each variable scope. */
21
- export const SCOPE_SIGILS = Object.fromEntries(
22
- Object.entries(SIGIL_SCOPES).map(([sigil, scope]) => [scope, sigil]),
23
- ) as Readonly<Record<VariableScope, Sigil>>;
24
-
25
- const SIGIL_CHARS: ReadonlySet<string> = new Set(Object.keys(SIGIL_SCOPES));
26
-
27
- /** Whether `c` is a variable sigil. */
28
- export function isSigil(c: string | undefined): c is Sigil {
29
- return c !== undefined && SIGIL_CHARS.has(c);
30
- }
31
-
32
- /** The `.class#id` selectors written before a link, variable or macro. */
33
- export interface Selectors {
34
- className?: string;
35
- id?: string;
36
- }
37
-
38
- /** Copy the selectors set in `from` onto `target`, and return it. */
39
- export function withSelectors<T extends Selectors>(
40
- target: T,
41
- from: Selectors,
42
- ): T {
43
- if (from.className) target.className = from.className;
44
- if (from.id) target.id = from.id;
45
- return target;
46
- }
47
-
48
- /** Where a token is in the input: from `start` up to `end`. */
49
- interface Span {
50
- start: number;
51
- end: number;
52
- }
53
-
54
- export interface TextToken extends Span {
55
- type: 'text';
56
- value: string;
57
- }
58
-
59
- export interface LinkToken extends Span, Selectors {
60
- type: 'link';
61
- display: string;
62
- target: string;
63
- }
64
-
65
- export interface MacroToken extends Span, Selectors {
66
- type: 'macro';
67
- name: string;
68
- rawArgs: string;
69
- isClose: boolean;
70
- }
71
-
72
- export interface VariableToken extends Span, Selectors {
73
- type: 'variable';
74
- name: string;
75
- scope: VariableScope;
76
- }
77
-
78
- export interface ExpressionToken extends Span, Selectors {
79
- type: 'expression';
80
- expression: string;
81
- }
82
-
83
- export interface HtmlToken extends Span {
84
- type: 'html';
85
- tag: string;
86
- attributes: Record<string, string>;
87
- isClose: boolean;
88
- isSelfClose: boolean;
89
- }
90
-
91
- export type Token =
92
- | TextToken
93
- | LinkToken
94
- | MacroToken
95
- | VariableToken
96
- | ExpressionToken
97
- | HtmlToken;
98
-
99
- /** Tag name must start with a letter (covers standard and custom elements). */
100
- const VALID_TAG_START = /[a-zA-Z]/;
101
-
102
- /** HTML void elements: never have children or a closing tag. */
103
- const HTML_VOID_TAGS = new Set([
104
- 'area',
105
- 'base',
106
- 'br',
107
- 'col',
108
- 'embed',
109
- 'hr',
110
- 'img',
111
- 'input',
112
- 'link',
113
- 'meta',
114
- 'param',
115
- 'source',
116
- 'track',
117
- 'wbr',
118
- ]);
119
-
120
- /**
121
- * Characters other than a sigil that open an expression after `{`:
122
- * `{(Math.max($a, 0))}`, `{!$done}`. Other characters that can start an
123
- * expression (quotes, digits, `[`, `-`) are left out, because braces around
124
- * them are common as literal text (JSON, regex quantifiers, `{[[link]]}`).
125
- */
126
- const EXPRESSION_START = new Set(['(', '!']);
127
-
128
- /** Macros whose body is JavaScript source, kept verbatim instead of tokenized. */
129
- const RAW_BODY_MACROS = new Set(['do']);
130
-
131
- /**
132
- * Link separators, in the order they are tried, and whether the target
133
- * comes first: display|target, display->target, target<-display.
134
- */
135
- const LINK_SEPARATORS: readonly (readonly [string, boolean])[] = [
136
- ['|', false],
137
- ['->', false],
138
- ['<-', true],
139
- ];
140
-
141
- /**
142
- * Parse a Twine link interior into display and target.
143
- * Supports: display|target, display->target, target<-display, plain
144
- */
145
- function parseLink(inner: string): { display: string; target: string } {
146
- for (const [separator, targetFirst] of LINK_SEPARATORS) {
147
- const idx = inner.indexOf(separator);
148
- if (idx === -1) continue;
149
- const before = inner.slice(0, idx).trim();
150
- const after = inner.slice(idx + separator.length).trim();
151
- return targetFirst
152
- ? { display: after, target: before }
153
- : { display: before, target: after };
154
- }
155
-
156
- // Plain: [[passage]]
157
- const trimmed = inner.trim();
158
- return { display: trimmed, target: trimmed };
159
- }
160
-
161
- /**
162
- * Parse a macro opening: extract name and rawArgs.
163
- * e.g. "set $x = 5" → { name: "set", rawArgs: "$x = 5" }
164
- * e.g. "/if" → { name: "if", rawArgs: "", isClose: true }
165
- * e.g. "elseif $x > 3" → { name: "elseif", rawArgs: "$x > 3" }
166
- */
167
- function parseMacroContent(content: string): {
168
- name: string;
169
- rawArgs: string;
170
- isClose: boolean;
171
- } {
172
- const trimmed = content.trim();
173
- const isClose = trimmed.startsWith('/');
174
- const rest = isClose ? trimmed.slice(1) : trimmed;
175
-
176
- const spaceIdx = rest.search(/\s/);
177
- if (spaceIdx === -1) {
178
- return { name: rest, rawArgs: '', isClose };
179
- }
180
-
181
- return {
182
- name: rest.slice(0, spaceIdx),
183
- rawArgs: rest.slice(spaceIdx + 1).trim(),
184
- isClose,
185
- };
186
- }
187
-
188
- /**
189
- * Parse CSS selectors: .foo.bar#baz → { className: "foo bar", id: "baz" }
190
- * Scans .[a-zA-Z0-9_-]+ and #[a-zA-Z0-9_-]+ segments in any order, and one
191
- * space after them. Returns space-joined class string, last id wins (each
192
- * left out if empty), and the position after them: `startIdx` if there are
193
- * none.
194
- */
195
- function parseSelectors(
196
- input: string,
197
- startIdx: number,
198
- ): { selectors: Selectors; end: number } {
199
- const classes: string[] = [];
200
- let id = '';
201
- let i = startIdx;
202
-
203
- while (i < input.length && (input[i] === '.' || input[i] === '#')) {
204
- const prefix = input[i]!;
205
- i++; // skip the . or #
206
- let name = '';
207
- while (i < input.length) {
208
- if (/[a-zA-Z0-9_-]/.test(input[i]!)) {
209
- name += input[i];
210
- i++;
211
- } else if (input[i] === '{' && isSigil(input[i + 1])) {
212
- // Consume interpolation: {$var}, {_var}, {@var}, {%var} (with optional dot paths)
213
- const braceStart = i;
214
- i += 2; // skip { and prefix
215
- while (i < input.length && /[\w.]/.test(input[i]!)) i++;
216
- if (i < input.length && input[i] === '}') {
217
- i++; // skip }
218
- name += input.slice(braceStart, i);
219
- } else {
220
- // Not a valid interpolation — stop
221
- i = braceStart;
222
- break;
223
- }
224
- } else {
225
- break;
226
- }
227
- }
228
- if (name) {
229
- if (prefix === '.') {
230
- classes.push(name);
231
- } else {
232
- id = name;
233
- }
234
- }
235
- }
236
-
237
- // Consume the space after the selectors
238
- if (i > startIdx && input[i] === ' ') i++;
239
- return {
240
- selectors: withSelectors<Selectors>(
241
- {},
242
- { className: classes.join(' '), id },
243
- ),
244
- end: i,
245
- };
246
- }
247
-
248
- /**
249
- * Parse HTML attributes from a string starting at position j.
250
- * Returns the attributes and the position after the last attribute.
251
- */
252
- function parseHtmlAttributes(
253
- input: string,
254
- j: number,
255
- memo: ScanMemo,
256
- ): { attributes: Record<string, string>; endIdx: number } {
257
- const attributes: Record<string, string> = {};
258
- // As in HTML, the first of attributes with the same (case-insensitive)
259
- // name wins. Defined as own properties so `__proto__` is kept too.
260
- const seen = new Set<string>();
261
- const endIdx = scanAttributes(input, j, memo, (name, value) => {
262
- const lower = name.toLowerCase();
263
- if (seen.has(lower)) return;
264
- seen.add(lower);
265
- Object.defineProperty(attributes, name, {
266
- value,
267
- enumerable: true,
268
- writable: true,
269
- configurable: true,
270
- });
271
- });
272
- return { attributes, endIdx };
273
- }
274
-
275
- /**
276
- * Index just past the attributes of a tag that start at `j` (after its
277
- * name), where its `>` or `/>` goes. Pass the same memo to repeated scans of
278
- * one input to share their work.
279
- */
280
- export function scanTagAttributes(
281
- input: string,
282
- j: number,
283
- memo: ScanMemo = createScanMemo(),
284
- ): number {
285
- return scanAttributes(input, j, memo);
286
- }
287
-
288
- /**
289
- * Scan the attributes of a tag from position j, handing each to `add` when
290
- * given. Returns the position after the last attribute.
291
- *
292
- * Where the attributes end depends only on where an attribute starts, so
293
- * without `add` the result is recorded for every attribute start passed (in
294
- * `memo.tag`) and a recorded one is used instead of scanning on. An unquoted
295
- * value takes in a `<`, so in `<a x=<a x=<a x=…` each opener's scan would
296
- * otherwise run over all the openers after it, taking quadratic time. A tag
297
- * whose attributes end in its `>` is consumed, so collecting the attributes
298
- * of those (with `add`, unrecorded) reads each character once.
299
- */
300
- function scanAttributes(
301
- input: string,
302
- j: number,
303
- memo: ScanMemo,
304
- add?: (name: string, value: string) => void,
305
- ): number {
306
- const passed: number[] = [];
307
- while (j < input.length) {
308
- // Skip whitespace
309
- while (j < input.length && /\s/.test(input[j]!)) j++;
310
- if (!add) {
311
- const known = memo.tag.get(j);
312
- if (known !== undefined) {
313
- j = known;
314
- break;
315
- }
316
- passed.push(j);
317
- }
318
- // End of tag?
319
- if (
320
- j >= input.length ||
321
- input[j] === '>' ||
322
- (input[j] === '/' && input[j + 1] === '>')
323
- )
324
- break;
325
-
326
- // Read attribute name
327
- const attrStart = j;
328
- while (j < input.length && /[a-zA-Z0-9_\-:@]/.test(input[j]!)) j++;
329
- const attrName = input.slice(attrStart, j);
330
- if (!attrName) break;
331
-
332
- // Check for = value. HTML allows whitespace on either side of the =
333
- // (`id = "x"`); whitespace not followed by = ends a boolean attribute.
334
- let eqIdx = j;
335
- while (eqIdx < input.length && /\s/.test(input[eqIdx]!)) eqIdx++;
336
- if (input[eqIdx] === '=') {
337
- j = eqIdx + 1; // skip =
338
- while (j < input.length && /\s/.test(input[j]!)) j++;
339
- if (input[j] === '"' || input[j] === "'") {
340
- const quote = input[j]!;
341
- j++; // skip opening quote
342
- const valStart = j;
343
- j = scanQuotedValue(input, j, quote, isCodeAttribute(attrName), memo);
344
- add?.(attrName, input.slice(valStart, j));
345
- if (j < input.length) j++; // skip closing quote
346
- } else {
347
- // Unquoted value, up to whitespace or `>`. It takes in a `<`, so in
348
- // `<a:=<a:=<a:=…` it holds every tag after it: the last run of such
349
- // characters found is kept, as each tag's value ends where it does.
350
- const valStart = j;
351
- const run = memo.unquoted;
352
- if (j < run.from || j > run.to) {
353
- run.from = j;
354
- while (j < input.length && /[^\s>]/.test(input[j]!)) j++;
355
- run.to = j;
356
- }
357
- j = run.to;
358
- add?.(attrName, input.slice(valStart, j));
359
- }
360
- } else {
361
- // Boolean attribute
362
- add?.(attrName, '');
363
- }
364
- }
365
- for (const at of passed) memo.tag.set(at, j);
366
- return j;
367
- }
368
-
369
- /**
370
- * Index of the quote ending the attribute value that starts at `j`, or the
371
- * end of the input. A `{…}` interpolation in it is skipped whole, so quotes
372
- * inside it don't end the value. A code attribute's value is no markup
373
- * (`isCodeAttribute`): only `{` and a sigil open a reference, other braces
374
- * and backslashes are text, as `splitSigilTemplate` reads them.
375
- *
376
- * The value reads the same from just past an interpolation, however the scan
377
- * got there, so the end is recorded there (in `memo.value`), and a recorded
378
- * one is used. An unclosed value that skips interpolations to the end of
379
- * the input (`<a x="}<a x={<a x="}…`) is then read once, not once per tag.
380
- */
381
- function scanQuotedValue(
382
- input: string,
383
- j: number,
384
- quote: string,
385
- code: boolean,
386
- memo: ScanMemo,
387
- ): number {
388
- const variant = (quote === '"' ? 0 : 2) + (code ? 1 : 0);
389
- const passed: number[] = [];
390
- let checkpoint = true;
391
- while (j < input.length) {
392
- if (checkpoint) {
393
- const known = memo.value.get(j * 4 + variant);
394
- if (known !== undefined) {
395
- j = known;
396
- break;
397
- }
398
- passed.push(j * 4 + variant);
399
- checkpoint = false;
400
- }
401
- if (!code && input[j] === '\\') {
402
- // A brace after an odd backslash run is escaped (`\{`) and opens
403
- // no interpolation, as in passage text.
404
- let k = j + 1;
405
- while (input[k] === '\\') k++;
406
- const brace = input[k] === '{' || input[k] === '}';
407
- j = brace && (k - j) % 2 === 1 ? k + 1 : k;
408
- continue;
409
- }
410
- if (input[j] === '{') {
411
- const closeIdx = !code
412
- ? scanBlockClose(input, j, memo)
413
- : isSigil(input[j + 1])
414
- ? scanBalancedBrace(input, j + 1, memo)
415
- : -1;
416
- if (closeIdx !== -1) {
417
- j = closeIdx + 1;
418
- checkpoint = true;
419
- continue;
420
- }
421
- } else if (input[j] === quote) break;
422
- j++;
423
- }
424
- for (const at of passed) memo.value.set(at, j);
425
- return j;
426
- }
427
-
428
- /**
429
- * Results of the brace and template scans of one input, shared by repeated
430
- * scans of it. A scan's result depends only on the input and where it
431
- * starts, so caching it is exact. Without the cache, each unclosed template
432
- * literal was scanned once as a template and again as plain text, so nested
433
- * unclosed ones (`` {$a`${$a`${… ``) took time exponential in their depth;
434
- * and each unclosed `{` was scanned to the end of the input, so many of them
435
- * took quadratic time. A memo must only be reused for scans of the same input
436
- * string.
437
- */
438
- export interface ScanMemo {
439
- /** `scanBalancedBrace` results, by lenient start. */
440
- code: Map<number, number>;
441
- /** JavaScript lexer results. */
442
- js: JsScanCache;
443
- /** Lenient brace scan results. */
444
- brace: Map<number, number>;
445
- /** Lenient template literal scan results. */
446
- template: Map<number, number>;
447
- /** Lenient template literal scan results, by a point in its text. */
448
- templateText: Map<number, number>;
449
- /** The last macro name run found: no whitespace or } in [from, to). */
450
- name: { from: number; to: number };
451
- /** The last unquoted attribute value run: no whitespace or > in [from, to). */
452
- unquoted: { from: number; to: number };
453
- /** Link scan results (`scanLinkClose`). */
454
- link: Map<number, number>;
455
- /** Where the attributes of a tag end, by attribute start (`scanAttributes`). */
456
- tag: Map<number, number>;
457
- /** Where quoted attribute values end, by checkpoint (`scanQuotedValue`). */
458
- value: Map<number, number>;
459
- /**
460
- * The last search for a raw-body closer, by macro name: the first one
461
- * from `from` on is at `at` (-1 for none).
462
- */
463
- rawClose: Map<string, { from: number; at: number }>;
464
- }
465
-
466
- export function createScanMemo(): ScanMemo {
467
- return {
468
- code: new Map(),
469
- js: createJsScanCache(),
470
- brace: new Map(),
471
- template: new Map(),
472
- templateText: new Map(),
473
- name: { from: 0, to: -1 },
474
- unquoted: { from: 0, to: -1 },
475
- link: new Map(),
476
- tag: new Map(),
477
- value: new Map(),
478
- rawClose: new Map(),
479
- };
480
- }
481
-
482
- /**
483
- * Find the } closing the code that starts at position i: a `{$…}`
484
- * expression from its sigil on, an attribute interpolation. The code is
485
- * lexed as JavaScript, so only a } in code outside the brackets it opened
486
- * counts — not one in a string, template or regex literal or a comment —
487
- * and `/` after an operand divides.
488
- *
489
- * Text that is not well-formed JavaScript (an apostrophe, an unterminated
490
- * literal or comment, unbalanced brackets) is scanned leniently instead, as
491
- * prose-like macro arguments are: braces count except inside string and
492
- * template literals, and a quote that can't start a string (apostrophe,
493
- * escaped, not closed on its line) is text.
494
- *
495
- * Returns the index of the closing } or -1 if there is none. Pass the same
496
- * memo to repeated scans of one input to share their work.
497
- */
498
- export function scanBalancedBrace(
499
- input: string,
500
- i: number,
501
- memo: ScanMemo = createScanMemo(),
502
- ): number {
503
- return scanClose(input, i, i, memo);
504
- }
505
-
506
- /**
507
- * Index of the } closing the macro whose content (name, then arguments)
508
- * starts at `contentStart`, or -1. The name runs up to whitespace or the }
509
- * (as `parseMacroContent` reads it); the arguments after it are code. The
510
- * lenient scan covers the whole content, as it always has.
511
- */
512
- function scanMacroClose(
513
- input: string,
514
- contentStart: number,
515
- memo: ScanMemo,
516
- ): number {
517
- const run = memo.name;
518
- if (contentStart < run.from || contentStart > run.to) {
519
- let k = contentStart;
520
- while (k < input.length && input[k] !== '}' && !/\s/.test(input[k]!)) k++;
521
- run.from = contentStart;
522
- run.to = k;
523
- }
524
- return scanClose(input, run.to, contentStart, memo);
525
- }
526
-
527
- /**
528
- * Index of the } closing the `{…}` block opened at `open`, read as passage
529
- * text reads it: a macro (`{name args}`, `{/name}`) has its arguments lexed
530
- * after its name, an expression (`{$…}`, `{(…)}`, `{!…}`) from its first
531
- * character, after any `.class#id` selectors. Any other block is scanned as
532
- * code from just past the {. Returns -1 if it is unclosed.
533
- */
534
- function scanBlockClose(input: string, open: number, memo: ScanMemo): number {
535
- const at = parseSelectors(input, open + 1).end;
536
- const first = input[at];
537
- if (first !== undefined && (first === '/' || /[a-zA-Z]/.test(first))) {
538
- return scanMacroClose(input, at, memo);
539
- }
540
- return scanBalancedBrace(input, at, memo);
541
- }
542
-
543
- /** Lex the code from `codeStart`, else scan leniently from `lenientStart`. */
544
- function scanClose(
545
- input: string,
546
- codeStart: number,
547
- lenientStart: number,
548
- memo: ScanMemo,
549
- ): number {
550
- let end = memo.code.get(lenientStart);
551
- if (end === undefined) {
552
- end = findCodeEnd(input, codeStart, { cache: memo.js });
553
- if (end === -1) end = scanBraceLenient(input, lenientStart, memo);
554
- memo.code.set(lenientStart, end);
555
- }
556
- return end;
557
- }
558
-
559
- /**
560
- * Index of the ]] closing the link whose text starts at `from`, or -1. A
561
- * [[ inside opens a nested pair. Each nested [[ starts a scan that reads the
562
- * rest the same way, so its result is recorded too: unclosed [[s don't each
563
- * scan to the end of the input.
564
- */
565
- function scanLinkClose(input: string, from: number, memo: ScanMemo): number {
566
- const known = memo.link.get(from);
567
- if (known !== undefined) return known;
568
- const opens = [from];
569
- let i = from;
570
- while (i < input.length) {
571
- if (input[i] === '[' && input[i + 1] === '[') {
572
- i += 2;
573
- const inner = memo.link.get(i);
574
- if (inner === undefined) opens.push(i);
575
- else if (inner === -1)
576
- break; // nor does this one close
577
- else i = inner + 2;
578
- } else if (input[i] === ']' && input[i + 1] === ']') {
579
- memo.link.set(opens.pop()!, i);
580
- if (!opens.length) return i;
581
- i += 2;
582
- } else {
583
- i++;
584
- }
585
- }
586
- for (const open of opens) memo.link.set(open, -1);
587
- return -1;
588
- }
589
-
590
- /**
591
- * Skip a '…' or "…" string literal opening at i.
592
- * Returns the index just past the closing quote, or -1 if the string is
593
- * not closed on the same line (JS strings can't span lines unescaped).
594
- */
595
- function skipQuoted(input: string, i: number): number {
596
- const quote = input[i];
597
- let j = i + 1;
598
- while (j < input.length) {
599
- const c = input[j];
600
- if (c === '\\') j += 2;
601
- else if (c === quote) return j + 1;
602
- else if (c === '\n') return -1;
603
- else j++;
604
- }
605
- return -1;
606
- }
607
-
608
- /**
609
- * A quote directly after a letter/digit is an apostrophe (don't), not a
610
- * string; after a backslash it is an escaped attribute delimiter (\").
611
- */
612
- const NON_STRING_QUOTE_PREFIX = /[\p{L}\p{N}_\\]/u;
613
-
614
- /**
615
- * A brace or template literal scan in progress: braces from `start` (just
616
- * past a {) to their closing }, or a template literal from its backtick at
617
- * `start` to the closing one.
618
- */
619
- interface LenientScan {
620
- template: boolean;
621
- start: number;
622
- /**
623
- * For braces, per brace depth from the outermost: the checkpoints at that
624
- * depth whose scans end where the depth does.
625
- */
626
- levels: number[][];
627
- /**
628
- * For a template literal, the points in its text it passed (just past its
629
- * backtick, an escape or an interpolation), whose scans end where it does.
630
- */
631
- passed?: number[];
632
- }
633
-
634
- /**
635
- * Scan for the balanced closing } starting at position i (just past the {).
636
- * Braces inside string and template literals are ignored. A quote that
637
- * can't start a string (apostrophe, escaped, unterminated) counts as text,
638
- * and so does a backtick without a closing one.
639
- * Returns the index of the closing } or -1 if unbalanced.
640
- *
641
- * Template literals and their ${…} parts are scans on a stack, not
642
- * recursive calls, so deep nesting can't overflow the call stack. How the
643
- * text reads depends only on where a scan is, so a scan passing a point —
644
- * just past a {, a string or a template literal — goes on as a scan
645
- * starting there would: its result there is recorded too, and a recorded
646
- * result is used instead of scanning on. Scans from many starts in one
647
- * input then take about linear time.
648
- */
649
- function scanBraceLenient(input: string, i: number, memo: ScanMemo): number {
650
- const known = memo.brace.get(i);
651
- if (known !== undefined) return known;
652
- const stack: LenientScan[] = [{ template: false, start: i, levels: [[]] }];
653
- for (;;) {
654
- const scan = stack[stack.length - 1]!;
655
- let end: number | undefined; // set when `scan` is done
656
- if (scan.template) {
657
- // Template text reads the same from such a point, however the scan got
658
- // there: in `` `\`\`\`… `` each backtick a scan from an earlier one
659
- // reads as escaped starts a template that reads the same rest.
660
- let point = true;
661
- while (end === undefined && i < input.length) {
662
- if (point) {
663
- const known = memo.templateText.get(i);
664
- if (known !== undefined) {
665
- end = known;
666
- break;
667
- }
668
- scan.passed!.push(i);
669
- point = false;
670
- }
671
- const c = input[i];
672
- if (c === '\\') {
673
- i += 2;
674
- point = true;
675
- } else if (c === '`') {
676
- end = i + 1;
677
- } else if (c === '$' && input[i + 1] === '{') {
678
- const inner = memo.brace.get(i + 2);
679
- if (inner === undefined) break; // scan the ${…} first
680
- if (inner === -1) end = -1;
681
- else {
682
- i = inner + 1;
683
- point = true;
684
- }
685
- } else {
686
- i++;
687
- }
688
- }
689
- if (end === undefined && i < input.length) {
690
- stack.push({ template: false, start: i + 2, levels: [[]] });
691
- i += 2;
692
- continue;
693
- }
694
- end ??= -1;
695
- } else {
696
- const { levels } = scan;
697
- /** The } at `close` ends the innermost level. */
698
- const closeLevel = (close: number) => {
699
- for (const at of levels.pop()!) memo.brace.set(at, close);
700
- if (!levels.length) end = close;
701
- };
702
- while (end === undefined && i < input.length) {
703
- const c = input[i]!;
704
- let at = -1; // a checkpoint, if one starts here
705
- if (c === '{') {
706
- levels.push([]);
707
- at = ++i;
708
- } else if (c === '}') {
709
- closeLevel(i++);
710
- } else if (
711
- (c === '"' || c === "'") &&
712
- !(i > 0 && NON_STRING_QUOTE_PREFIX.test(input[i - 1]!))
713
- ) {
714
- const close = skipQuoted(input, i);
715
- if (close === -1) i++;
716
- else at = i = close;
717
- } else if (c === '`') {
718
- const close = memo.template.get(i);
719
- if (close === undefined) break; // scan the template first
720
- if (close === -1) i++;
721
- else at = i = close;
722
- } else {
723
- i++;
724
- }
725
- if (at === -1) continue;
726
- const known = memo.brace.get(at);
727
- if (known === undefined) {
728
- levels[levels.length - 1]!.push(at);
729
- } else if (known === -1) {
730
- // The innermost level never closes, so neither do the others
731
- i = input.length;
732
- } else {
733
- closeLevel(known);
734
- i = known + 1;
735
- }
736
- }
737
- if (end === undefined && i < input.length) {
738
- stack.push({ template: true, start: i, levels: [], passed: [] });
739
- i++;
740
- continue;
741
- }
742
- if (end === undefined) {
743
- end = -1;
744
- for (const level of levels) {
745
- for (const at of level) memo.brace.set(at, -1);
746
- }
747
- }
748
- }
749
- // `scan` is done: hand its result to the scan that started it
750
- stack.pop();
751
- (scan.template ? memo.template : memo.brace).set(scan.start, end);
752
- for (const at of scan.passed ?? []) memo.templateText.set(at, end);
753
- const parent = stack[stack.length - 1];
754
- if (!parent) return end;
755
- if (parent.template) {
756
- // An unclosed ${…} leaves the template unclosed: go on to its end
757
- i = end === -1 ? input.length : end + 1;
758
- } else {
759
- // An unclosed template's backtick is text
760
- i = end === -1 ? scan.start + 1 : end;
761
- }
762
- }
763
- }
764
-
765
- /**
766
- * Options for {@link tokenize}.
767
- */
768
- export interface TokenizeOptions {
769
- /**
770
- * Text mode, for markup that becomes a string (HTML attribute values,
771
- * macro labels): only `{…}` markup and brace escapes are recognized, while
772
- * `[[` and `<` are text. With no markdown to pair up the backslashes of a
773
- * run before a brace, they are paired up here: `\\{` is one backslash
774
- * before a live brace, `\\\{` one before a literal one.
775
- */
776
- text?: boolean;
777
- }
778
-
779
- /**
780
- * Single-pass tokenizer for Twine passage content.
781
- * Recognizes: [[links]], {$variable}, {_temporary}, {macroName args}
782
- */
783
- export function tokenize(
784
- input: string,
785
- options: TokenizeOptions = {},
786
- ): Token[] {
787
- const textMode = options.text === true;
788
- const tokens: Token[] = [];
789
- const memo = createScanMemo();
790
- let i = 0;
791
- let textStart = 0;
792
-
793
- /** Push a text token for the text not yet tokenized before `end`. */
794
- function flushText(end: number) {
795
- if (end > textStart) {
796
- tokens.push({
797
- type: 'text',
798
- value: input.slice(textStart, end),
799
- start: textStart,
800
- end,
801
- });
802
- textStart = end;
803
- }
804
- }
805
-
806
- /** Push `token` and go on after it. */
807
- function emit(token: Token) {
808
- tokens.push(token);
809
- i = textStart = token.end;
810
- }
811
-
812
- /**
813
- * After an opening raw-body macro ({do}), emit everything up to its
814
- * {/name} as a single text token so JavaScript source is not parsed as
815
- * markup. The body is lexed as JavaScript statements: a {/do} in code ends
816
- * it, one inside a string, template or regex literal or a comment does
817
- * not. If the body is not well-formed JavaScript up to a {/do} in code,
818
- * the first {/do} ends it. Without a closer, nothing is consumed and the
819
- * AST builder reports it.
820
- */
821
- function consumeRawBody(name: string, isClose: boolean) {
822
- const lower = name.toLowerCase();
823
- if (isClose || !RAW_BODY_MACROS.has(lower)) return;
824
- const closer = `\\{/${lower}\\s*\\}`;
825
- // The first closer from here on. Without one from `from` on there is
826
- // none later either, so many unclosed {do}s don't each search the rest.
827
- let last = memo.rawClose.get(lower);
828
- if (!last || i < last.from || (last.at >= 0 && i > last.at)) {
829
- const firstRe = new RegExp(closer, 'gi');
830
- firstRe.lastIndex = i;
831
- last = { from: i, at: firstRe.exec(input)?.index ?? -1 };
832
- memo.rawClose.set(lower, last);
833
- }
834
- const first = last.at;
835
- if (first === -1) return;
836
- const atRe = new RegExp(closer, 'iy');
837
- const closerAt = (k: number) => {
838
- atRe.lastIndex = k;
839
- return atRe.test(input);
840
- };
841
- let closeStart = findCodeEnd(input, i, {
842
- goal: 'statements',
843
- stop: closerAt,
844
- stopKey: lower,
845
- cache: memo.js,
846
- });
847
- if (closeStart === -1) closeStart = first;
848
- atRe.lastIndex = closeStart;
849
- const closeEnd = closeStart + atRe.exec(input)![0].length;
850
- flushText(closeStart);
851
- emit({
852
- type: 'macro',
853
- ...parseMacroContent(input.slice(closeStart + 1, closeEnd - 1)),
854
- start: closeStart,
855
- end: closeEnd,
856
- });
857
- }
858
-
859
- /**
860
- * Push an expression token for the `{…}` block opened at `start` whose
861
- * expression starts at `exprStart`, flushing the text before it first.
862
- * Returns false, consuming nothing, if the block is unclosed.
863
- */
864
- function pushExpression(
865
- exprStart: number,
866
- start: number,
867
- selectors: Selectors,
868
- ): boolean {
869
- const closeIdx = scanBalancedBrace(input, exprStart, memo);
870
- if (closeIdx === -1) return false;
871
- flushText(start);
872
- emit(
873
- withSelectors<ExpressionToken>(
874
- {
875
- type: 'expression',
876
- expression: input.slice(exprStart, closeIdx),
877
- start,
878
- end: closeIdx + 1,
879
- },
880
- selectors,
881
- ),
882
- );
883
- return true;
884
- }
885
-
886
- /**
887
- * Push a variable token for the `{…}` block opened at `start` whose sigil
888
- * is at `sigilAt`: `{$name}`, `{_name.field.subfield}`. Anything else
889
- * after the name makes the block an expression from the sigil on
890
- * (`{$expr[...]}`). Returns false, consuming nothing, if it is unclosed.
891
- */
892
- function pushVariable(
893
- sigilAt: number,
894
- start: number,
895
- selectors: Selectors,
896
- ): boolean {
897
- let nameEnd = sigilAt + 1;
898
- while (nameEnd < input.length && /[\w.]/.test(input[nameEnd]!)) nameEnd++;
899
- if (input[nameEnd] !== '}') {
900
- // Complex expression — scan for balanced closing }
901
- return pushExpression(sigilAt, start, selectors);
902
- }
903
- emit(
904
- withSelectors<VariableToken>(
905
- {
906
- type: 'variable',
907
- name: input.slice(sigilAt + 1, nameEnd),
908
- scope: SIGIL_SCOPES[input[sigilAt] as Sigil],
909
- start,
910
- end: nameEnd + 1,
911
- },
912
- selectors,
913
- ),
914
- );
915
- return true;
916
- }
917
-
918
- /**
919
- * Push a macro token for the `{…}` block opened at `start` whose content
920
- * (name, then arguments) starts at `contentStart`, then consume the body
921
- * of a raw-body macro. Returns false, consuming nothing, if the block is
922
- * unclosed.
923
- */
924
- function pushMacro(
925
- contentStart: number,
926
- start: number,
927
- selectors: Selectors,
928
- ): boolean {
929
- // Scan to closing }, tracking brace nesting (object literals)
930
- // and string literals
931
- const closeIdx = scanMacroClose(input, contentStart, memo);
932
- if (closeIdx === -1) return false;
933
- const token = withSelectors<MacroToken>(
934
- {
935
- type: 'macro',
936
- ...parseMacroContent(input.slice(contentStart, closeIdx)),
937
- start,
938
- end: closeIdx + 1,
939
- },
940
- selectors,
941
- );
942
- emit(token);
943
- consumeRawBody(token.name, token.isClose);
944
- return true;
945
- }
946
-
947
- while (i < input.length) {
948
- // Escaped braces: \{ and \}. Count the whole backslash run so \\{ is a
949
- // backslash pair before a live brace. In an odd run the last backslash
950
- // escapes the brace; the even rest stays text, which markdown collapses
951
- // pair by pair like any other \\ in the passage.
952
- if (input[i] === '\\') {
953
- let k = i + 1;
954
- while (input[k] === '\\') k++;
955
- const next = input[k];
956
- if (textMode && (next === '{' || next === '}')) {
957
- const escaped = (k - i) % 2 === 1;
958
- const end = escaped ? k + 1 : k;
959
- const value = '\\'.repeat((k - i) >> 1) + (escaped ? next : '');
960
- flushText(i);
961
- if (value) tokens.push({ type: 'text', value, start: i, end });
962
- i = textStart = end;
963
- continue;
964
- }
965
- if ((next === '{' || next === '}') && (k - i) % 2 === 1) {
966
- flushText(k - 1);
967
- emit({ type: 'text', value: next, start: k - 1, end: k + 1 });
968
- continue;
969
- }
970
- i = k;
971
- continue;
972
- }
973
-
974
- // Check for [[ link, with optional .class or #id syntax after [[
975
- if (!textMode && input[i] === '[' && input[i + 1] === '[') {
976
- flushText(i);
977
- const start = i;
978
- const { selectors, end: innerStart } = parseSelectors(input, i + 2);
979
-
980
- // Find closing ]]
981
- const closeIdx = scanLinkClose(input, innerStart, memo);
982
- if (closeIdx === -1) {
983
- // Unclosed link — treat as text
984
- i = start + 2;
985
- continue;
986
- }
987
-
988
- emit(
989
- withSelectors<LinkToken>(
990
- {
991
- type: 'link',
992
- ...parseLink(input.slice(innerStart, closeIdx)),
993
- start,
994
- end: closeIdx + 2, // skip ]]
995
- },
996
- selectors,
997
- ),
998
- );
999
- continue;
1000
- }
1001
-
1002
- // Check for { — variable, expression or macro, with an optional
1003
- // .class/#id prefix: {.foo#bar $var} or {#id.foo macroName ...}
1004
- if (input[i] === '{') {
1005
- const start = i;
1006
- const { selectors, end: at } = parseSelectors(input, i + 1);
1007
- const prefixed = at > i + 1;
1008
- // The text before a selector prefix ends there, whatever follows it
1009
- if (prefixed) flushText(start);
1010
- const c = input[at];
1011
-
1012
- if (isSigil(c)) {
1013
- // {$variable.field} or {_temporary} or {@local} or {%expr[...]}
1014
- flushText(start);
1015
- if (pushVariable(at, start, selectors)) continue;
1016
- } else if (EXPRESSION_START.has(c!)) {
1017
- // {(expr)} or {!expr}: an expression that doesn't start with a variable
1018
- if (pushExpression(at, start, selectors)) continue;
1019
- } else if (
1020
- c !== undefined &&
1021
- (/[a-zA-Z]/.test(c) || (c === '/' && !prefixed))
1022
- ) {
1023
- // {macro ...} or {/macro}; a closing tag takes no selectors
1024
- flushText(start);
1025
- if (pushMacro(at, start, selectors)) continue;
1026
- }
1027
-
1028
- // Unclosed, or a bare { — treat as regular text
1029
- i = start + 1;
1030
- continue;
1031
- }
1032
-
1033
- // Check for < — HTML tag
1034
- if (!textMode && input[i] === '<') {
1035
- const start = i;
1036
- let j = i + 1;
1037
-
1038
- // Closing tag?
1039
- const isClose = input[j] === '/';
1040
- if (isClose) j++;
1041
-
1042
- // Read tag name (letters, digits, hyphens for custom elements)
1043
- const tagStart = j;
1044
- while (j < input.length && /[a-zA-Z0-9-]/.test(input[j]!)) j++;
1045
- const tag = input.slice(tagStart, j);
1046
- const tagLower = tag.toLowerCase();
1047
-
1048
- // Valid tag name must start with a letter
1049
- if (tag && VALID_TAG_START.test(tag[0]!)) {
1050
- if (isClose) {
1051
- // Closing tag: skip whitespace, expect >
1052
- while (j < input.length && /\s/.test(input[j]!)) j++;
1053
- if (input[j] === '>') {
1054
- j++;
1055
- flushText(start);
1056
- if (HTML_VOID_TAGS.has(tagLower)) {
1057
- // Void elements never take a closer; drop a redundant </input>
1058
- i = textStart = j;
1059
- continue;
1060
- }
1061
- emit({
1062
- type: 'html',
1063
- tag,
1064
- attributes: {},
1065
- isClose: true,
1066
- isSelfClose: false,
1067
- start,
1068
- end: j,
1069
- });
1070
- continue;
1071
- }
1072
- } else {
1073
- // Opening or self-closing tag: find where its attributes end, and
1074
- // read them only if the tag closes there
1075
- const attrsStart = j;
1076
- j = scanAttributes(input, attrsStart, memo);
1077
-
1078
- let isSelfClose = HTML_VOID_TAGS.has(tagLower);
1079
- if (input[j] === '/') {
1080
- isSelfClose = true;
1081
- j++;
1082
- }
1083
-
1084
- if (input[j] === '>') {
1085
- j++;
1086
- flushText(start);
1087
- emit({
1088
- type: 'html',
1089
- tag,
1090
- attributes: parseHtmlAttributes(input, attrsStart, memo)
1091
- .attributes,
1092
- isClose: false,
1093
- isSelfClose,
1094
- start,
1095
- end: j,
1096
- });
1097
- continue;
1098
- }
1099
- }
1100
- }
1101
-
1102
- // Not a valid HTML tag — treat as text
1103
- i++;
1104
- continue;
1105
- }
1106
-
1107
- i++;
1108
- }
1109
-
1110
- flushText(input.length);
1111
- return tokens;
1112
- }