@fuzdev/fuz_ui 0.204.0 → 0.205.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (123) hide show
  1. package/dist/ApiModule.svelte +14 -1
  2. package/dist/ApiModule.svelte.d.ts.map +1 -1
  3. package/dist/Breadcrumb.svelte +2 -2
  4. package/dist/Card.svelte +1 -7
  5. package/dist/ColorSchemeInput.svelte +1 -2
  6. package/dist/ContextmenuEntry.svelte +1 -3
  7. package/dist/ContextmenuLinkEntry.svelte +1 -4
  8. package/dist/ContextmenuSeparator.svelte +1 -1
  9. package/dist/ContextmenuSubmenu.svelte +1 -2
  10. package/dist/CopyToClipboard.svelte +5 -5
  11. package/dist/CopyToClipboard.svelte.d.ts.map +1 -1
  12. package/dist/DeclarationDetail.svelte +15 -4
  13. package/dist/DeclarationDetail.svelte.d.ts.map +1 -1
  14. package/dist/Dialog.svelte +1 -2
  15. package/dist/DialogContent.svelte +1 -1
  16. package/dist/DocsList.svelte +1 -1
  17. package/dist/DocsMenu.svelte +1 -1
  18. package/dist/DocsPageLinks.svelte +9 -8
  19. package/dist/DocsPageLinks.svelte.d.ts.map +1 -1
  20. package/dist/DocsPrimaryNav.svelte +1 -1
  21. package/dist/DocsSecondaryNav.svelte +1 -1
  22. package/dist/EcosystemLinks.svelte +1 -1
  23. package/dist/LibraryDetail.svelte +10 -7
  24. package/dist/LibraryDetail.svelte.d.ts.map +1 -1
  25. package/dist/LibrarySummary.svelte +1 -2
  26. package/dist/PendingAnimation.svelte +6 -4
  27. package/dist/PendingAnimation.svelte.d.ts.map +1 -1
  28. package/dist/PendingButton.svelte +1 -2
  29. package/dist/StyleVariableButton.svelte +1 -2
  30. package/dist/Svg.svelte +1 -1
  31. package/dist/ThemeInput.svelte +1 -2
  32. package/dist/TomeSectionHeader.svelte +1 -1
  33. package/package.json +14 -38
  34. package/dist/Mdz.svelte +0 -34
  35. package/dist/Mdz.svelte.d.ts +0 -11
  36. package/dist/Mdz.svelte.d.ts.map +0 -1
  37. package/dist/MdzNodeView.svelte +0 -107
  38. package/dist/MdzNodeView.svelte.d.ts +0 -9
  39. package/dist/MdzNodeView.svelte.d.ts.map +0 -1
  40. package/dist/MdzPrecompiled.svelte +0 -30
  41. package/dist/MdzPrecompiled.svelte.d.ts +0 -11
  42. package/dist/MdzPrecompiled.svelte.d.ts.map +0 -1
  43. package/dist/MdzRoot.svelte +0 -30
  44. package/dist/MdzRoot.svelte.d.ts +0 -12
  45. package/dist/MdzRoot.svelte.d.ts.map +0 -1
  46. package/dist/MdzStream.svelte +0 -32
  47. package/dist/MdzStream.svelte.d.ts +0 -12
  48. package/dist/MdzStream.svelte.d.ts.map +0 -1
  49. package/dist/MdzStreamNodeView.svelte +0 -106
  50. package/dist/MdzStreamNodeView.svelte.d.ts +0 -9
  51. package/dist/MdzStreamNodeView.svelte.d.ts.map +0 -1
  52. package/dist/mdz.d.ts +0 -112
  53. package/dist/mdz.d.ts.map +0 -1
  54. package/dist/mdz.js +0 -1186
  55. package/dist/mdz_components.d.ts +0 -63
  56. package/dist/mdz_components.d.ts.map +0 -1
  57. package/dist/mdz_components.js +0 -32
  58. package/dist/mdz_helpers.d.ts +0 -164
  59. package/dist/mdz_helpers.d.ts.map +0 -1
  60. package/dist/mdz_helpers.js +0 -424
  61. package/dist/mdz_lexer.d.ts +0 -93
  62. package/dist/mdz_lexer.d.ts.map +0 -1
  63. package/dist/mdz_lexer.js +0 -732
  64. package/dist/mdz_opcodes.d.ts +0 -174
  65. package/dist/mdz_opcodes.d.ts.map +0 -1
  66. package/dist/mdz_opcodes.js +0 -14
  67. package/dist/mdz_opcodes_to_nodes.d.ts +0 -20
  68. package/dist/mdz_opcodes_to_nodes.d.ts.map +0 -1
  69. package/dist/mdz_opcodes_to_nodes.js +0 -332
  70. package/dist/mdz_stream_parser.d.ts +0 -80
  71. package/dist/mdz_stream_parser.d.ts.map +0 -1
  72. package/dist/mdz_stream_parser.js +0 -354
  73. package/dist/mdz_stream_parser_block.d.ts +0 -76
  74. package/dist/mdz_stream_parser_block.d.ts.map +0 -1
  75. package/dist/mdz_stream_parser_block.js +0 -458
  76. package/dist/mdz_stream_parser_inline.d.ts +0 -52
  77. package/dist/mdz_stream_parser_inline.d.ts.map +0 -1
  78. package/dist/mdz_stream_parser_inline.js +0 -378
  79. package/dist/mdz_stream_parser_link.d.ts +0 -22
  80. package/dist/mdz_stream_parser_link.d.ts.map +0 -1
  81. package/dist/mdz_stream_parser_link.js +0 -215
  82. package/dist/mdz_stream_parser_state.d.ts +0 -187
  83. package/dist/mdz_stream_parser_state.d.ts.map +0 -1
  84. package/dist/mdz_stream_parser_state.js +0 -398
  85. package/dist/mdz_stream_parser_text.d.ts +0 -14
  86. package/dist/mdz_stream_parser_text.d.ts.map +0 -1
  87. package/dist/mdz_stream_parser_text.js +0 -50
  88. package/dist/mdz_stream_parser_url.d.ts +0 -60
  89. package/dist/mdz_stream_parser_url.d.ts.map +0 -1
  90. package/dist/mdz_stream_parser_url.js +0 -307
  91. package/dist/mdz_stream_state.svelte.d.ts +0 -44
  92. package/dist/mdz_stream_state.svelte.d.ts.map +0 -1
  93. package/dist/mdz_stream_state.svelte.js +0 -357
  94. package/dist/mdz_to_svelte.d.ts +0 -41
  95. package/dist/mdz_to_svelte.d.ts.map +0 -1
  96. package/dist/mdz_to_svelte.js +0 -100
  97. package/dist/mdz_token_parser.d.ts +0 -14
  98. package/dist/mdz_token_parser.d.ts.map +0 -1
  99. package/dist/mdz_token_parser.js +0 -344
  100. package/dist/svelte_preprocess_mdz.d.ts +0 -65
  101. package/dist/svelte_preprocess_mdz.d.ts.map +0 -1
  102. package/dist/svelte_preprocess_mdz.js +0 -529
  103. package/dist/tsdoc_mdz.d.ts +0 -45
  104. package/dist/tsdoc_mdz.d.ts.map +0 -1
  105. package/dist/tsdoc_mdz.js +0 -88
  106. package/src/lib/mdz.ts +0 -1532
  107. package/src/lib/mdz_components.ts +0 -64
  108. package/src/lib/mdz_helpers.ts +0 -449
  109. package/src/lib/mdz_lexer.ts +0 -1012
  110. package/src/lib/mdz_opcodes.ts +0 -205
  111. package/src/lib/mdz_opcodes_to_nodes.ts +0 -375
  112. package/src/lib/mdz_stream_parser.ts +0 -415
  113. package/src/lib/mdz_stream_parser_block.ts +0 -532
  114. package/src/lib/mdz_stream_parser_inline.ts +0 -414
  115. package/src/lib/mdz_stream_parser_link.ts +0 -271
  116. package/src/lib/mdz_stream_parser_state.ts +0 -539
  117. package/src/lib/mdz_stream_parser_text.ts +0 -77
  118. package/src/lib/mdz_stream_parser_url.ts +0 -365
  119. package/src/lib/mdz_stream_state.svelte.ts +0 -387
  120. package/src/lib/mdz_to_svelte.ts +0 -141
  121. package/src/lib/mdz_token_parser.ts +0 -434
  122. package/src/lib/svelte_preprocess_mdz.ts +0 -742
  123. package/src/lib/tsdoc_mdz.ts +0 -97
package/src/lib/mdz.ts DELETED
@@ -1,1532 +0,0 @@
1
- /**
2
- * mdz - minimal markdown dialect for Fuz documentation.
3
- *
4
- * Parses an enhanced markdown dialect with:
5
- * - inline formatting: `code`, **bold**, _italic_, ~strikethrough~
6
- * - auto-detected links: external URLs (`https://...`) and internal paths (`/path`)
7
- * - markdown links: `[text](url)` with custom display text
8
- * - inline code in backticks (creates `Code` nodes; auto-linking to identifiers/modules
9
- * is handled by the rendering layer via `MdzNodeView.svelte`)
10
- * - paragraph breaks (double newline)
11
- * - block elements: headings, horizontal rules, code blocks
12
- * - HTML elements and Svelte components (opt-in via context)
13
- *
14
- * Key constraint: preserves ALL whitespace exactly as authored,
15
- * and is rendered with white-space pre or pre-wrap.
16
- *
17
- * ## Design philosophy
18
- *
19
- * - **False negatives over false positives**: When in doubt, treat as plain text.
20
- * Block elements can interrupt paragraphs without blank lines; inline formatting is strict.
21
- * - **One way to do things**: Single unambiguous syntax per feature. No alternatives.
22
- * - **Explicit over implicit**: Clear delimiters and column-0 requirements avoid ambiguity.
23
- * - **Simple over complete**: Prefer simple parsing rules over complex edge case handling.
24
- *
25
- * ## Status
26
- *
27
- * This is an early proof of concept with missing features and edge cases.
28
- *
29
- * @module
30
- */
31
-
32
- import {
33
- BACKTICK,
34
- ASTERISK,
35
- UNDERSCORE,
36
- TILDE,
37
- NEWLINE,
38
- HYPHEN,
39
- HASH,
40
- SPACE,
41
- TAB,
42
- LEFT_ANGLE,
43
- RIGHT_ANGLE,
44
- SLASH,
45
- LEFT_BRACKET,
46
- RIGHT_BRACKET,
47
- LEFT_PAREN,
48
- RIGHT_PAREN,
49
- A_UPPER,
50
- Z_UPPER,
51
- HR_HYPHEN_COUNT,
52
- MIN_CODEBLOCK_BACKTICKS,
53
- MAX_HEADING_LEVEL,
54
- match_url_prefix_case_insensitive,
55
- is_letter,
56
- is_tag_name_char,
57
- is_word_char,
58
- PERIOD,
59
- is_valid_path_char,
60
- trim_trailing_punctuation,
61
- is_at_absolute_path,
62
- is_at_relative_path,
63
- extract_single_tag,
64
- mdz_heading_id,
65
- mdz_is_url,
66
- mdz_is_safe_reference,
67
- mdz_push_merging_text,
68
- } from './mdz_helpers.js';
69
-
70
- // TODO design incremental parsing or some system that preserves Svelte components across re-renders when possible
71
-
72
- /**
73
- * Parses text to an array of `MdzNode`.
74
- */
75
- export const mdz_parse = (text: string): Array<MdzNode> => new MdzParser(text).parse();
76
-
77
- export type MdzNode =
78
- | MdzTextNode
79
- | MdzCodeNode
80
- | MdzCodeblockNode
81
- | MdzBoldNode
82
- | MdzItalicNode
83
- | MdzStrikethroughNode
84
- | MdzLinkNode
85
- | MdzParagraphNode
86
- | MdzHrNode
87
- | MdzHeadingNode
88
- | MdzElementNode
89
- | MdzComponentNode;
90
-
91
- export interface MdzBaseNode {
92
- type: string;
93
- start: number;
94
- end: number;
95
- }
96
-
97
- export interface MdzTextNode extends MdzBaseNode {
98
- type: 'Text';
99
- content: string;
100
- }
101
-
102
- export interface MdzCodeNode extends MdzBaseNode {
103
- type: 'Code';
104
- content: string; // The code content (identifier/module name)
105
- }
106
-
107
- export interface MdzCodeblockNode extends MdzBaseNode {
108
- type: 'Codeblock';
109
- lang: string | null; // language hint, if provided
110
- content: string; // raw code content
111
- }
112
-
113
- export interface MdzBoldNode extends MdzBaseNode {
114
- type: 'Bold';
115
- children: Array<MdzNode>;
116
- }
117
-
118
- export interface MdzItalicNode extends MdzBaseNode {
119
- type: 'Italic';
120
- children: Array<MdzNode>;
121
- }
122
-
123
- export interface MdzStrikethroughNode extends MdzBaseNode {
124
- type: 'Strikethrough';
125
- children: Array<MdzNode>;
126
- }
127
-
128
- export interface MdzLinkNode extends MdzBaseNode {
129
- type: 'Link';
130
- reference: string; // URL or path
131
- children: Array<MdzNode>; // Display content (can include inline formatting)
132
- link_type: 'external' | 'internal'; // external: https/http, internal: /path, ./path, ../path
133
- }
134
-
135
- export interface MdzParagraphNode extends MdzBaseNode {
136
- type: 'Paragraph';
137
- children: Array<MdzNode>;
138
- }
139
-
140
- export interface MdzHrNode extends MdzBaseNode {
141
- type: 'Hr';
142
- }
143
-
144
- export interface MdzHeadingNode extends MdzBaseNode {
145
- type: 'Heading';
146
- level: 1 | 2 | 3 | 4 | 5 | 6;
147
- id: string; // slugified heading text for fragment links
148
- children: Array<MdzNode>; // inline formatting allowed
149
- }
150
-
151
- export interface MdzElementNode extends MdzBaseNode {
152
- type: 'Element';
153
- name: string; // HTML element name (e.g., 'div', 'span', 'code')
154
- children: Array<MdzNode>;
155
- }
156
-
157
- export interface MdzComponentNode extends MdzBaseNode {
158
- type: 'Component';
159
- name: string; // Svelte component name (e.g., 'Alert', 'Card')
160
- children: Array<MdzNode>;
161
- }
162
-
163
- /**
164
- * Parser for mdz format.
165
- * Single-pass lexer/parser with text accumulation for efficiency.
166
- * Used by `mdz_parse`, which should be preferred for simple usage.
167
- */
168
- export class MdzParser {
169
- #index: number = 0;
170
- #template: string;
171
- #accumulated_text: string = '';
172
- #accumulated_start: number = 0;
173
- #nodes: Array<MdzNode> = [];
174
- #max_search_index: number = Number.MAX_SAFE_INTEGER; // Boundary for delimiter searches
175
-
176
- constructor(template: string) {
177
- this.#template = template;
178
- }
179
-
180
- /**
181
- * Main parse method. Returns flat array of nodes,
182
- * with paragraph nodes wrapping content between double newlines.
183
- *
184
- * Block elements (headings, HR, codeblocks) are detected at every column-0
185
- * position — they can interrupt paragraphs without requiring blank lines.
186
- */
187
- parse(): Array<MdzNode> {
188
- this.#nodes.length = 0;
189
- const root_nodes: Array<MdzNode> = [];
190
- const paragraph_children: Array<MdzNode> = [];
191
-
192
- // Skip leading newlines
193
- this.#skip_newlines();
194
-
195
- while (this.#index < this.#template.length) {
196
- // Block elements only start at column 0 — skip peek for mid-line characters
197
- const block_type =
198
- (this.#index === 0 || this.#template.charCodeAt(this.#index - 1) === NEWLINE) &&
199
- this.#peek_block_element();
200
- if (block_type) {
201
- const flushed = this.#flush_paragraph(paragraph_children, true);
202
- if (flushed) root_nodes.push(flushed);
203
- if (block_type === 'heading') root_nodes.push(this.#parse_heading());
204
- else if (block_type === 'hr') root_nodes.push(this.#parse_hr());
205
- else root_nodes.push(this.#parse_code_block());
206
- this.#skip_newlines();
207
- continue;
208
- }
209
-
210
- // Check for paragraph break (double newline)
211
- if (this.#is_at_paragraph_break()) {
212
- const flushed = this.#flush_paragraph(paragraph_children, true);
213
- if (flushed) root_nodes.push(flushed);
214
- this.#skip_newlines();
215
- continue;
216
- }
217
-
218
- // Parse inline content
219
- const node = this.#parse_node();
220
- if (node.type === 'Text') {
221
- this.#accumulate_text(node.content, node.start);
222
- } else {
223
- this.#flush_text();
224
- this.#nodes.push(node);
225
- }
226
- if (this.#nodes.length > 0) {
227
- paragraph_children.push(...this.#nodes);
228
- this.#nodes.length = 0;
229
- }
230
- }
231
-
232
- // Flush remaining content as final paragraph
233
- const final_paragraph = this.#flush_paragraph(paragraph_children, true);
234
- if (final_paragraph) root_nodes.push(final_paragraph);
235
-
236
- return root_nodes;
237
- }
238
-
239
- /**
240
- * Flush accumulated inline content as a paragraph node (or single tag).
241
- * When `trim_trailing` is true, trims trailing newlines from the last text node
242
- * (the line break before a block element or paragraph break).
243
- *
244
- * @mutates paragraph_children - trims last text node, removes whitespace-only nodes, clears array
245
- */
246
- #flush_paragraph(paragraph_children: Array<MdzNode>, trim_trailing = false): MdzNode | null {
247
- this.#flush_text();
248
- if (this.#nodes.length > 0) {
249
- paragraph_children.push(...this.#nodes);
250
- this.#nodes.length = 0;
251
- }
252
-
253
- if (paragraph_children.length === 0) {
254
- return null;
255
- }
256
-
257
- if (trim_trailing) {
258
- // Trim trailing newlines from the last text node (line break before block element)
259
- const last = paragraph_children[paragraph_children.length - 1]!;
260
- if (last.type === 'Text') {
261
- const trimmed = last.content.replace(/\n+$/, '');
262
- if (trimmed) {
263
- last.content = trimmed;
264
- last.end = last.start + trimmed.length;
265
- } else {
266
- paragraph_children.pop();
267
- }
268
- }
269
-
270
- // Skip whitespace-only paragraphs (e.g. newlines between consecutive blocks)
271
- const has_content = paragraph_children.some(
272
- (n) => n.type !== 'Text' || n.content.trim().length > 0,
273
- );
274
- if (!has_content) {
275
- paragraph_children.length = 0;
276
- return null;
277
- }
278
- }
279
-
280
- // Single tag (component/element) - add directly without paragraph wrapper (MDX convention)
281
- const single_tag = extract_single_tag(paragraph_children);
282
- if (single_tag) {
283
- paragraph_children.length = 0;
284
- return single_tag;
285
- }
286
-
287
- // Regular paragraph
288
- const result: MdzParagraphNode = {
289
- type: 'Paragraph',
290
- children: paragraph_children.slice(),
291
- start: paragraph_children[0]!.start,
292
- end: paragraph_children[paragraph_children.length - 1]!.end,
293
- };
294
- paragraph_children.length = 0;
295
- return result;
296
- }
297
-
298
- /**
299
- * Consume consecutive newline characters.
300
- */
301
- #skip_newlines(): void {
302
- while (
303
- this.#index < this.#template.length &&
304
- this.#template.charCodeAt(this.#index) === NEWLINE
305
- ) {
306
- this.#index++;
307
- }
308
- }
309
-
310
- /**
311
- * Accumulate text for later flushing (performance optimization).
312
- */
313
- #accumulate_text(text: string, start: number): void {
314
- if (this.#accumulated_text === '') {
315
- this.#accumulated_start = start;
316
- }
317
- this.#accumulated_text += text;
318
- }
319
-
320
- #flush_text(): void {
321
- if (this.#accumulated_text !== '') {
322
- this.#nodes.push({
323
- type: 'Text',
324
- content: this.#accumulated_text,
325
- start: this.#accumulated_start,
326
- end: this.#accumulated_start + this.#accumulated_text.length,
327
- });
328
- this.#accumulated_text = '';
329
- }
330
- }
331
-
332
- /**
333
- * Create a text node and advance index past the content.
334
- * Used when formatting delimiters fail to match and need to be treated as literal text.
335
- */
336
- #make_text_node(content: string, start: number): MdzTextNode {
337
- this.#index = start + content.length;
338
- return {
339
- type: 'Text',
340
- content,
341
- start,
342
- end: this.#index,
343
- };
344
- }
345
-
346
- /**
347
- * Parse next node based on current character.
348
- * Uses switch for performance (avoids regex in hot loop).
349
- */
350
- #parse_node(): MdzNode {
351
- const char_code = this.#template.charCodeAt(this.#index);
352
-
353
- // Use character codes for performance in hot path
354
- switch (char_code) {
355
- case BACKTICK:
356
- return this.#parse_code();
357
- case ASTERISK:
358
- return this.#parse_bold();
359
- case UNDERSCORE:
360
- return this.#parse_italic();
361
- case TILDE:
362
- return this.#parse_strikethrough();
363
- case LEFT_BRACKET:
364
- return this.#parse_markdown_link();
365
- case LEFT_ANGLE:
366
- return this.#parse_tag();
367
- default:
368
- return this.#parse_text();
369
- }
370
- }
371
-
372
- /**
373
- * Parse backtick code: `code`
374
- * Auto-links to identifiers/modules if match found.
375
- * Falls back to text if unclosed, empty, or if newline encountered before closing backtick.
376
- */
377
- #parse_code(): MdzCodeNode | MdzTextNode {
378
- const start = this.#index;
379
- this.#eat('`');
380
- const content_start = this.#index;
381
-
382
- // Find closing backtick, but stop at newline (respect boundary for greedy matching)
383
- let content_end = -1;
384
- const search_limit = Math.min(this.#max_search_index, this.#template.length);
385
- for (let i = this.#index; i < search_limit; i++) {
386
- const char_code = this.#template.charCodeAt(i);
387
- if (char_code === BACKTICK) {
388
- content_end = i;
389
- break;
390
- }
391
- if (char_code === NEWLINE) {
392
- // Newline before closing backtick - treat as unclosed
393
- break;
394
- }
395
- }
396
-
397
- if (content_end === -1) {
398
- // Unclosed backtick or newline encountered, treat as text
399
- return this.#make_text_node('`', start);
400
- }
401
-
402
- const content = this.#template.slice(content_start, content_end);
403
-
404
- // Empty inline code has no semantic meaning, treat as literal text
405
- if (content.length === 0) {
406
- return this.#make_text_node('``', start);
407
- }
408
-
409
- this.#index = content_end + 1;
410
-
411
- return {
412
- type: 'Code',
413
- content,
414
- start,
415
- end: this.#index,
416
- };
417
- }
418
-
419
- /**
420
- * Parse bold starting with double asterisk.
421
- *
422
- * - **bold** = Bold node
423
- *
424
- * Falls back to text if unclosed or single asterisk.
425
- *
426
- * Bold has no word boundary restrictions and works everywhere including intraword.
427
- * Examples:
428
- * - `foo**bar**baz` → foo<strong>bar</strong>baz (creates bold)
429
- * - `word **bold** word` → word <strong>bold</strong> word (also works)
430
- */
431
- #parse_bold(): MdzBoldNode | MdzTextNode {
432
- const start = this.#index;
433
-
434
- // Check for ** (bold)
435
- if (this.#match('**')) {
436
- // Bold (**) has no word boundary restrictions - works everywhere including intraword
437
- this.#eat('**');
438
-
439
- // Find closing ** (greedy matching - first occurrence within boundary)
440
- const search_end = Math.min(this.#max_search_index, this.#template.length);
441
- let close_index = this.#template.indexOf('**', this.#index);
442
- // Check if close_index exceeds search boundary
443
- if (close_index !== -1 && close_index >= search_end) {
444
- close_index = -1;
445
- }
446
- if (close_index === -1) {
447
- // Unclosed, treat as text
448
- return this.#make_text_node('**', start);
449
- }
450
-
451
- // No word boundary check for closing ** - works everywhere
452
- // Parse children up to closing delimiter (bounded parsing)
453
- const children = this.#parse_nodes_until('**', close_index);
454
-
455
- // Verify we're at the closing delimiter (could have stopped early due to paragraph break)
456
- if (!this.#match('**')) {
457
- // Interrupted before closing - treat as unclosed
458
- return this.#make_text_node('**', start);
459
- }
460
-
461
- // Empty bold has no semantic meaning, treat as literal text
462
- if (children.length === 0) {
463
- this.#index = start;
464
- return this.#make_text_node('****', start);
465
- }
466
-
467
- // Consume closing **
468
- this.#eat('**');
469
- return {
470
- type: 'Bold',
471
- children,
472
- start,
473
- end: this.#index,
474
- };
475
- }
476
-
477
- // Single asterisk - treat as text
478
- const content = this.#template[this.#index]!;
479
- return this.#make_text_node(content, start);
480
- }
481
-
482
- /**
483
- * Common parser for single-delimiter formatting (italic and strikethrough).
484
- * Both formats use identical parsing logic with different delimiters and node types.
485
- *
486
- * Word boundary requirements:
487
- * - Opening delimiter must be at word boundary (not preceded by alphanumeric)
488
- * - Closing delimiter must be at word boundary (not followed by alphanumeric)
489
- * - This prevents false positives with `snake_case` and `foo~bar~baz` text
490
- *
491
- * Falls back to literal text if:
492
- * - Delimiter not at word boundary
493
- * - Unclosed (no matching closing delimiter found)
494
- * - Empty content (e.g., `__` or `~~`)
495
- * - Paragraph break interrupts before closing delimiter
496
- *
497
- * @param delimiter - the delimiter character (`_` for italic, `~` for strikethrough)
498
- * @param node_type - the node type to create ('Italic' or 'Strikethrough')
499
- * @returns formatted node or text node if validation fails
500
- */
501
- #parse_single_delimiter_formatting(
502
- delimiter: '_',
503
- node_type: 'Italic',
504
- ): MdzItalicNode | MdzTextNode;
505
- #parse_single_delimiter_formatting(
506
- delimiter: '~',
507
- node_type: 'Strikethrough',
508
- ): MdzStrikethroughNode | MdzTextNode;
509
- #parse_single_delimiter_formatting(
510
- delimiter: '_' | '~',
511
- node_type: 'Italic' | 'Strikethrough',
512
- ): MdzItalicNode | MdzStrikethroughNode | MdzTextNode {
513
- const start = this.#index;
514
-
515
- // Check if opening delimiter is at word boundary
516
- if (!this.#is_at_word_boundary(this.#index, true, false)) {
517
- // Intraword delimiter - treat as literal text
518
- const content = this.#template[this.#index]!;
519
- return this.#make_text_node(content, start);
520
- }
521
-
522
- this.#eat(delimiter);
523
-
524
- // Find closing delimiter (greedy matching - first occurrence within boundary)
525
- const search_end = Math.min(this.#max_search_index, this.#template.length);
526
- let close_index = this.#template.indexOf(delimiter, this.#index);
527
- // Check if close_index exceeds search boundary
528
- if (close_index !== -1 && close_index >= search_end) {
529
- close_index = -1;
530
- }
531
- if (close_index === -1) {
532
- // Unclosed, treat as text
533
- return this.#make_text_node(delimiter, start);
534
- }
535
-
536
- // Check if closing delimiter is at word boundary
537
- if (!this.#is_at_word_boundary(close_index + 1, false, true)) {
538
- // Closing delimiter not at boundary - treat whole thing as text
539
- return this.#make_text_node(delimiter, start);
540
- }
541
-
542
- // Parse children up to closing delimiter (bounded parsing)
543
- const children = this.#parse_nodes_until(delimiter, close_index);
544
-
545
- // Verify we're at the closing delimiter (could have stopped early due to paragraph break)
546
- if (!this.#match(delimiter)) {
547
- // Interrupted before closing - treat as unclosed
548
- return this.#make_text_node(delimiter, start);
549
- }
550
-
551
- // Empty formatting has no semantic meaning, treat as literal text
552
- if (children.length === 0) {
553
- this.#index = start;
554
- return this.#make_text_node(delimiter + delimiter, start);
555
- }
556
-
557
- // Consume closing delimiter
558
- this.#eat(delimiter);
559
- return {
560
- type: node_type,
561
- children,
562
- start,
563
- end: this.#index,
564
- };
565
- }
566
-
567
- /**
568
- * Parse italic starting with underscore.
569
- * _italic_ = Italic node
570
- * Falls back to text if unclosed or not at word boundary.
571
- *
572
- * Following GFM spec: underscores cannot create emphasis in middle of words.
573
- * Examples:
574
- * - `foo_bar_baz` → literal text (intraword)
575
- * - `word _emphasis_ word` → emphasis (at word boundaries)
576
- */
577
- #parse_italic(): MdzItalicNode | MdzTextNode {
578
- return this.#parse_single_delimiter_formatting('_', 'Italic');
579
- }
580
-
581
- /**
582
- * Parse strikethrough starting with tilde.
583
- * ~strikethrough~ = Strikethrough node
584
- * Falls back to text if unclosed or not at word boundary.
585
- *
586
- * Following mdz philosophy (false negatives over false positives):
587
- * Strikethrough requires word boundaries to prevent intraword formatting.
588
- * Examples:
589
- * - `foo~bar~baz` → literal text (intraword)
590
- * - `word ~strike~ word` → strikethrough (at word boundaries)
591
- */
592
- #parse_strikethrough(): MdzStrikethroughNode | MdzTextNode {
593
- return this.#parse_single_delimiter_formatting('~', 'Strikethrough');
594
- }
595
-
596
- /**
597
- * Parse markdown link: `[text](url)`.
598
- * Falls back to text if malformed.
599
- */
600
- #parse_markdown_link(): MdzLinkNode | MdzTextNode {
601
- const start = this.#index;
602
-
603
- // Consume opening [
604
- if (!this.#match('[')) {
605
- const content = this.#template[this.#index]!;
606
- this.#index++;
607
- return {
608
- type: 'Text',
609
- content,
610
- start,
611
- end: this.#index,
612
- };
613
- }
614
- this.#index++;
615
-
616
- // Parse children nodes until closing ]
617
- const children = this.#parse_nodes_until(']');
618
-
619
- // Check if we found the closing ]
620
- if (!this.#match(']')) {
621
- // No closing ], treat as text
622
- this.#index = start + 1;
623
- return {
624
- type: 'Text',
625
- content: '[',
626
- start,
627
- end: this.#index,
628
- };
629
- }
630
- this.#index++; // consume ]
631
-
632
- // Check for opening (
633
- if (
634
- this.#index >= this.#template.length ||
635
- this.#template.charCodeAt(this.#index) !== LEFT_PAREN
636
- ) {
637
- // No opening (, treat as text
638
- this.#index = start + 1;
639
- return {
640
- type: 'Text',
641
- content: '[',
642
- start,
643
- end: this.#index,
644
- };
645
- }
646
- this.#index++;
647
-
648
- // Find closing )
649
- const close_paren = this.#template.indexOf(')', this.#index);
650
- if (close_paren === -1) {
651
- // No closing ), treat as text
652
- this.#index = start + 1;
653
- return {
654
- type: 'Text',
655
- content: '[',
656
- start,
657
- end: this.#index,
658
- };
659
- }
660
-
661
- // Extract URL/path
662
- const reference = this.#template.slice(this.#index, close_paren);
663
-
664
- // Validate reference is not empty or whitespace-only
665
- if (!reference.trim()) {
666
- // Empty reference, treat as text
667
- this.#index = start + 1;
668
- return {
669
- type: 'Text',
670
- content: '[',
671
- start,
672
- end: this.#index,
673
- };
674
- }
675
-
676
- // Validate all characters in reference are valid URI characters per RFC 3986
677
- // This prevents spaces and other invalid characters from being in markdown link URLs
678
- // Follows GFM behavior: invalid chars cause fallback to text, then auto-detection
679
- for (let i = 0; i < reference.length; i++) {
680
- const char_code = reference.charCodeAt(i);
681
- if (!is_valid_path_char(char_code)) {
682
- // Invalid character in URL, treat as text and let auto-detection handle it
683
- this.#index = start + 1;
684
- return {
685
- type: 'Text',
686
- content: '[',
687
- start,
688
- end: this.#index,
689
- };
690
- }
691
- }
692
-
693
- // Reject unsafe protocols (javascript:, data:, etc.) — fall back to text.
694
- // `is_valid_path_char` permits `:` so a malicious `[x](javascript:...)` would
695
- // otherwise pass character validation and reach the renderer.
696
- if (!mdz_is_safe_reference(reference)) {
697
- this.#index = start + 1;
698
- return {
699
- type: 'Text',
700
- content: '[',
701
- start,
702
- end: this.#index,
703
- };
704
- }
705
-
706
- this.#index = close_paren + 1;
707
-
708
- // Determine link type (external vs internal)
709
- const link_type = mdz_is_url(reference) ? 'external' : 'internal';
710
-
711
- return {
712
- type: 'Link',
713
- reference,
714
- children,
715
- link_type,
716
- start,
717
- end: this.#index,
718
- };
719
- }
720
-
721
- /**
722
- * Parse component/element tag: `<TagName>content</TagName>` or `<TagName />`
723
- *
724
- * Formats:
725
- * - `<Alert>content</Alert>` - Svelte component with children (uppercase first letter)
726
- * - `<div>content</div>` - HTML element with children (lowercase first letter)
727
- * - `<Alert />` - self-closing component/element
728
- *
729
- * Tag names must start with a letter and can contain letters, numbers, hyphens, underscores.
730
- *
731
- * Falls back to text if malformed or unclosed.
732
- *
733
- * TODO: Add attribute support like `<Alert status="error">` or `<div class="container">`
734
- */
735
- #parse_tag(): MdzElementNode | MdzComponentNode | MdzTextNode {
736
- const start = this.#index;
737
-
738
- // Phase 1: Validate tag structure using a local scan index.
739
- // Avoids save/restore of accumulation state on the fast path
740
- // (most `<` characters are not valid tags).
741
- let i = start + 1; // skip <
742
-
743
- // Tag name must start with a letter
744
- if (i >= this.#template.length || !is_letter(this.#template.charCodeAt(i))) {
745
- this.#index = start + 1;
746
- return this.#make_text_node('<', start);
747
- }
748
-
749
- // Collect tag name (letters, numbers, hyphens, underscores)
750
- while (i < this.#template.length && is_tag_name_char(this.#template.charCodeAt(i))) {
751
- i++;
752
- }
753
-
754
- const tag_name = this.#template.slice(start + 1, i);
755
- const first_char_code = tag_name.charCodeAt(0);
756
- const node_type: 'Component' | 'Element' =
757
- first_char_code >= A_UPPER && first_char_code <= Z_UPPER ? 'Component' : 'Element';
758
-
759
- // Skip whitespace after tag name (for future attribute support)
760
- while (i < this.#template.length && this.#template.charCodeAt(i) === SPACE) {
761
- i++;
762
- }
763
-
764
- // TODO: Parse attributes here
765
-
766
- // Check for self-closing />
767
- if (
768
- i + 1 < this.#template.length &&
769
- this.#template.charCodeAt(i) === SLASH &&
770
- this.#template.charCodeAt(i + 1) === RIGHT_ANGLE
771
- ) {
772
- this.#index = i + 2;
773
- return {
774
- type: node_type,
775
- name: tag_name,
776
- children: [],
777
- start,
778
- end: this.#index,
779
- };
780
- }
781
-
782
- // Must have closing >
783
- if (i >= this.#template.length || this.#template.charCodeAt(i) !== RIGHT_ANGLE) {
784
- this.#index = start + 1;
785
- return this.#make_text_node('<', start);
786
- }
787
-
788
- const content_start = i + 1; // past >
789
-
790
- // Bail early if closing tag is missing, past a paragraph break,
791
- // or past the search boundary (e.g. heading line end)
792
- const closing_tag = `</${tag_name}>`;
793
- const search_limit = Math.min(this.#max_search_index, this.#template.length);
794
- const close_tag_index = this.#template.indexOf(closing_tag, content_start);
795
- if (close_tag_index === -1 || close_tag_index >= search_limit) {
796
- this.#index = start + 1;
797
- return this.#make_text_node('<', start);
798
- }
799
- const para_break = this.#template.indexOf('\n\n', content_start);
800
- if (para_break !== -1 && para_break < close_tag_index) {
801
- this.#index = start + 1;
802
- return this.#make_text_node('<', start);
803
- }
804
-
805
- // Phase 2: Tag validated — save accumulation state and parse children.
806
- const saved_state = this.#save_accumulation_state();
807
- this.#accumulated_text = '';
808
- this.#nodes.length = 0;
809
- this.#index = content_start;
810
-
811
- const children: Array<MdzNode> = [];
812
-
813
- while (this.#index < this.#template.length) {
814
- if (this.#match(closing_tag)) {
815
- this.#flush_text();
816
- children.push(...this.#nodes);
817
- this.#nodes.length = 0;
818
-
819
- this.#index += closing_tag.length;
820
- this.#restore_accumulation_state(saved_state);
821
-
822
- return {
823
- type: node_type,
824
- name: tag_name,
825
- children,
826
- start,
827
- end: this.#index,
828
- };
829
- }
830
-
831
- const node = this.#parse_node();
832
- if (node.type === 'Text') {
833
- this.#accumulate_text(node.content, node.start);
834
- } else {
835
- this.#flush_text();
836
- children.push(...this.#nodes);
837
- this.#nodes.length = 0;
838
- children.push(node);
839
- }
840
- }
841
-
842
- // Defensive: pre-check guarantees closing tag exists, but handle EOF gracefully
843
- this.#restore_accumulation_state(saved_state);
844
- this.#index = start + 1;
845
- return this.#make_text_node('<', start);
846
- }
847
-
848
- /**
849
- * Read-only check if current position matches a block element.
850
- * Does not modify parser state — used to peek before flushing paragraph.
851
- * Caller must verify column-0 position before calling.
852
- */
853
- #peek_block_element(): 'heading' | 'hr' | 'codeblock' | null {
854
- if (this.#match_heading()) return 'heading';
855
- if (this.#match_hr()) return 'hr';
856
- if (this.#match_code_block()) return 'codeblock';
857
- return null;
858
- }
859
-
860
- /**
861
- * Save current text accumulation state.
862
- * Used when parsing nested structures (like components/elements) that need isolated accumulation.
863
- * Returns state object that can be passed to `#restore_accumulation_state()`.
864
- */
865
- #save_accumulation_state(): {
866
- accumulated_text: string;
867
- accumulated_start: number;
868
- nodes: Array<MdzNode>;
869
- } {
870
- return {
871
- accumulated_text: this.#accumulated_text,
872
- accumulated_start: this.#accumulated_start,
873
- nodes: this.#nodes.slice(),
874
- };
875
- }
876
-
877
- /**
878
- * Restore previously saved text accumulation state.
879
- * Used to restore parent state when exiting nested structure parsing.
880
- * @param state - state object returned from `#save_accumulation_state()`
881
- */
882
- #restore_accumulation_state(state: {
883
- accumulated_text: string;
884
- accumulated_start: number;
885
- nodes: Array<MdzNode>;
886
- }): void {
887
- this.#accumulated_text = state.accumulated_text;
888
- this.#accumulated_start = state.accumulated_start;
889
- this.#nodes.length = 0;
890
- this.#nodes.push(...state.nodes);
891
- }
892
-
893
- /**
894
- * Check if position is at a word boundary.
895
- * Word boundary = not surrounded by word characters (A-Z, a-z, 0-9).
896
- * Used to prevent intraword emphasis for underscores and tildes.
897
- *
898
- * @param index - position to check
899
- * @param check_before - whether to check the character before this position
900
- * @param check_after - whether to check the character after this position
901
- */
902
- #is_at_word_boundary(index: number, check_before: boolean, check_after: boolean): boolean {
903
- if (check_before && index > 0) {
904
- const prev = this.#template.charCodeAt(index - 1);
905
- // If preceded by word char, not at boundary
906
- if (is_word_char(prev)) {
907
- return false;
908
- }
909
- }
910
-
911
- if (check_after && index < this.#template.length) {
912
- const next = this.#template.charCodeAt(index);
913
- // If followed by word char, not at boundary
914
- if (is_word_char(next)) {
915
- return false;
916
- }
917
- }
918
-
919
- return true;
920
- }
921
-
922
- /**
923
- * Check if current position is the start of an external URL (`https://` or `http://`).
924
- *
925
- * Requires the preceding character (if any) to not be a word character. This
926
- * matches the streaming parser and the spec's "false negatives over false
927
- * positives" policy — `xhttps://...` is treated as plain text rather than
928
- * `x` followed by a link.
929
- *
930
- * Scheme matching is case-insensitive (RFC 3986 §3.1); the original casing
931
- * is preserved in the emitted reference.
932
- */
933
- #is_at_url(): boolean {
934
- // word boundary check: skip when preceded by [A-Za-z0-9]
935
- if (this.#index > 0 && is_word_char(this.#template.charCodeAt(this.#index - 1))) {
936
- return false;
937
- }
938
- const prefix_len = match_url_prefix_case_insensitive(this.#template, this.#index);
939
- if (prefix_len === 0) return false;
940
- // Must have at least one non-whitespace character after protocol
941
- if (this.#index + prefix_len >= this.#template.length) return false;
942
- const next_char = this.#template.charCodeAt(this.#index + prefix_len);
943
- return next_char !== SPACE && next_char !== NEWLINE;
944
- }
945
-
946
- /**
947
- * Parse auto-detected external URL (`https://` or `http://`).
948
- * Uses RFC 3986 whitelist validation for valid URI characters.
949
- */
950
- #parse_auto_link_url(): MdzLinkNode {
951
- const start = this.#index;
952
-
953
- // Consume protocol (case-insensitive match; original casing preserved in reference)
954
- this.#index += match_url_prefix_case_insensitive(this.#template, this.#index);
955
-
956
- // Collect URL characters using RFC 3986 whitelist
957
- // Stop at whitespace or any character invalid in URIs
958
- while (this.#index < this.#template.length) {
959
- const char_code = this.#template.charCodeAt(this.#index);
960
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
961
- break;
962
- }
963
- this.#index++;
964
- }
965
-
966
- let reference = this.#template.slice(start, this.#index);
967
-
968
- // Apply GFM trailing punctuation trimming with balanced parentheses
969
- reference = trim_trailing_punctuation(reference);
970
-
971
- // Update index after trimming
972
- this.#index = start + reference.length;
973
-
974
- return {
975
- type: 'Link',
976
- reference,
977
- children: [{type: 'Text', content: reference, start, end: this.#index}],
978
- link_type: 'external',
979
- start,
980
- end: this.#index,
981
- };
982
- }
983
-
984
- /**
985
- * Parse auto-detected path (absolute `/`, relative `./` or `../`).
986
- * Uses RFC 3986 whitelist validation for valid URI characters.
987
- */
988
- #parse_auto_link_path(): MdzLinkNode {
989
- const start = this.#index;
990
-
991
- // Collect path characters using RFC 3986 whitelist
992
- // Stop at whitespace or any character invalid in URIs
993
- while (this.#index < this.#template.length) {
994
- const char_code = this.#template.charCodeAt(this.#index);
995
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
996
- break;
997
- }
998
- this.#index++;
999
- }
1000
-
1001
- let reference = this.#template.slice(start, this.#index);
1002
-
1003
- // Apply GFM trailing punctuation trimming
1004
- reference = trim_trailing_punctuation(reference);
1005
-
1006
- // Update index after trimming
1007
- this.#index = start + reference.length;
1008
-
1009
- return {
1010
- type: 'Link',
1011
- reference,
1012
- children: [{type: 'Text', content: reference, start, end: this.#index}],
1013
- link_type: 'internal',
1014
- start,
1015
- end: this.#index,
1016
- };
1017
- }
1018
-
1019
- /**
1020
- * Parse plain text until special character encountered.
1021
- * Preserves all whitespace (except paragraph breaks handled separately).
1022
- * Detects and delegates to URL/path parsing when encountered.
1023
- */
1024
- #parse_text(): MdzTextNode | MdzLinkNode {
1025
- const start = this.#index;
1026
-
1027
- // Check for URL or internal absolute/relative path at current position
1028
- if (this.#is_at_url()) {
1029
- return this.#parse_auto_link_url();
1030
- }
1031
- if (
1032
- is_at_absolute_path(this.#template, this.#index) ||
1033
- is_at_relative_path(this.#template, this.#index)
1034
- ) {
1035
- return this.#parse_auto_link_path();
1036
- }
1037
-
1038
- while (this.#index < this.#template.length) {
1039
- const char_code = this.#template.charCodeAt(this.#index);
1040
-
1041
- // Stop at special characters (but preserve single newlines)
1042
- if (
1043
- char_code === BACKTICK ||
1044
- char_code === ASTERISK ||
1045
- char_code === UNDERSCORE ||
1046
- char_code === TILDE ||
1047
- char_code === LEFT_BRACKET ||
1048
- char_code === RIGHT_BRACKET ||
1049
- char_code === RIGHT_PAREN ||
1050
- char_code === LEFT_ANGLE
1051
- ) {
1052
- break;
1053
- }
1054
-
1055
- // Check for paragraph break (double newline)
1056
- if (this.#is_at_paragraph_break()) {
1057
- break;
1058
- }
1059
-
1060
- // When next line could start a block element, consume the newline and stop.
1061
- // The main loop will try block detection at the next character.
1062
- // Consuming the newline here avoids a 3-iteration detour (break before \n,
1063
- // fail peek at \n, safety-increment, then succeed peek at block char).
1064
- if (char_code === NEWLINE) {
1065
- const next_i = this.#index + 1;
1066
- if (next_i < this.#template.length) {
1067
- const next_char = this.#template.charCodeAt(next_i);
1068
- if (next_char === HASH || next_char === HYPHEN || next_char === BACKTICK) {
1069
- this.#index++; // consume the newline
1070
- break;
1071
- }
1072
- }
1073
- }
1074
-
1075
- // Check for URL or internal absolute/relative path mid-text (char code guard avoids startsWith on every char)
1076
- if (
1077
- ((char_code === 104 /* h */ || char_code === 72) /* H */ && this.#is_at_url()) ||
1078
- (char_code === SLASH && is_at_absolute_path(this.#template, this.#index)) ||
1079
- (char_code === PERIOD && is_at_relative_path(this.#template, this.#index))
1080
- ) {
1081
- break;
1082
- }
1083
-
1084
- this.#index++;
1085
- }
1086
-
1087
- // Ensure we always consume at least one character to prevent infinite loops
1088
- if (this.#index === start && this.#index < this.#template.length) {
1089
- this.#index++;
1090
- }
1091
-
1092
- // Use slice instead of concatenation for performance
1093
- const content = this.#template.slice(start, this.#index);
1094
-
1095
- return {
1096
- type: 'Text',
1097
- content,
1098
- start,
1099
- end: this.#index,
1100
- };
1101
- }
1102
-
1103
- /**
1104
- * Parse nodes until delimiter string is found.
1105
- * Used for parsing children of inline formatting (bold, italic, strikethrough) and markdown links.
1106
- *
1107
- * Implements greedy/bounded parsing to prevent nested formatters from consuming parent delimiters:
1108
- * - When parsing `**bold with _italic_**`, the outer `**` parser finds its closing delimiter at position Y
1109
- * - Sets `#max_search_index = Y` to create a boundary
1110
- * - Parses children only within range, preventing `_italic_` from finding delimiters beyond Y
1111
- * - This ensures proper nesting without backtracking
1112
- *
1113
- * Stops parsing when:
1114
- * - Delimiter string is found
1115
- * - Paragraph break (double newline) is encountered (allows block elements to interrupt inline formatting)
1116
- * - `end_index` boundary is reached
1117
- *
1118
- * @param delimiter - the delimiter string to stop at (e.g., '**', '_', ']')
1119
- * @param end_index - optional maximum index to parse up to (for greedy/bounded parsing)
1120
- * @returns array of parsed nodes (may be empty if delimiter found immediately)
1121
- */
1122
- #parse_nodes_until(delimiter: string, end_index?: number): Array<MdzNode> {
1123
- const nodes: Array<MdzNode> = [];
1124
- const max_index = end_index ?? this.#template.length;
1125
-
1126
- // Save and set max search boundary for nested parsers
1127
- const saved_max_search_index = this.#max_search_index;
1128
- this.#max_search_index = max_index;
1129
-
1130
- while (this.#index < max_index) {
1131
- if (this.#match(delimiter)) {
1132
- break;
1133
- }
1134
-
1135
- // Check for paragraph break (block element interruption)
1136
- if (this.#is_at_paragraph_break()) {
1137
- // Paragraph break interrupts inline formatting
1138
- break;
1139
- }
1140
-
1141
- // merge adjacent Text nodes (e.g., text + failed delimiter + text)
1142
- mdz_push_merging_text(nodes, this.#parse_node());
1143
- }
1144
-
1145
- // Restore previous boundary
1146
- this.#max_search_index = saved_max_search_index;
1147
-
1148
- return nodes;
1149
- }
1150
-
1151
- /**
1152
- * Check if current position is at a paragraph break (double newline).
1153
- */
1154
- #is_at_paragraph_break(): boolean {
1155
- return (
1156
- this.#index + 1 < this.#template.length &&
1157
- this.#template.charCodeAt(this.#index) === NEWLINE &&
1158
- this.#template.charCodeAt(this.#index + 1) === NEWLINE
1159
- );
1160
- }
1161
-
1162
- #match(str: string): boolean {
1163
- return this.#template.startsWith(str, this.#index);
1164
- }
1165
-
1166
- /**
1167
- * Consume string at current index, or throw error.
1168
- */
1169
- #eat(str: string): void {
1170
- if (this.#match(str)) {
1171
- this.#index += str.length;
1172
- } else {
1173
- throw Error(`Expected "${str}" at index ${this.#index}`);
1174
- }
1175
- }
1176
-
1177
- /**
1178
- * Check if current position matches a horizontal rule.
1179
- * HR must be exactly `---` at column 0, followed by newline or EOF.
1180
- *
1181
- * mdz has no setext headings, so `---` after a paragraph is unambiguous
1182
- * (always an HR, unlike CommonMark where it becomes a setext heading).
1183
- */
1184
- #match_hr(): boolean {
1185
- let i = this.#index;
1186
-
1187
- // Must have exactly three hyphens
1188
- if (
1189
- i + HR_HYPHEN_COUNT > this.#template.length ||
1190
- this.#template.charCodeAt(i) !== HYPHEN ||
1191
- this.#template.charCodeAt(i + 1) !== HYPHEN ||
1192
- this.#template.charCodeAt(i + 2) !== HYPHEN
1193
- ) {
1194
- return false;
1195
- }
1196
- i += HR_HYPHEN_COUNT;
1197
-
1198
- // After the three hyphens, only whitespace and newline (or EOF) allowed
1199
- while (i < this.#template.length) {
1200
- const char_code = this.#template.charCodeAt(i);
1201
- if (char_code === NEWLINE) {
1202
- return true;
1203
- }
1204
- if (char_code !== SPACE) {
1205
- return false; // Non-whitespace after ---, not an hr
1206
- }
1207
- i++;
1208
- }
1209
-
1210
- // Reached EOF after ---, valid hr
1211
- return true;
1212
- }
1213
-
1214
- /**
1215
- * Parse horizontal rule: `---`
1216
- * Assumes #match_hr() already verified this is an hr.
1217
- */
1218
- #parse_hr(): MdzHrNode {
1219
- const start = this.#index;
1220
-
1221
- // Consume the three hyphens (no leading whitespace - already verified)
1222
- this.#index += HR_HYPHEN_COUNT;
1223
-
1224
- // Skip trailing whitespace
1225
- while (
1226
- this.#index < this.#template.length &&
1227
- this.#template.charCodeAt(this.#index) === SPACE
1228
- ) {
1229
- this.#index++;
1230
- }
1231
-
1232
- // Don't consume the newline - let the main parse loop handle it
1233
-
1234
- return {
1235
- type: 'Hr',
1236
- start,
1237
- end: this.#index,
1238
- };
1239
- }
1240
-
1241
- /**
1242
- * Check if current position matches a heading.
1243
- * Heading must be 1-6 hashes at column 0, followed by space and content,
1244
- * followed by newline or EOF.
1245
- */
1246
- #match_heading(): boolean {
1247
- let i = this.#index;
1248
-
1249
- // Count hashes (must be 1-6)
1250
- let hash_count = 0;
1251
- while (
1252
- i < this.#template.length &&
1253
- this.#template.charCodeAt(i) === HASH &&
1254
- hash_count <= MAX_HEADING_LEVEL
1255
- ) {
1256
- hash_count++;
1257
- i++;
1258
- }
1259
-
1260
- if (hash_count === 0 || hash_count > MAX_HEADING_LEVEL) {
1261
- return false;
1262
- }
1263
-
1264
- // Must have space after hashes
1265
- if (i >= this.#template.length || this.#template.charCodeAt(i) !== SPACE) {
1266
- return false;
1267
- }
1268
- i++; // consume the space
1269
-
1270
- // Must have at least one non-whitespace character after the space
1271
- while (i < this.#template.length) {
1272
- const char_code = this.#template.charCodeAt(i);
1273
- if (char_code === NEWLINE) return false; // reached end of line with only whitespace
1274
- if (char_code !== SPACE && char_code !== TAB) return true;
1275
- i++;
1276
- }
1277
-
1278
- // Reached EOF with only whitespace after hashes
1279
- return false;
1280
- }
1281
-
1282
- /**
1283
- * Parse heading: `# Heading text`
1284
- * Assumes #match_heading() already verified this is a heading.
1285
- */
1286
- #parse_heading(): MdzHeadingNode {
1287
- const start = this.#index;
1288
-
1289
- // Count and consume hashes
1290
- let level = 0;
1291
- while (this.#index < this.#template.length && this.#template.charCodeAt(this.#index) === HASH) {
1292
- level++;
1293
- this.#index++;
1294
- }
1295
-
1296
- // Consume the space after hashes (already verified to exist)
1297
- this.#index++;
1298
-
1299
- // Find end-of-line to bound nested parsers (prevents tag scanner from scanning past heading)
1300
- let eol = this.#template.indexOf('\n', this.#index);
1301
- if (eol === -1) eol = this.#template.length;
1302
-
1303
- const saved_max_search_index = this.#max_search_index;
1304
- this.#max_search_index = eol;
1305
-
1306
- // Parse inline content until end of line
1307
- const content_nodes: Array<MdzNode> = [];
1308
-
1309
- while (this.#index < eol) {
1310
- const node = this.#parse_node();
1311
- if (node.type === 'Text') {
1312
- // Trim if #parse_text overshot past the newline
1313
- if (node.end > eol) {
1314
- const trimmed_content = node.content.slice(0, eol - node.start);
1315
- if (trimmed_content) {
1316
- this.#accumulate_text(trimmed_content, node.start);
1317
- }
1318
- this.#index = eol;
1319
- break;
1320
- }
1321
- this.#accumulate_text(node.content, node.start);
1322
- } else {
1323
- this.#flush_text();
1324
- content_nodes.push(...this.#nodes);
1325
- this.#nodes.length = 0;
1326
- content_nodes.push(node);
1327
- }
1328
- }
1329
-
1330
- this.#max_search_index = saved_max_search_index;
1331
-
1332
- this.#flush_text();
1333
- content_nodes.push(...this.#nodes);
1334
- this.#nodes.length = 0;
1335
-
1336
- // Don't consume the newline - let the main parse loop handle it
1337
-
1338
- return {
1339
- type: 'Heading',
1340
- level: level as 1 | 2 | 3 | 4 | 5 | 6,
1341
- id: mdz_heading_id(content_nodes),
1342
- children: content_nodes,
1343
- start,
1344
- end: this.#index,
1345
- };
1346
- }
1347
-
1348
- /**
1349
- * Check if current position matches a code block.
1350
- * Code block must be 3+ backticks at column 0, closing fence followed by newline or EOF.
1351
- * Empty code blocks (no content) are treated as invalid.
1352
- */
1353
- #match_code_block(): boolean {
1354
- let i = this.#index;
1355
-
1356
- // Must have at least three backticks
1357
- let backtick_count = 0;
1358
- while (i < this.#template.length && this.#template.charCodeAt(i) === BACKTICK) {
1359
- backtick_count++;
1360
- i++;
1361
- }
1362
-
1363
- if (backtick_count < MIN_CODEBLOCK_BACKTICKS) {
1364
- return false;
1365
- }
1366
-
1367
- // Skip optional language hint (consume until space or newline)
1368
- while (i < this.#template.length) {
1369
- const char_code = this.#template.charCodeAt(i);
1370
- if (char_code === SPACE || char_code === NEWLINE) {
1371
- break;
1372
- }
1373
- i++;
1374
- }
1375
-
1376
- // Skip any trailing spaces on opening fence line
1377
- while (i < this.#template.length && this.#template.charCodeAt(i) === SPACE) {
1378
- i++;
1379
- }
1380
-
1381
- // Must have newline after opening fence (or be at EOF)
1382
- if (i >= this.#template.length) {
1383
- return false; // No newline, can't be a valid code block
1384
- }
1385
-
1386
- if (this.#template.charCodeAt(i) !== NEWLINE) {
1387
- return false;
1388
- }
1389
- i++; // consume the newline
1390
-
1391
- // Mark content start position (after opening fence newline)
1392
- const content_start = i;
1393
-
1394
- // Now search for closing fence
1395
- const closing_fence = '`'.repeat(backtick_count);
1396
- while (i < this.#template.length) {
1397
- // Check if we're at a potential closing fence (must be at start of line)
1398
- if (this.#template.startsWith(closing_fence, i)) {
1399
- // Verify it's at column 0 by checking previous character
1400
- const prev_char = i > 0 ? this.#template.charCodeAt(i - 1) : NEWLINE;
1401
- if (prev_char === NEWLINE || i === 0) {
1402
- // Found closing fence - check for empty content first
1403
- const content = this.#template.slice(content_start, i);
1404
- const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
1405
- if (final_content.length === 0) {
1406
- return false; // Empty code block has no semantic meaning
1407
- }
1408
-
1409
- // Now verify what comes after closing fence
1410
- let j = i + backtick_count;
1411
-
1412
- // Skip trailing whitespace on closing fence line
1413
- while (j < this.#template.length && this.#template.charCodeAt(j) === SPACE) {
1414
- j++;
1415
- }
1416
-
1417
- // Must have newline after closing fence (or be at EOF)
1418
- if (j >= this.#template.length) {
1419
- return true;
1420
- }
1421
-
1422
- if (this.#template.charCodeAt(j) !== NEWLINE) {
1423
- // closing fence has non-whitespace after it on same line - not a code block
1424
- return false;
1425
- }
1426
-
1427
- return true; // code block followed by newline or EOF
1428
- }
1429
- }
1430
- i++;
1431
- }
1432
-
1433
- // No closing fence found - not a valid code block
1434
- return false;
1435
- }
1436
-
1437
- /**
1438
- * Parse code block: ```lang\ncode\n```
1439
- * Assumes #match_code_block() already verified this is a code block.
1440
- */
1441
- #parse_code_block(): MdzCodeblockNode {
1442
- const start = this.#index;
1443
-
1444
- // Count and consume opening backticks
1445
- let backtick_count = 0;
1446
- while (
1447
- this.#index < this.#template.length &&
1448
- this.#template.charCodeAt(this.#index) === BACKTICK
1449
- ) {
1450
- backtick_count++;
1451
- this.#index++;
1452
- }
1453
-
1454
- // Parse optional language hint (consume until space or newline)
1455
- let lang: string | null = null;
1456
- const lang_start = this.#index;
1457
- while (this.#index < this.#template.length) {
1458
- const char_code = this.#template.charCodeAt(this.#index);
1459
- if (char_code === SPACE || char_code === NEWLINE) {
1460
- break;
1461
- }
1462
- this.#index++;
1463
- }
1464
- if (this.#index > lang_start) {
1465
- lang = this.#template.slice(lang_start, this.#index);
1466
- }
1467
-
1468
- // Skip any trailing spaces on opening fence line
1469
- while (
1470
- this.#index < this.#template.length &&
1471
- this.#template.charCodeAt(this.#index) === SPACE
1472
- ) {
1473
- this.#index++;
1474
- }
1475
-
1476
- // Consume the newline after opening fence (first newline is consumed per spec)
1477
- if (this.#index < this.#template.length && this.#template.charCodeAt(this.#index) === NEWLINE) {
1478
- this.#index++;
1479
- }
1480
-
1481
- // Collect content until closing fence
1482
- const content_start = this.#index;
1483
- const closing_fence = '`'.repeat(backtick_count);
1484
-
1485
- while (this.#index < this.#template.length) {
1486
- // Check if we're at the closing fence (must be at start of line)
1487
- if (this.#template.startsWith(closing_fence, this.#index)) {
1488
- // Verify it's at column 0 by checking previous character
1489
- const prev_char = this.#index > 0 ? this.#template.charCodeAt(this.#index - 1) : NEWLINE;
1490
- if (prev_char === NEWLINE || this.#index === 0) {
1491
- // Check if it's exactly the right number of backticks at line start
1492
- let j = this.#index + backtick_count;
1493
- // After closing fence, only whitespace and newline allowed
1494
- while (j < this.#template.length && this.#template.charCodeAt(j) === SPACE) {
1495
- j++;
1496
- }
1497
- if (j >= this.#template.length || this.#template.charCodeAt(j) === NEWLINE) {
1498
- // Valid closing fence
1499
- const content = this.#template.slice(content_start, this.#index);
1500
- // Remove trailing newline if present (closing fence comes after a newline)
1501
- const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
1502
-
1503
- // Consume closing fence
1504
- this.#index += backtick_count;
1505
-
1506
- // Skip trailing whitespace on closing fence line
1507
- while (
1508
- this.#index < this.#template.length &&
1509
- this.#template.charCodeAt(this.#index) === SPACE
1510
- ) {
1511
- this.#index++;
1512
- }
1513
-
1514
- // Don't consume the newline - let the main parse loop handle it
1515
-
1516
- return {
1517
- type: 'Codeblock',
1518
- lang,
1519
- content: final_content,
1520
- start,
1521
- end: this.#index,
1522
- };
1523
- }
1524
- }
1525
- }
1526
- this.#index++;
1527
- }
1528
-
1529
- // Should not reach here if #match_code_block() validated correctly
1530
- throw Error('Code block not properly closed');
1531
- }
1532
- }