@fuzdev/fuz_ui 0.204.0 → 0.205.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ApiModule.svelte +14 -1
- package/dist/ApiModule.svelte.d.ts.map +1 -1
- package/dist/DeclarationDetail.svelte +15 -4
- package/dist/DeclarationDetail.svelte.d.ts.map +1 -1
- package/package.json +12 -36
- package/dist/Mdz.svelte +0 -34
- package/dist/Mdz.svelte.d.ts +0 -11
- package/dist/Mdz.svelte.d.ts.map +0 -1
- package/dist/MdzNodeView.svelte +0 -107
- package/dist/MdzNodeView.svelte.d.ts +0 -9
- package/dist/MdzNodeView.svelte.d.ts.map +0 -1
- package/dist/MdzPrecompiled.svelte +0 -30
- package/dist/MdzPrecompiled.svelte.d.ts +0 -11
- package/dist/MdzPrecompiled.svelte.d.ts.map +0 -1
- package/dist/MdzRoot.svelte +0 -30
- package/dist/MdzRoot.svelte.d.ts +0 -12
- package/dist/MdzRoot.svelte.d.ts.map +0 -1
- package/dist/MdzStream.svelte +0 -32
- package/dist/MdzStream.svelte.d.ts +0 -12
- package/dist/MdzStream.svelte.d.ts.map +0 -1
- package/dist/MdzStreamNodeView.svelte +0 -106
- package/dist/MdzStreamNodeView.svelte.d.ts +0 -9
- package/dist/MdzStreamNodeView.svelte.d.ts.map +0 -1
- package/dist/mdz.d.ts +0 -112
- package/dist/mdz.d.ts.map +0 -1
- package/dist/mdz.js +0 -1186
- package/dist/mdz_components.d.ts +0 -63
- package/dist/mdz_components.d.ts.map +0 -1
- package/dist/mdz_components.js +0 -32
- package/dist/mdz_helpers.d.ts +0 -164
- package/dist/mdz_helpers.d.ts.map +0 -1
- package/dist/mdz_helpers.js +0 -424
- package/dist/mdz_lexer.d.ts +0 -93
- package/dist/mdz_lexer.d.ts.map +0 -1
- package/dist/mdz_lexer.js +0 -732
- package/dist/mdz_opcodes.d.ts +0 -174
- package/dist/mdz_opcodes.d.ts.map +0 -1
- package/dist/mdz_opcodes.js +0 -14
- package/dist/mdz_opcodes_to_nodes.d.ts +0 -20
- package/dist/mdz_opcodes_to_nodes.d.ts.map +0 -1
- package/dist/mdz_opcodes_to_nodes.js +0 -332
- package/dist/mdz_stream_parser.d.ts +0 -80
- package/dist/mdz_stream_parser.d.ts.map +0 -1
- package/dist/mdz_stream_parser.js +0 -354
- package/dist/mdz_stream_parser_block.d.ts +0 -76
- package/dist/mdz_stream_parser_block.d.ts.map +0 -1
- package/dist/mdz_stream_parser_block.js +0 -458
- package/dist/mdz_stream_parser_inline.d.ts +0 -52
- package/dist/mdz_stream_parser_inline.d.ts.map +0 -1
- package/dist/mdz_stream_parser_inline.js +0 -378
- package/dist/mdz_stream_parser_link.d.ts +0 -22
- package/dist/mdz_stream_parser_link.d.ts.map +0 -1
- package/dist/mdz_stream_parser_link.js +0 -215
- package/dist/mdz_stream_parser_state.d.ts +0 -187
- package/dist/mdz_stream_parser_state.d.ts.map +0 -1
- package/dist/mdz_stream_parser_state.js +0 -398
- package/dist/mdz_stream_parser_text.d.ts +0 -14
- package/dist/mdz_stream_parser_text.d.ts.map +0 -1
- package/dist/mdz_stream_parser_text.js +0 -50
- package/dist/mdz_stream_parser_url.d.ts +0 -60
- package/dist/mdz_stream_parser_url.d.ts.map +0 -1
- package/dist/mdz_stream_parser_url.js +0 -307
- package/dist/mdz_stream_state.svelte.d.ts +0 -44
- package/dist/mdz_stream_state.svelte.d.ts.map +0 -1
- package/dist/mdz_stream_state.svelte.js +0 -357
- package/dist/mdz_to_svelte.d.ts +0 -41
- package/dist/mdz_to_svelte.d.ts.map +0 -1
- package/dist/mdz_to_svelte.js +0 -100
- package/dist/mdz_token_parser.d.ts +0 -14
- package/dist/mdz_token_parser.d.ts.map +0 -1
- package/dist/mdz_token_parser.js +0 -344
- package/dist/svelte_preprocess_mdz.d.ts +0 -65
- package/dist/svelte_preprocess_mdz.d.ts.map +0 -1
- package/dist/svelte_preprocess_mdz.js +0 -529
- package/dist/tsdoc_mdz.d.ts +0 -45
- package/dist/tsdoc_mdz.d.ts.map +0 -1
- package/dist/tsdoc_mdz.js +0 -88
- package/src/lib/mdz.ts +0 -1532
- package/src/lib/mdz_components.ts +0 -64
- package/src/lib/mdz_helpers.ts +0 -449
- package/src/lib/mdz_lexer.ts +0 -1012
- package/src/lib/mdz_opcodes.ts +0 -205
- package/src/lib/mdz_opcodes_to_nodes.ts +0 -375
- package/src/lib/mdz_stream_parser.ts +0 -415
- package/src/lib/mdz_stream_parser_block.ts +0 -532
- package/src/lib/mdz_stream_parser_inline.ts +0 -414
- package/src/lib/mdz_stream_parser_link.ts +0 -271
- package/src/lib/mdz_stream_parser_state.ts +0 -539
- package/src/lib/mdz_stream_parser_text.ts +0 -77
- package/src/lib/mdz_stream_parser_url.ts +0 -365
- package/src/lib/mdz_stream_state.svelte.ts +0 -387
- package/src/lib/mdz_to_svelte.ts +0 -141
- package/src/lib/mdz_token_parser.ts +0 -434
- package/src/lib/svelte_preprocess_mdz.ts +0 -742
- package/src/lib/tsdoc_mdz.ts +0 -97
package/dist/mdz.js
DELETED
|
@@ -1,1186 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* mdz - minimal markdown dialect for Fuz documentation.
|
|
3
|
-
*
|
|
4
|
-
* Parses an enhanced markdown dialect with:
|
|
5
|
-
* - inline formatting: `code`, **bold**, _italic_, ~strikethrough~
|
|
6
|
-
* - auto-detected links: external URLs (`https://...`) and internal paths (`/path`)
|
|
7
|
-
* - markdown links: `[text](url)` with custom display text
|
|
8
|
-
* - inline code in backticks (creates `Code` nodes; auto-linking to identifiers/modules
|
|
9
|
-
* is handled by the rendering layer via `MdzNodeView.svelte`)
|
|
10
|
-
* - paragraph breaks (double newline)
|
|
11
|
-
* - block elements: headings, horizontal rules, code blocks
|
|
12
|
-
* - HTML elements and Svelte components (opt-in via context)
|
|
13
|
-
*
|
|
14
|
-
* Key constraint: preserves ALL whitespace exactly as authored,
|
|
15
|
-
* and is rendered with white-space pre or pre-wrap.
|
|
16
|
-
*
|
|
17
|
-
* ## Design philosophy
|
|
18
|
-
*
|
|
19
|
-
* - **False negatives over false positives**: When in doubt, treat as plain text.
|
|
20
|
-
* Block elements can interrupt paragraphs without blank lines; inline formatting is strict.
|
|
21
|
-
* - **One way to do things**: Single unambiguous syntax per feature. No alternatives.
|
|
22
|
-
* - **Explicit over implicit**: Clear delimiters and column-0 requirements avoid ambiguity.
|
|
23
|
-
* - **Simple over complete**: Prefer simple parsing rules over complex edge case handling.
|
|
24
|
-
*
|
|
25
|
-
* ## Status
|
|
26
|
-
*
|
|
27
|
-
* This is an early proof of concept with missing features and edge cases.
|
|
28
|
-
*
|
|
29
|
-
* @module
|
|
30
|
-
*/
|
|
31
|
-
import { BACKTICK, ASTERISK, UNDERSCORE, TILDE, NEWLINE, HYPHEN, HASH, SPACE, TAB, LEFT_ANGLE, RIGHT_ANGLE, SLASH, LEFT_BRACKET, RIGHT_BRACKET, LEFT_PAREN, RIGHT_PAREN, A_UPPER, Z_UPPER, HR_HYPHEN_COUNT, MIN_CODEBLOCK_BACKTICKS, MAX_HEADING_LEVEL, match_url_prefix_case_insensitive, is_letter, is_tag_name_char, is_word_char, PERIOD, is_valid_path_char, trim_trailing_punctuation, is_at_absolute_path, is_at_relative_path, extract_single_tag, mdz_heading_id, mdz_is_url, mdz_is_safe_reference, mdz_push_merging_text, } from './mdz_helpers.js';
|
|
32
|
-
// TODO design incremental parsing or some system that preserves Svelte components across re-renders when possible
|
|
33
|
-
/**
|
|
34
|
-
* Parses text to an array of `MdzNode`.
|
|
35
|
-
*/
|
|
36
|
-
export const mdz_parse = (text) => new MdzParser(text).parse();
|
|
37
|
-
/**
|
|
38
|
-
* Parser for mdz format.
|
|
39
|
-
* Single-pass lexer/parser with text accumulation for efficiency.
|
|
40
|
-
* Used by `mdz_parse`, which should be preferred for simple usage.
|
|
41
|
-
*/
|
|
42
|
-
export class MdzParser {
|
|
43
|
-
#index = 0;
|
|
44
|
-
#template;
|
|
45
|
-
#accumulated_text = '';
|
|
46
|
-
#accumulated_start = 0;
|
|
47
|
-
#nodes = [];
|
|
48
|
-
#max_search_index = Number.MAX_SAFE_INTEGER; // Boundary for delimiter searches
|
|
49
|
-
constructor(template) {
|
|
50
|
-
this.#template = template;
|
|
51
|
-
}
|
|
52
|
-
/**
|
|
53
|
-
* Main parse method. Returns flat array of nodes,
|
|
54
|
-
* with paragraph nodes wrapping content between double newlines.
|
|
55
|
-
*
|
|
56
|
-
* Block elements (headings, HR, codeblocks) are detected at every column-0
|
|
57
|
-
* position — they can interrupt paragraphs without requiring blank lines.
|
|
58
|
-
*/
|
|
59
|
-
parse() {
|
|
60
|
-
this.#nodes.length = 0;
|
|
61
|
-
const root_nodes = [];
|
|
62
|
-
const paragraph_children = [];
|
|
63
|
-
// Skip leading newlines
|
|
64
|
-
this.#skip_newlines();
|
|
65
|
-
while (this.#index < this.#template.length) {
|
|
66
|
-
// Block elements only start at column 0 — skip peek for mid-line characters
|
|
67
|
-
const block_type = (this.#index === 0 || this.#template.charCodeAt(this.#index - 1) === NEWLINE) &&
|
|
68
|
-
this.#peek_block_element();
|
|
69
|
-
if (block_type) {
|
|
70
|
-
const flushed = this.#flush_paragraph(paragraph_children, true);
|
|
71
|
-
if (flushed)
|
|
72
|
-
root_nodes.push(flushed);
|
|
73
|
-
if (block_type === 'heading')
|
|
74
|
-
root_nodes.push(this.#parse_heading());
|
|
75
|
-
else if (block_type === 'hr')
|
|
76
|
-
root_nodes.push(this.#parse_hr());
|
|
77
|
-
else
|
|
78
|
-
root_nodes.push(this.#parse_code_block());
|
|
79
|
-
this.#skip_newlines();
|
|
80
|
-
continue;
|
|
81
|
-
}
|
|
82
|
-
// Check for paragraph break (double newline)
|
|
83
|
-
if (this.#is_at_paragraph_break()) {
|
|
84
|
-
const flushed = this.#flush_paragraph(paragraph_children, true);
|
|
85
|
-
if (flushed)
|
|
86
|
-
root_nodes.push(flushed);
|
|
87
|
-
this.#skip_newlines();
|
|
88
|
-
continue;
|
|
89
|
-
}
|
|
90
|
-
// Parse inline content
|
|
91
|
-
const node = this.#parse_node();
|
|
92
|
-
if (node.type === 'Text') {
|
|
93
|
-
this.#accumulate_text(node.content, node.start);
|
|
94
|
-
}
|
|
95
|
-
else {
|
|
96
|
-
this.#flush_text();
|
|
97
|
-
this.#nodes.push(node);
|
|
98
|
-
}
|
|
99
|
-
if (this.#nodes.length > 0) {
|
|
100
|
-
paragraph_children.push(...this.#nodes);
|
|
101
|
-
this.#nodes.length = 0;
|
|
102
|
-
}
|
|
103
|
-
}
|
|
104
|
-
// Flush remaining content as final paragraph
|
|
105
|
-
const final_paragraph = this.#flush_paragraph(paragraph_children, true);
|
|
106
|
-
if (final_paragraph)
|
|
107
|
-
root_nodes.push(final_paragraph);
|
|
108
|
-
return root_nodes;
|
|
109
|
-
}
|
|
110
|
-
/**
|
|
111
|
-
* Flush accumulated inline content as a paragraph node (or single tag).
|
|
112
|
-
* When `trim_trailing` is true, trims trailing newlines from the last text node
|
|
113
|
-
* (the line break before a block element or paragraph break).
|
|
114
|
-
*
|
|
115
|
-
* @mutates paragraph_children - trims last text node, removes whitespace-only nodes, clears array
|
|
116
|
-
*/
|
|
117
|
-
#flush_paragraph(paragraph_children, trim_trailing = false) {
|
|
118
|
-
this.#flush_text();
|
|
119
|
-
if (this.#nodes.length > 0) {
|
|
120
|
-
paragraph_children.push(...this.#nodes);
|
|
121
|
-
this.#nodes.length = 0;
|
|
122
|
-
}
|
|
123
|
-
if (paragraph_children.length === 0) {
|
|
124
|
-
return null;
|
|
125
|
-
}
|
|
126
|
-
if (trim_trailing) {
|
|
127
|
-
// Trim trailing newlines from the last text node (line break before block element)
|
|
128
|
-
const last = paragraph_children[paragraph_children.length - 1];
|
|
129
|
-
if (last.type === 'Text') {
|
|
130
|
-
const trimmed = last.content.replace(/\n+$/, '');
|
|
131
|
-
if (trimmed) {
|
|
132
|
-
last.content = trimmed;
|
|
133
|
-
last.end = last.start + trimmed.length;
|
|
134
|
-
}
|
|
135
|
-
else {
|
|
136
|
-
paragraph_children.pop();
|
|
137
|
-
}
|
|
138
|
-
}
|
|
139
|
-
// Skip whitespace-only paragraphs (e.g. newlines between consecutive blocks)
|
|
140
|
-
const has_content = paragraph_children.some((n) => n.type !== 'Text' || n.content.trim().length > 0);
|
|
141
|
-
if (!has_content) {
|
|
142
|
-
paragraph_children.length = 0;
|
|
143
|
-
return null;
|
|
144
|
-
}
|
|
145
|
-
}
|
|
146
|
-
// Single tag (component/element) - add directly without paragraph wrapper (MDX convention)
|
|
147
|
-
const single_tag = extract_single_tag(paragraph_children);
|
|
148
|
-
if (single_tag) {
|
|
149
|
-
paragraph_children.length = 0;
|
|
150
|
-
return single_tag;
|
|
151
|
-
}
|
|
152
|
-
// Regular paragraph
|
|
153
|
-
const result = {
|
|
154
|
-
type: 'Paragraph',
|
|
155
|
-
children: paragraph_children.slice(),
|
|
156
|
-
start: paragraph_children[0].start,
|
|
157
|
-
end: paragraph_children[paragraph_children.length - 1].end,
|
|
158
|
-
};
|
|
159
|
-
paragraph_children.length = 0;
|
|
160
|
-
return result;
|
|
161
|
-
}
|
|
162
|
-
/**
|
|
163
|
-
* Consume consecutive newline characters.
|
|
164
|
-
*/
|
|
165
|
-
#skip_newlines() {
|
|
166
|
-
while (this.#index < this.#template.length &&
|
|
167
|
-
this.#template.charCodeAt(this.#index) === NEWLINE) {
|
|
168
|
-
this.#index++;
|
|
169
|
-
}
|
|
170
|
-
}
|
|
171
|
-
/**
|
|
172
|
-
* Accumulate text for later flushing (performance optimization).
|
|
173
|
-
*/
|
|
174
|
-
#accumulate_text(text, start) {
|
|
175
|
-
if (this.#accumulated_text === '') {
|
|
176
|
-
this.#accumulated_start = start;
|
|
177
|
-
}
|
|
178
|
-
this.#accumulated_text += text;
|
|
179
|
-
}
|
|
180
|
-
#flush_text() {
|
|
181
|
-
if (this.#accumulated_text !== '') {
|
|
182
|
-
this.#nodes.push({
|
|
183
|
-
type: 'Text',
|
|
184
|
-
content: this.#accumulated_text,
|
|
185
|
-
start: this.#accumulated_start,
|
|
186
|
-
end: this.#accumulated_start + this.#accumulated_text.length,
|
|
187
|
-
});
|
|
188
|
-
this.#accumulated_text = '';
|
|
189
|
-
}
|
|
190
|
-
}
|
|
191
|
-
/**
|
|
192
|
-
* Create a text node and advance index past the content.
|
|
193
|
-
* Used when formatting delimiters fail to match and need to be treated as literal text.
|
|
194
|
-
*/
|
|
195
|
-
#make_text_node(content, start) {
|
|
196
|
-
this.#index = start + content.length;
|
|
197
|
-
return {
|
|
198
|
-
type: 'Text',
|
|
199
|
-
content,
|
|
200
|
-
start,
|
|
201
|
-
end: this.#index,
|
|
202
|
-
};
|
|
203
|
-
}
|
|
204
|
-
/**
|
|
205
|
-
* Parse next node based on current character.
|
|
206
|
-
* Uses switch for performance (avoids regex in hot loop).
|
|
207
|
-
*/
|
|
208
|
-
#parse_node() {
|
|
209
|
-
const char_code = this.#template.charCodeAt(this.#index);
|
|
210
|
-
// Use character codes for performance in hot path
|
|
211
|
-
switch (char_code) {
|
|
212
|
-
case BACKTICK:
|
|
213
|
-
return this.#parse_code();
|
|
214
|
-
case ASTERISK:
|
|
215
|
-
return this.#parse_bold();
|
|
216
|
-
case UNDERSCORE:
|
|
217
|
-
return this.#parse_italic();
|
|
218
|
-
case TILDE:
|
|
219
|
-
return this.#parse_strikethrough();
|
|
220
|
-
case LEFT_BRACKET:
|
|
221
|
-
return this.#parse_markdown_link();
|
|
222
|
-
case LEFT_ANGLE:
|
|
223
|
-
return this.#parse_tag();
|
|
224
|
-
default:
|
|
225
|
-
return this.#parse_text();
|
|
226
|
-
}
|
|
227
|
-
}
|
|
228
|
-
/**
|
|
229
|
-
* Parse backtick code: `code`
|
|
230
|
-
* Auto-links to identifiers/modules if match found.
|
|
231
|
-
* Falls back to text if unclosed, empty, or if newline encountered before closing backtick.
|
|
232
|
-
*/
|
|
233
|
-
#parse_code() {
|
|
234
|
-
const start = this.#index;
|
|
235
|
-
this.#eat('`');
|
|
236
|
-
const content_start = this.#index;
|
|
237
|
-
// Find closing backtick, but stop at newline (respect boundary for greedy matching)
|
|
238
|
-
let content_end = -1;
|
|
239
|
-
const search_limit = Math.min(this.#max_search_index, this.#template.length);
|
|
240
|
-
for (let i = this.#index; i < search_limit; i++) {
|
|
241
|
-
const char_code = this.#template.charCodeAt(i);
|
|
242
|
-
if (char_code === BACKTICK) {
|
|
243
|
-
content_end = i;
|
|
244
|
-
break;
|
|
245
|
-
}
|
|
246
|
-
if (char_code === NEWLINE) {
|
|
247
|
-
// Newline before closing backtick - treat as unclosed
|
|
248
|
-
break;
|
|
249
|
-
}
|
|
250
|
-
}
|
|
251
|
-
if (content_end === -1) {
|
|
252
|
-
// Unclosed backtick or newline encountered, treat as text
|
|
253
|
-
return this.#make_text_node('`', start);
|
|
254
|
-
}
|
|
255
|
-
const content = this.#template.slice(content_start, content_end);
|
|
256
|
-
// Empty inline code has no semantic meaning, treat as literal text
|
|
257
|
-
if (content.length === 0) {
|
|
258
|
-
return this.#make_text_node('``', start);
|
|
259
|
-
}
|
|
260
|
-
this.#index = content_end + 1;
|
|
261
|
-
return {
|
|
262
|
-
type: 'Code',
|
|
263
|
-
content,
|
|
264
|
-
start,
|
|
265
|
-
end: this.#index,
|
|
266
|
-
};
|
|
267
|
-
}
|
|
268
|
-
/**
|
|
269
|
-
* Parse bold starting with double asterisk.
|
|
270
|
-
*
|
|
271
|
-
* - **bold** = Bold node
|
|
272
|
-
*
|
|
273
|
-
* Falls back to text if unclosed or single asterisk.
|
|
274
|
-
*
|
|
275
|
-
* Bold has no word boundary restrictions and works everywhere including intraword.
|
|
276
|
-
* Examples:
|
|
277
|
-
* - `foo**bar**baz` → foo<strong>bar</strong>baz (creates bold)
|
|
278
|
-
* - `word **bold** word` → word <strong>bold</strong> word (also works)
|
|
279
|
-
*/
|
|
280
|
-
#parse_bold() {
|
|
281
|
-
const start = this.#index;
|
|
282
|
-
// Check for ** (bold)
|
|
283
|
-
if (this.#match('**')) {
|
|
284
|
-
// Bold (**) has no word boundary restrictions - works everywhere including intraword
|
|
285
|
-
this.#eat('**');
|
|
286
|
-
// Find closing ** (greedy matching - first occurrence within boundary)
|
|
287
|
-
const search_end = Math.min(this.#max_search_index, this.#template.length);
|
|
288
|
-
let close_index = this.#template.indexOf('**', this.#index);
|
|
289
|
-
// Check if close_index exceeds search boundary
|
|
290
|
-
if (close_index !== -1 && close_index >= search_end) {
|
|
291
|
-
close_index = -1;
|
|
292
|
-
}
|
|
293
|
-
if (close_index === -1) {
|
|
294
|
-
// Unclosed, treat as text
|
|
295
|
-
return this.#make_text_node('**', start);
|
|
296
|
-
}
|
|
297
|
-
// No word boundary check for closing ** - works everywhere
|
|
298
|
-
// Parse children up to closing delimiter (bounded parsing)
|
|
299
|
-
const children = this.#parse_nodes_until('**', close_index);
|
|
300
|
-
// Verify we're at the closing delimiter (could have stopped early due to paragraph break)
|
|
301
|
-
if (!this.#match('**')) {
|
|
302
|
-
// Interrupted before closing - treat as unclosed
|
|
303
|
-
return this.#make_text_node('**', start);
|
|
304
|
-
}
|
|
305
|
-
// Empty bold has no semantic meaning, treat as literal text
|
|
306
|
-
if (children.length === 0) {
|
|
307
|
-
this.#index = start;
|
|
308
|
-
return this.#make_text_node('****', start);
|
|
309
|
-
}
|
|
310
|
-
// Consume closing **
|
|
311
|
-
this.#eat('**');
|
|
312
|
-
return {
|
|
313
|
-
type: 'Bold',
|
|
314
|
-
children,
|
|
315
|
-
start,
|
|
316
|
-
end: this.#index,
|
|
317
|
-
};
|
|
318
|
-
}
|
|
319
|
-
// Single asterisk - treat as text
|
|
320
|
-
const content = this.#template[this.#index];
|
|
321
|
-
return this.#make_text_node(content, start);
|
|
322
|
-
}
|
|
323
|
-
#parse_single_delimiter_formatting(delimiter, node_type) {
|
|
324
|
-
const start = this.#index;
|
|
325
|
-
// Check if opening delimiter is at word boundary
|
|
326
|
-
if (!this.#is_at_word_boundary(this.#index, true, false)) {
|
|
327
|
-
// Intraword delimiter - treat as literal text
|
|
328
|
-
const content = this.#template[this.#index];
|
|
329
|
-
return this.#make_text_node(content, start);
|
|
330
|
-
}
|
|
331
|
-
this.#eat(delimiter);
|
|
332
|
-
// Find closing delimiter (greedy matching - first occurrence within boundary)
|
|
333
|
-
const search_end = Math.min(this.#max_search_index, this.#template.length);
|
|
334
|
-
let close_index = this.#template.indexOf(delimiter, this.#index);
|
|
335
|
-
// Check if close_index exceeds search boundary
|
|
336
|
-
if (close_index !== -1 && close_index >= search_end) {
|
|
337
|
-
close_index = -1;
|
|
338
|
-
}
|
|
339
|
-
if (close_index === -1) {
|
|
340
|
-
// Unclosed, treat as text
|
|
341
|
-
return this.#make_text_node(delimiter, start);
|
|
342
|
-
}
|
|
343
|
-
// Check if closing delimiter is at word boundary
|
|
344
|
-
if (!this.#is_at_word_boundary(close_index + 1, false, true)) {
|
|
345
|
-
// Closing delimiter not at boundary - treat whole thing as text
|
|
346
|
-
return this.#make_text_node(delimiter, start);
|
|
347
|
-
}
|
|
348
|
-
// Parse children up to closing delimiter (bounded parsing)
|
|
349
|
-
const children = this.#parse_nodes_until(delimiter, close_index);
|
|
350
|
-
// Verify we're at the closing delimiter (could have stopped early due to paragraph break)
|
|
351
|
-
if (!this.#match(delimiter)) {
|
|
352
|
-
// Interrupted before closing - treat as unclosed
|
|
353
|
-
return this.#make_text_node(delimiter, start);
|
|
354
|
-
}
|
|
355
|
-
// Empty formatting has no semantic meaning, treat as literal text
|
|
356
|
-
if (children.length === 0) {
|
|
357
|
-
this.#index = start;
|
|
358
|
-
return this.#make_text_node(delimiter + delimiter, start);
|
|
359
|
-
}
|
|
360
|
-
// Consume closing delimiter
|
|
361
|
-
this.#eat(delimiter);
|
|
362
|
-
return {
|
|
363
|
-
type: node_type,
|
|
364
|
-
children,
|
|
365
|
-
start,
|
|
366
|
-
end: this.#index,
|
|
367
|
-
};
|
|
368
|
-
}
|
|
369
|
-
/**
|
|
370
|
-
* Parse italic starting with underscore.
|
|
371
|
-
* _italic_ = Italic node
|
|
372
|
-
* Falls back to text if unclosed or not at word boundary.
|
|
373
|
-
*
|
|
374
|
-
* Following GFM spec: underscores cannot create emphasis in middle of words.
|
|
375
|
-
* Examples:
|
|
376
|
-
* - `foo_bar_baz` → literal text (intraword)
|
|
377
|
-
* - `word _emphasis_ word` → emphasis (at word boundaries)
|
|
378
|
-
*/
|
|
379
|
-
#parse_italic() {
|
|
380
|
-
return this.#parse_single_delimiter_formatting('_', 'Italic');
|
|
381
|
-
}
|
|
382
|
-
/**
|
|
383
|
-
* Parse strikethrough starting with tilde.
|
|
384
|
-
* ~strikethrough~ = Strikethrough node
|
|
385
|
-
* Falls back to text if unclosed or not at word boundary.
|
|
386
|
-
*
|
|
387
|
-
* Following mdz philosophy (false negatives over false positives):
|
|
388
|
-
* Strikethrough requires word boundaries to prevent intraword formatting.
|
|
389
|
-
* Examples:
|
|
390
|
-
* - `foo~bar~baz` → literal text (intraword)
|
|
391
|
-
* - `word ~strike~ word` → strikethrough (at word boundaries)
|
|
392
|
-
*/
|
|
393
|
-
#parse_strikethrough() {
|
|
394
|
-
return this.#parse_single_delimiter_formatting('~', 'Strikethrough');
|
|
395
|
-
}
|
|
396
|
-
/**
|
|
397
|
-
* Parse markdown link: `[text](url)`.
|
|
398
|
-
* Falls back to text if malformed.
|
|
399
|
-
*/
|
|
400
|
-
#parse_markdown_link() {
|
|
401
|
-
const start = this.#index;
|
|
402
|
-
// Consume opening [
|
|
403
|
-
if (!this.#match('[')) {
|
|
404
|
-
const content = this.#template[this.#index];
|
|
405
|
-
this.#index++;
|
|
406
|
-
return {
|
|
407
|
-
type: 'Text',
|
|
408
|
-
content,
|
|
409
|
-
start,
|
|
410
|
-
end: this.#index,
|
|
411
|
-
};
|
|
412
|
-
}
|
|
413
|
-
this.#index++;
|
|
414
|
-
// Parse children nodes until closing ]
|
|
415
|
-
const children = this.#parse_nodes_until(']');
|
|
416
|
-
// Check if we found the closing ]
|
|
417
|
-
if (!this.#match(']')) {
|
|
418
|
-
// No closing ], treat as text
|
|
419
|
-
this.#index = start + 1;
|
|
420
|
-
return {
|
|
421
|
-
type: 'Text',
|
|
422
|
-
content: '[',
|
|
423
|
-
start,
|
|
424
|
-
end: this.#index,
|
|
425
|
-
};
|
|
426
|
-
}
|
|
427
|
-
this.#index++; // consume ]
|
|
428
|
-
// Check for opening (
|
|
429
|
-
if (this.#index >= this.#template.length ||
|
|
430
|
-
this.#template.charCodeAt(this.#index) !== LEFT_PAREN) {
|
|
431
|
-
// No opening (, treat as text
|
|
432
|
-
this.#index = start + 1;
|
|
433
|
-
return {
|
|
434
|
-
type: 'Text',
|
|
435
|
-
content: '[',
|
|
436
|
-
start,
|
|
437
|
-
end: this.#index,
|
|
438
|
-
};
|
|
439
|
-
}
|
|
440
|
-
this.#index++;
|
|
441
|
-
// Find closing )
|
|
442
|
-
const close_paren = this.#template.indexOf(')', this.#index);
|
|
443
|
-
if (close_paren === -1) {
|
|
444
|
-
// No closing ), treat as text
|
|
445
|
-
this.#index = start + 1;
|
|
446
|
-
return {
|
|
447
|
-
type: 'Text',
|
|
448
|
-
content: '[',
|
|
449
|
-
start,
|
|
450
|
-
end: this.#index,
|
|
451
|
-
};
|
|
452
|
-
}
|
|
453
|
-
// Extract URL/path
|
|
454
|
-
const reference = this.#template.slice(this.#index, close_paren);
|
|
455
|
-
// Validate reference is not empty or whitespace-only
|
|
456
|
-
if (!reference.trim()) {
|
|
457
|
-
// Empty reference, treat as text
|
|
458
|
-
this.#index = start + 1;
|
|
459
|
-
return {
|
|
460
|
-
type: 'Text',
|
|
461
|
-
content: '[',
|
|
462
|
-
start,
|
|
463
|
-
end: this.#index,
|
|
464
|
-
};
|
|
465
|
-
}
|
|
466
|
-
// Validate all characters in reference are valid URI characters per RFC 3986
|
|
467
|
-
// This prevents spaces and other invalid characters from being in markdown link URLs
|
|
468
|
-
// Follows GFM behavior: invalid chars cause fallback to text, then auto-detection
|
|
469
|
-
for (let i = 0; i < reference.length; i++) {
|
|
470
|
-
const char_code = reference.charCodeAt(i);
|
|
471
|
-
if (!is_valid_path_char(char_code)) {
|
|
472
|
-
// Invalid character in URL, treat as text and let auto-detection handle it
|
|
473
|
-
this.#index = start + 1;
|
|
474
|
-
return {
|
|
475
|
-
type: 'Text',
|
|
476
|
-
content: '[',
|
|
477
|
-
start,
|
|
478
|
-
end: this.#index,
|
|
479
|
-
};
|
|
480
|
-
}
|
|
481
|
-
}
|
|
482
|
-
// Reject unsafe protocols (javascript:, data:, etc.) — fall back to text.
|
|
483
|
-
// `is_valid_path_char` permits `:` so a malicious `[x](javascript:...)` would
|
|
484
|
-
// otherwise pass character validation and reach the renderer.
|
|
485
|
-
if (!mdz_is_safe_reference(reference)) {
|
|
486
|
-
this.#index = start + 1;
|
|
487
|
-
return {
|
|
488
|
-
type: 'Text',
|
|
489
|
-
content: '[',
|
|
490
|
-
start,
|
|
491
|
-
end: this.#index,
|
|
492
|
-
};
|
|
493
|
-
}
|
|
494
|
-
this.#index = close_paren + 1;
|
|
495
|
-
// Determine link type (external vs internal)
|
|
496
|
-
const link_type = mdz_is_url(reference) ? 'external' : 'internal';
|
|
497
|
-
return {
|
|
498
|
-
type: 'Link',
|
|
499
|
-
reference,
|
|
500
|
-
children,
|
|
501
|
-
link_type,
|
|
502
|
-
start,
|
|
503
|
-
end: this.#index,
|
|
504
|
-
};
|
|
505
|
-
}
|
|
506
|
-
/**
|
|
507
|
-
* Parse component/element tag: `<TagName>content</TagName>` or `<TagName />`
|
|
508
|
-
*
|
|
509
|
-
* Formats:
|
|
510
|
-
* - `<Alert>content</Alert>` - Svelte component with children (uppercase first letter)
|
|
511
|
-
* - `<div>content</div>` - HTML element with children (lowercase first letter)
|
|
512
|
-
* - `<Alert />` - self-closing component/element
|
|
513
|
-
*
|
|
514
|
-
* Tag names must start with a letter and can contain letters, numbers, hyphens, underscores.
|
|
515
|
-
*
|
|
516
|
-
* Falls back to text if malformed or unclosed.
|
|
517
|
-
*
|
|
518
|
-
* TODO: Add attribute support like `<Alert status="error">` or `<div class="container">`
|
|
519
|
-
*/
|
|
520
|
-
#parse_tag() {
|
|
521
|
-
const start = this.#index;
|
|
522
|
-
// Phase 1: Validate tag structure using a local scan index.
|
|
523
|
-
// Avoids save/restore of accumulation state on the fast path
|
|
524
|
-
// (most `<` characters are not valid tags).
|
|
525
|
-
let i = start + 1; // skip <
|
|
526
|
-
// Tag name must start with a letter
|
|
527
|
-
if (i >= this.#template.length || !is_letter(this.#template.charCodeAt(i))) {
|
|
528
|
-
this.#index = start + 1;
|
|
529
|
-
return this.#make_text_node('<', start);
|
|
530
|
-
}
|
|
531
|
-
// Collect tag name (letters, numbers, hyphens, underscores)
|
|
532
|
-
while (i < this.#template.length && is_tag_name_char(this.#template.charCodeAt(i))) {
|
|
533
|
-
i++;
|
|
534
|
-
}
|
|
535
|
-
const tag_name = this.#template.slice(start + 1, i);
|
|
536
|
-
const first_char_code = tag_name.charCodeAt(0);
|
|
537
|
-
const node_type = first_char_code >= A_UPPER && first_char_code <= Z_UPPER ? 'Component' : 'Element';
|
|
538
|
-
// Skip whitespace after tag name (for future attribute support)
|
|
539
|
-
while (i < this.#template.length && this.#template.charCodeAt(i) === SPACE) {
|
|
540
|
-
i++;
|
|
541
|
-
}
|
|
542
|
-
// TODO: Parse attributes here
|
|
543
|
-
// Check for self-closing />
|
|
544
|
-
if (i + 1 < this.#template.length &&
|
|
545
|
-
this.#template.charCodeAt(i) === SLASH &&
|
|
546
|
-
this.#template.charCodeAt(i + 1) === RIGHT_ANGLE) {
|
|
547
|
-
this.#index = i + 2;
|
|
548
|
-
return {
|
|
549
|
-
type: node_type,
|
|
550
|
-
name: tag_name,
|
|
551
|
-
children: [],
|
|
552
|
-
start,
|
|
553
|
-
end: this.#index,
|
|
554
|
-
};
|
|
555
|
-
}
|
|
556
|
-
// Must have closing >
|
|
557
|
-
if (i >= this.#template.length || this.#template.charCodeAt(i) !== RIGHT_ANGLE) {
|
|
558
|
-
this.#index = start + 1;
|
|
559
|
-
return this.#make_text_node('<', start);
|
|
560
|
-
}
|
|
561
|
-
const content_start = i + 1; // past >
|
|
562
|
-
// Bail early if closing tag is missing, past a paragraph break,
|
|
563
|
-
// or past the search boundary (e.g. heading line end)
|
|
564
|
-
const closing_tag = `</${tag_name}>`;
|
|
565
|
-
const search_limit = Math.min(this.#max_search_index, this.#template.length);
|
|
566
|
-
const close_tag_index = this.#template.indexOf(closing_tag, content_start);
|
|
567
|
-
if (close_tag_index === -1 || close_tag_index >= search_limit) {
|
|
568
|
-
this.#index = start + 1;
|
|
569
|
-
return this.#make_text_node('<', start);
|
|
570
|
-
}
|
|
571
|
-
const para_break = this.#template.indexOf('\n\n', content_start);
|
|
572
|
-
if (para_break !== -1 && para_break < close_tag_index) {
|
|
573
|
-
this.#index = start + 1;
|
|
574
|
-
return this.#make_text_node('<', start);
|
|
575
|
-
}
|
|
576
|
-
// Phase 2: Tag validated — save accumulation state and parse children.
|
|
577
|
-
const saved_state = this.#save_accumulation_state();
|
|
578
|
-
this.#accumulated_text = '';
|
|
579
|
-
this.#nodes.length = 0;
|
|
580
|
-
this.#index = content_start;
|
|
581
|
-
const children = [];
|
|
582
|
-
while (this.#index < this.#template.length) {
|
|
583
|
-
if (this.#match(closing_tag)) {
|
|
584
|
-
this.#flush_text();
|
|
585
|
-
children.push(...this.#nodes);
|
|
586
|
-
this.#nodes.length = 0;
|
|
587
|
-
this.#index += closing_tag.length;
|
|
588
|
-
this.#restore_accumulation_state(saved_state);
|
|
589
|
-
return {
|
|
590
|
-
type: node_type,
|
|
591
|
-
name: tag_name,
|
|
592
|
-
children,
|
|
593
|
-
start,
|
|
594
|
-
end: this.#index,
|
|
595
|
-
};
|
|
596
|
-
}
|
|
597
|
-
const node = this.#parse_node();
|
|
598
|
-
if (node.type === 'Text') {
|
|
599
|
-
this.#accumulate_text(node.content, node.start);
|
|
600
|
-
}
|
|
601
|
-
else {
|
|
602
|
-
this.#flush_text();
|
|
603
|
-
children.push(...this.#nodes);
|
|
604
|
-
this.#nodes.length = 0;
|
|
605
|
-
children.push(node);
|
|
606
|
-
}
|
|
607
|
-
}
|
|
608
|
-
// Defensive: pre-check guarantees closing tag exists, but handle EOF gracefully
|
|
609
|
-
this.#restore_accumulation_state(saved_state);
|
|
610
|
-
this.#index = start + 1;
|
|
611
|
-
return this.#make_text_node('<', start);
|
|
612
|
-
}
|
|
613
|
-
/**
|
|
614
|
-
* Read-only check if current position matches a block element.
|
|
615
|
-
* Does not modify parser state — used to peek before flushing paragraph.
|
|
616
|
-
* Caller must verify column-0 position before calling.
|
|
617
|
-
*/
|
|
618
|
-
#peek_block_element() {
|
|
619
|
-
if (this.#match_heading())
|
|
620
|
-
return 'heading';
|
|
621
|
-
if (this.#match_hr())
|
|
622
|
-
return 'hr';
|
|
623
|
-
if (this.#match_code_block())
|
|
624
|
-
return 'codeblock';
|
|
625
|
-
return null;
|
|
626
|
-
}
|
|
627
|
-
/**
|
|
628
|
-
* Save current text accumulation state.
|
|
629
|
-
* Used when parsing nested structures (like components/elements) that need isolated accumulation.
|
|
630
|
-
* Returns state object that can be passed to `#restore_accumulation_state()`.
|
|
631
|
-
*/
|
|
632
|
-
#save_accumulation_state() {
|
|
633
|
-
return {
|
|
634
|
-
accumulated_text: this.#accumulated_text,
|
|
635
|
-
accumulated_start: this.#accumulated_start,
|
|
636
|
-
nodes: this.#nodes.slice(),
|
|
637
|
-
};
|
|
638
|
-
}
|
|
639
|
-
/**
|
|
640
|
-
* Restore previously saved text accumulation state.
|
|
641
|
-
* Used to restore parent state when exiting nested structure parsing.
|
|
642
|
-
* @param state - state object returned from `#save_accumulation_state()`
|
|
643
|
-
*/
|
|
644
|
-
#restore_accumulation_state(state) {
|
|
645
|
-
this.#accumulated_text = state.accumulated_text;
|
|
646
|
-
this.#accumulated_start = state.accumulated_start;
|
|
647
|
-
this.#nodes.length = 0;
|
|
648
|
-
this.#nodes.push(...state.nodes);
|
|
649
|
-
}
|
|
650
|
-
/**
|
|
651
|
-
* Check if position is at a word boundary.
|
|
652
|
-
* Word boundary = not surrounded by word characters (A-Z, a-z, 0-9).
|
|
653
|
-
* Used to prevent intraword emphasis for underscores and tildes.
|
|
654
|
-
*
|
|
655
|
-
* @param index - position to check
|
|
656
|
-
* @param check_before - whether to check the character before this position
|
|
657
|
-
* @param check_after - whether to check the character after this position
|
|
658
|
-
*/
|
|
659
|
-
#is_at_word_boundary(index, check_before, check_after) {
|
|
660
|
-
if (check_before && index > 0) {
|
|
661
|
-
const prev = this.#template.charCodeAt(index - 1);
|
|
662
|
-
// If preceded by word char, not at boundary
|
|
663
|
-
if (is_word_char(prev)) {
|
|
664
|
-
return false;
|
|
665
|
-
}
|
|
666
|
-
}
|
|
667
|
-
if (check_after && index < this.#template.length) {
|
|
668
|
-
const next = this.#template.charCodeAt(index);
|
|
669
|
-
// If followed by word char, not at boundary
|
|
670
|
-
if (is_word_char(next)) {
|
|
671
|
-
return false;
|
|
672
|
-
}
|
|
673
|
-
}
|
|
674
|
-
return true;
|
|
675
|
-
}
|
|
676
|
-
/**
|
|
677
|
-
* Check if current position is the start of an external URL (`https://` or `http://`).
|
|
678
|
-
*
|
|
679
|
-
* Requires the preceding character (if any) to not be a word character. This
|
|
680
|
-
* matches the streaming parser and the spec's "false negatives over false
|
|
681
|
-
* positives" policy — `xhttps://...` is treated as plain text rather than
|
|
682
|
-
* `x` followed by a link.
|
|
683
|
-
*
|
|
684
|
-
* Scheme matching is case-insensitive (RFC 3986 §3.1); the original casing
|
|
685
|
-
* is preserved in the emitted reference.
|
|
686
|
-
*/
|
|
687
|
-
#is_at_url() {
|
|
688
|
-
// word boundary check: skip when preceded by [A-Za-z0-9]
|
|
689
|
-
if (this.#index > 0 && is_word_char(this.#template.charCodeAt(this.#index - 1))) {
|
|
690
|
-
return false;
|
|
691
|
-
}
|
|
692
|
-
const prefix_len = match_url_prefix_case_insensitive(this.#template, this.#index);
|
|
693
|
-
if (prefix_len === 0)
|
|
694
|
-
return false;
|
|
695
|
-
// Must have at least one non-whitespace character after protocol
|
|
696
|
-
if (this.#index + prefix_len >= this.#template.length)
|
|
697
|
-
return false;
|
|
698
|
-
const next_char = this.#template.charCodeAt(this.#index + prefix_len);
|
|
699
|
-
return next_char !== SPACE && next_char !== NEWLINE;
|
|
700
|
-
}
|
|
701
|
-
/**
|
|
702
|
-
* Parse auto-detected external URL (`https://` or `http://`).
|
|
703
|
-
* Uses RFC 3986 whitelist validation for valid URI characters.
|
|
704
|
-
*/
|
|
705
|
-
#parse_auto_link_url() {
|
|
706
|
-
const start = this.#index;
|
|
707
|
-
// Consume protocol (case-insensitive match; original casing preserved in reference)
|
|
708
|
-
this.#index += match_url_prefix_case_insensitive(this.#template, this.#index);
|
|
709
|
-
// Collect URL characters using RFC 3986 whitelist
|
|
710
|
-
// Stop at whitespace or any character invalid in URIs
|
|
711
|
-
while (this.#index < this.#template.length) {
|
|
712
|
-
const char_code = this.#template.charCodeAt(this.#index);
|
|
713
|
-
if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
|
|
714
|
-
break;
|
|
715
|
-
}
|
|
716
|
-
this.#index++;
|
|
717
|
-
}
|
|
718
|
-
let reference = this.#template.slice(start, this.#index);
|
|
719
|
-
// Apply GFM trailing punctuation trimming with balanced parentheses
|
|
720
|
-
reference = trim_trailing_punctuation(reference);
|
|
721
|
-
// Update index after trimming
|
|
722
|
-
this.#index = start + reference.length;
|
|
723
|
-
return {
|
|
724
|
-
type: 'Link',
|
|
725
|
-
reference,
|
|
726
|
-
children: [{ type: 'Text', content: reference, start, end: this.#index }],
|
|
727
|
-
link_type: 'external',
|
|
728
|
-
start,
|
|
729
|
-
end: this.#index,
|
|
730
|
-
};
|
|
731
|
-
}
|
|
732
|
-
/**
|
|
733
|
-
* Parse auto-detected path (absolute `/`, relative `./` or `../`).
|
|
734
|
-
* Uses RFC 3986 whitelist validation for valid URI characters.
|
|
735
|
-
*/
|
|
736
|
-
#parse_auto_link_path() {
|
|
737
|
-
const start = this.#index;
|
|
738
|
-
// Collect path characters using RFC 3986 whitelist
|
|
739
|
-
// Stop at whitespace or any character invalid in URIs
|
|
740
|
-
while (this.#index < this.#template.length) {
|
|
741
|
-
const char_code = this.#template.charCodeAt(this.#index);
|
|
742
|
-
if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
|
|
743
|
-
break;
|
|
744
|
-
}
|
|
745
|
-
this.#index++;
|
|
746
|
-
}
|
|
747
|
-
let reference = this.#template.slice(start, this.#index);
|
|
748
|
-
// Apply GFM trailing punctuation trimming
|
|
749
|
-
reference = trim_trailing_punctuation(reference);
|
|
750
|
-
// Update index after trimming
|
|
751
|
-
this.#index = start + reference.length;
|
|
752
|
-
return {
|
|
753
|
-
type: 'Link',
|
|
754
|
-
reference,
|
|
755
|
-
children: [{ type: 'Text', content: reference, start, end: this.#index }],
|
|
756
|
-
link_type: 'internal',
|
|
757
|
-
start,
|
|
758
|
-
end: this.#index,
|
|
759
|
-
};
|
|
760
|
-
}
|
|
761
|
-
/**
|
|
762
|
-
* Parse plain text until special character encountered.
|
|
763
|
-
* Preserves all whitespace (except paragraph breaks handled separately).
|
|
764
|
-
* Detects and delegates to URL/path parsing when encountered.
|
|
765
|
-
*/
|
|
766
|
-
#parse_text() {
|
|
767
|
-
const start = this.#index;
|
|
768
|
-
// Check for URL or internal absolute/relative path at current position
|
|
769
|
-
if (this.#is_at_url()) {
|
|
770
|
-
return this.#parse_auto_link_url();
|
|
771
|
-
}
|
|
772
|
-
if (is_at_absolute_path(this.#template, this.#index) ||
|
|
773
|
-
is_at_relative_path(this.#template, this.#index)) {
|
|
774
|
-
return this.#parse_auto_link_path();
|
|
775
|
-
}
|
|
776
|
-
while (this.#index < this.#template.length) {
|
|
777
|
-
const char_code = this.#template.charCodeAt(this.#index);
|
|
778
|
-
// Stop at special characters (but preserve single newlines)
|
|
779
|
-
if (char_code === BACKTICK ||
|
|
780
|
-
char_code === ASTERISK ||
|
|
781
|
-
char_code === UNDERSCORE ||
|
|
782
|
-
char_code === TILDE ||
|
|
783
|
-
char_code === LEFT_BRACKET ||
|
|
784
|
-
char_code === RIGHT_BRACKET ||
|
|
785
|
-
char_code === RIGHT_PAREN ||
|
|
786
|
-
char_code === LEFT_ANGLE) {
|
|
787
|
-
break;
|
|
788
|
-
}
|
|
789
|
-
// Check for paragraph break (double newline)
|
|
790
|
-
if (this.#is_at_paragraph_break()) {
|
|
791
|
-
break;
|
|
792
|
-
}
|
|
793
|
-
// When next line could start a block element, consume the newline and stop.
|
|
794
|
-
// The main loop will try block detection at the next character.
|
|
795
|
-
// Consuming the newline here avoids a 3-iteration detour (break before \n,
|
|
796
|
-
// fail peek at \n, safety-increment, then succeed peek at block char).
|
|
797
|
-
if (char_code === NEWLINE) {
|
|
798
|
-
const next_i = this.#index + 1;
|
|
799
|
-
if (next_i < this.#template.length) {
|
|
800
|
-
const next_char = this.#template.charCodeAt(next_i);
|
|
801
|
-
if (next_char === HASH || next_char === HYPHEN || next_char === BACKTICK) {
|
|
802
|
-
this.#index++; // consume the newline
|
|
803
|
-
break;
|
|
804
|
-
}
|
|
805
|
-
}
|
|
806
|
-
}
|
|
807
|
-
// Check for URL or internal absolute/relative path mid-text (char code guard avoids startsWith on every char)
|
|
808
|
-
if (((char_code === 104 /* h */ || char_code === 72) /* H */ && this.#is_at_url()) ||
|
|
809
|
-
(char_code === SLASH && is_at_absolute_path(this.#template, this.#index)) ||
|
|
810
|
-
(char_code === PERIOD && is_at_relative_path(this.#template, this.#index))) {
|
|
811
|
-
break;
|
|
812
|
-
}
|
|
813
|
-
this.#index++;
|
|
814
|
-
}
|
|
815
|
-
// Ensure we always consume at least one character to prevent infinite loops
|
|
816
|
-
if (this.#index === start && this.#index < this.#template.length) {
|
|
817
|
-
this.#index++;
|
|
818
|
-
}
|
|
819
|
-
// Use slice instead of concatenation for performance
|
|
820
|
-
const content = this.#template.slice(start, this.#index);
|
|
821
|
-
return {
|
|
822
|
-
type: 'Text',
|
|
823
|
-
content,
|
|
824
|
-
start,
|
|
825
|
-
end: this.#index,
|
|
826
|
-
};
|
|
827
|
-
}
|
|
828
|
-
/**
|
|
829
|
-
* Parse nodes until delimiter string is found.
|
|
830
|
-
* Used for parsing children of inline formatting (bold, italic, strikethrough) and markdown links.
|
|
831
|
-
*
|
|
832
|
-
* Implements greedy/bounded parsing to prevent nested formatters from consuming parent delimiters:
|
|
833
|
-
* - When parsing `**bold with _italic_**`, the outer `**` parser finds its closing delimiter at position Y
|
|
834
|
-
* - Sets `#max_search_index = Y` to create a boundary
|
|
835
|
-
* - Parses children only within range, preventing `_italic_` from finding delimiters beyond Y
|
|
836
|
-
* - This ensures proper nesting without backtracking
|
|
837
|
-
*
|
|
838
|
-
* Stops parsing when:
|
|
839
|
-
* - Delimiter string is found
|
|
840
|
-
* - Paragraph break (double newline) is encountered (allows block elements to interrupt inline formatting)
|
|
841
|
-
* - `end_index` boundary is reached
|
|
842
|
-
*
|
|
843
|
-
* @param delimiter - the delimiter string to stop at (e.g., '**', '_', ']')
|
|
844
|
-
* @param end_index - optional maximum index to parse up to (for greedy/bounded parsing)
|
|
845
|
-
* @returns array of parsed nodes (may be empty if delimiter found immediately)
|
|
846
|
-
*/
|
|
847
|
-
#parse_nodes_until(delimiter, end_index) {
|
|
848
|
-
const nodes = [];
|
|
849
|
-
const max_index = end_index ?? this.#template.length;
|
|
850
|
-
// Save and set max search boundary for nested parsers
|
|
851
|
-
const saved_max_search_index = this.#max_search_index;
|
|
852
|
-
this.#max_search_index = max_index;
|
|
853
|
-
while (this.#index < max_index) {
|
|
854
|
-
if (this.#match(delimiter)) {
|
|
855
|
-
break;
|
|
856
|
-
}
|
|
857
|
-
// Check for paragraph break (block element interruption)
|
|
858
|
-
if (this.#is_at_paragraph_break()) {
|
|
859
|
-
// Paragraph break interrupts inline formatting
|
|
860
|
-
break;
|
|
861
|
-
}
|
|
862
|
-
// merge adjacent Text nodes (e.g., text + failed delimiter + text)
|
|
863
|
-
mdz_push_merging_text(nodes, this.#parse_node());
|
|
864
|
-
}
|
|
865
|
-
// Restore previous boundary
|
|
866
|
-
this.#max_search_index = saved_max_search_index;
|
|
867
|
-
return nodes;
|
|
868
|
-
}
|
|
869
|
-
/**
|
|
870
|
-
* Check if current position is at a paragraph break (double newline).
|
|
871
|
-
*/
|
|
872
|
-
#is_at_paragraph_break() {
|
|
873
|
-
return (this.#index + 1 < this.#template.length &&
|
|
874
|
-
this.#template.charCodeAt(this.#index) === NEWLINE &&
|
|
875
|
-
this.#template.charCodeAt(this.#index + 1) === NEWLINE);
|
|
876
|
-
}
|
|
877
|
-
#match(str) {
|
|
878
|
-
return this.#template.startsWith(str, this.#index);
|
|
879
|
-
}
|
|
880
|
-
/**
|
|
881
|
-
* Consume string at current index, or throw error.
|
|
882
|
-
*/
|
|
883
|
-
#eat(str) {
|
|
884
|
-
if (this.#match(str)) {
|
|
885
|
-
this.#index += str.length;
|
|
886
|
-
}
|
|
887
|
-
else {
|
|
888
|
-
throw Error(`Expected "${str}" at index ${this.#index}`);
|
|
889
|
-
}
|
|
890
|
-
}
|
|
891
|
-
/**
|
|
892
|
-
* Check if current position matches a horizontal rule.
|
|
893
|
-
* HR must be exactly `---` at column 0, followed by newline or EOF.
|
|
894
|
-
*
|
|
895
|
-
* mdz has no setext headings, so `---` after a paragraph is unambiguous
|
|
896
|
-
* (always an HR, unlike CommonMark where it becomes a setext heading).
|
|
897
|
-
*/
|
|
898
|
-
#match_hr() {
|
|
899
|
-
let i = this.#index;
|
|
900
|
-
// Must have exactly three hyphens
|
|
901
|
-
if (i + HR_HYPHEN_COUNT > this.#template.length ||
|
|
902
|
-
this.#template.charCodeAt(i) !== HYPHEN ||
|
|
903
|
-
this.#template.charCodeAt(i + 1) !== HYPHEN ||
|
|
904
|
-
this.#template.charCodeAt(i + 2) !== HYPHEN) {
|
|
905
|
-
return false;
|
|
906
|
-
}
|
|
907
|
-
i += HR_HYPHEN_COUNT;
|
|
908
|
-
// After the three hyphens, only whitespace and newline (or EOF) allowed
|
|
909
|
-
while (i < this.#template.length) {
|
|
910
|
-
const char_code = this.#template.charCodeAt(i);
|
|
911
|
-
if (char_code === NEWLINE) {
|
|
912
|
-
return true;
|
|
913
|
-
}
|
|
914
|
-
if (char_code !== SPACE) {
|
|
915
|
-
return false; // Non-whitespace after ---, not an hr
|
|
916
|
-
}
|
|
917
|
-
i++;
|
|
918
|
-
}
|
|
919
|
-
// Reached EOF after ---, valid hr
|
|
920
|
-
return true;
|
|
921
|
-
}
|
|
922
|
-
/**
|
|
923
|
-
* Parse horizontal rule: `---`
|
|
924
|
-
* Assumes #match_hr() already verified this is an hr.
|
|
925
|
-
*/
|
|
926
|
-
#parse_hr() {
|
|
927
|
-
const start = this.#index;
|
|
928
|
-
// Consume the three hyphens (no leading whitespace - already verified)
|
|
929
|
-
this.#index += HR_HYPHEN_COUNT;
|
|
930
|
-
// Skip trailing whitespace
|
|
931
|
-
while (this.#index < this.#template.length &&
|
|
932
|
-
this.#template.charCodeAt(this.#index) === SPACE) {
|
|
933
|
-
this.#index++;
|
|
934
|
-
}
|
|
935
|
-
// Don't consume the newline - let the main parse loop handle it
|
|
936
|
-
return {
|
|
937
|
-
type: 'Hr',
|
|
938
|
-
start,
|
|
939
|
-
end: this.#index,
|
|
940
|
-
};
|
|
941
|
-
}
|
|
942
|
-
/**
|
|
943
|
-
* Check if current position matches a heading.
|
|
944
|
-
* Heading must be 1-6 hashes at column 0, followed by space and content,
|
|
945
|
-
* followed by newline or EOF.
|
|
946
|
-
*/
|
|
947
|
-
#match_heading() {
|
|
948
|
-
let i = this.#index;
|
|
949
|
-
// Count hashes (must be 1-6)
|
|
950
|
-
let hash_count = 0;
|
|
951
|
-
while (i < this.#template.length &&
|
|
952
|
-
this.#template.charCodeAt(i) === HASH &&
|
|
953
|
-
hash_count <= MAX_HEADING_LEVEL) {
|
|
954
|
-
hash_count++;
|
|
955
|
-
i++;
|
|
956
|
-
}
|
|
957
|
-
if (hash_count === 0 || hash_count > MAX_HEADING_LEVEL) {
|
|
958
|
-
return false;
|
|
959
|
-
}
|
|
960
|
-
// Must have space after hashes
|
|
961
|
-
if (i >= this.#template.length || this.#template.charCodeAt(i) !== SPACE) {
|
|
962
|
-
return false;
|
|
963
|
-
}
|
|
964
|
-
i++; // consume the space
|
|
965
|
-
// Must have at least one non-whitespace character after the space
|
|
966
|
-
while (i < this.#template.length) {
|
|
967
|
-
const char_code = this.#template.charCodeAt(i);
|
|
968
|
-
if (char_code === NEWLINE)
|
|
969
|
-
return false; // reached end of line with only whitespace
|
|
970
|
-
if (char_code !== SPACE && char_code !== TAB)
|
|
971
|
-
return true;
|
|
972
|
-
i++;
|
|
973
|
-
}
|
|
974
|
-
// Reached EOF with only whitespace after hashes
|
|
975
|
-
return false;
|
|
976
|
-
}
|
|
977
|
-
/**
|
|
978
|
-
* Parse heading: `# Heading text`
|
|
979
|
-
* Assumes #match_heading() already verified this is a heading.
|
|
980
|
-
*/
|
|
981
|
-
#parse_heading() {
|
|
982
|
-
const start = this.#index;
|
|
983
|
-
// Count and consume hashes
|
|
984
|
-
let level = 0;
|
|
985
|
-
while (this.#index < this.#template.length && this.#template.charCodeAt(this.#index) === HASH) {
|
|
986
|
-
level++;
|
|
987
|
-
this.#index++;
|
|
988
|
-
}
|
|
989
|
-
// Consume the space after hashes (already verified to exist)
|
|
990
|
-
this.#index++;
|
|
991
|
-
// Find end-of-line to bound nested parsers (prevents tag scanner from scanning past heading)
|
|
992
|
-
let eol = this.#template.indexOf('\n', this.#index);
|
|
993
|
-
if (eol === -1)
|
|
994
|
-
eol = this.#template.length;
|
|
995
|
-
const saved_max_search_index = this.#max_search_index;
|
|
996
|
-
this.#max_search_index = eol;
|
|
997
|
-
// Parse inline content until end of line
|
|
998
|
-
const content_nodes = [];
|
|
999
|
-
while (this.#index < eol) {
|
|
1000
|
-
const node = this.#parse_node();
|
|
1001
|
-
if (node.type === 'Text') {
|
|
1002
|
-
// Trim if #parse_text overshot past the newline
|
|
1003
|
-
if (node.end > eol) {
|
|
1004
|
-
const trimmed_content = node.content.slice(0, eol - node.start);
|
|
1005
|
-
if (trimmed_content) {
|
|
1006
|
-
this.#accumulate_text(trimmed_content, node.start);
|
|
1007
|
-
}
|
|
1008
|
-
this.#index = eol;
|
|
1009
|
-
break;
|
|
1010
|
-
}
|
|
1011
|
-
this.#accumulate_text(node.content, node.start);
|
|
1012
|
-
}
|
|
1013
|
-
else {
|
|
1014
|
-
this.#flush_text();
|
|
1015
|
-
content_nodes.push(...this.#nodes);
|
|
1016
|
-
this.#nodes.length = 0;
|
|
1017
|
-
content_nodes.push(node);
|
|
1018
|
-
}
|
|
1019
|
-
}
|
|
1020
|
-
this.#max_search_index = saved_max_search_index;
|
|
1021
|
-
this.#flush_text();
|
|
1022
|
-
content_nodes.push(...this.#nodes);
|
|
1023
|
-
this.#nodes.length = 0;
|
|
1024
|
-
// Don't consume the newline - let the main parse loop handle it
|
|
1025
|
-
return {
|
|
1026
|
-
type: 'Heading',
|
|
1027
|
-
level: level,
|
|
1028
|
-
id: mdz_heading_id(content_nodes),
|
|
1029
|
-
children: content_nodes,
|
|
1030
|
-
start,
|
|
1031
|
-
end: this.#index,
|
|
1032
|
-
};
|
|
1033
|
-
}
|
|
1034
|
-
/**
|
|
1035
|
-
* Check if current position matches a code block.
|
|
1036
|
-
* Code block must be 3+ backticks at column 0, closing fence followed by newline or EOF.
|
|
1037
|
-
* Empty code blocks (no content) are treated as invalid.
|
|
1038
|
-
*/
|
|
1039
|
-
#match_code_block() {
|
|
1040
|
-
let i = this.#index;
|
|
1041
|
-
// Must have at least three backticks
|
|
1042
|
-
let backtick_count = 0;
|
|
1043
|
-
while (i < this.#template.length && this.#template.charCodeAt(i) === BACKTICK) {
|
|
1044
|
-
backtick_count++;
|
|
1045
|
-
i++;
|
|
1046
|
-
}
|
|
1047
|
-
if (backtick_count < MIN_CODEBLOCK_BACKTICKS) {
|
|
1048
|
-
return false;
|
|
1049
|
-
}
|
|
1050
|
-
// Skip optional language hint (consume until space or newline)
|
|
1051
|
-
while (i < this.#template.length) {
|
|
1052
|
-
const char_code = this.#template.charCodeAt(i);
|
|
1053
|
-
if (char_code === SPACE || char_code === NEWLINE) {
|
|
1054
|
-
break;
|
|
1055
|
-
}
|
|
1056
|
-
i++;
|
|
1057
|
-
}
|
|
1058
|
-
// Skip any trailing spaces on opening fence line
|
|
1059
|
-
while (i < this.#template.length && this.#template.charCodeAt(i) === SPACE) {
|
|
1060
|
-
i++;
|
|
1061
|
-
}
|
|
1062
|
-
// Must have newline after opening fence (or be at EOF)
|
|
1063
|
-
if (i >= this.#template.length) {
|
|
1064
|
-
return false; // No newline, can't be a valid code block
|
|
1065
|
-
}
|
|
1066
|
-
if (this.#template.charCodeAt(i) !== NEWLINE) {
|
|
1067
|
-
return false;
|
|
1068
|
-
}
|
|
1069
|
-
i++; // consume the newline
|
|
1070
|
-
// Mark content start position (after opening fence newline)
|
|
1071
|
-
const content_start = i;
|
|
1072
|
-
// Now search for closing fence
|
|
1073
|
-
const closing_fence = '`'.repeat(backtick_count);
|
|
1074
|
-
while (i < this.#template.length) {
|
|
1075
|
-
// Check if we're at a potential closing fence (must be at start of line)
|
|
1076
|
-
if (this.#template.startsWith(closing_fence, i)) {
|
|
1077
|
-
// Verify it's at column 0 by checking previous character
|
|
1078
|
-
const prev_char = i > 0 ? this.#template.charCodeAt(i - 1) : NEWLINE;
|
|
1079
|
-
if (prev_char === NEWLINE || i === 0) {
|
|
1080
|
-
// Found closing fence - check for empty content first
|
|
1081
|
-
const content = this.#template.slice(content_start, i);
|
|
1082
|
-
const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
|
|
1083
|
-
if (final_content.length === 0) {
|
|
1084
|
-
return false; // Empty code block has no semantic meaning
|
|
1085
|
-
}
|
|
1086
|
-
// Now verify what comes after closing fence
|
|
1087
|
-
let j = i + backtick_count;
|
|
1088
|
-
// Skip trailing whitespace on closing fence line
|
|
1089
|
-
while (j < this.#template.length && this.#template.charCodeAt(j) === SPACE) {
|
|
1090
|
-
j++;
|
|
1091
|
-
}
|
|
1092
|
-
// Must have newline after closing fence (or be at EOF)
|
|
1093
|
-
if (j >= this.#template.length) {
|
|
1094
|
-
return true;
|
|
1095
|
-
}
|
|
1096
|
-
if (this.#template.charCodeAt(j) !== NEWLINE) {
|
|
1097
|
-
// closing fence has non-whitespace after it on same line - not a code block
|
|
1098
|
-
return false;
|
|
1099
|
-
}
|
|
1100
|
-
return true; // code block followed by newline or EOF
|
|
1101
|
-
}
|
|
1102
|
-
}
|
|
1103
|
-
i++;
|
|
1104
|
-
}
|
|
1105
|
-
// No closing fence found - not a valid code block
|
|
1106
|
-
return false;
|
|
1107
|
-
}
|
|
1108
|
-
/**
|
|
1109
|
-
* Parse code block: ```lang\ncode\n```
|
|
1110
|
-
* Assumes #match_code_block() already verified this is a code block.
|
|
1111
|
-
*/
|
|
1112
|
-
#parse_code_block() {
|
|
1113
|
-
const start = this.#index;
|
|
1114
|
-
// Count and consume opening backticks
|
|
1115
|
-
let backtick_count = 0;
|
|
1116
|
-
while (this.#index < this.#template.length &&
|
|
1117
|
-
this.#template.charCodeAt(this.#index) === BACKTICK) {
|
|
1118
|
-
backtick_count++;
|
|
1119
|
-
this.#index++;
|
|
1120
|
-
}
|
|
1121
|
-
// Parse optional language hint (consume until space or newline)
|
|
1122
|
-
let lang = null;
|
|
1123
|
-
const lang_start = this.#index;
|
|
1124
|
-
while (this.#index < this.#template.length) {
|
|
1125
|
-
const char_code = this.#template.charCodeAt(this.#index);
|
|
1126
|
-
if (char_code === SPACE || char_code === NEWLINE) {
|
|
1127
|
-
break;
|
|
1128
|
-
}
|
|
1129
|
-
this.#index++;
|
|
1130
|
-
}
|
|
1131
|
-
if (this.#index > lang_start) {
|
|
1132
|
-
lang = this.#template.slice(lang_start, this.#index);
|
|
1133
|
-
}
|
|
1134
|
-
// Skip any trailing spaces on opening fence line
|
|
1135
|
-
while (this.#index < this.#template.length &&
|
|
1136
|
-
this.#template.charCodeAt(this.#index) === SPACE) {
|
|
1137
|
-
this.#index++;
|
|
1138
|
-
}
|
|
1139
|
-
// Consume the newline after opening fence (first newline is consumed per spec)
|
|
1140
|
-
if (this.#index < this.#template.length && this.#template.charCodeAt(this.#index) === NEWLINE) {
|
|
1141
|
-
this.#index++;
|
|
1142
|
-
}
|
|
1143
|
-
// Collect content until closing fence
|
|
1144
|
-
const content_start = this.#index;
|
|
1145
|
-
const closing_fence = '`'.repeat(backtick_count);
|
|
1146
|
-
while (this.#index < this.#template.length) {
|
|
1147
|
-
// Check if we're at the closing fence (must be at start of line)
|
|
1148
|
-
if (this.#template.startsWith(closing_fence, this.#index)) {
|
|
1149
|
-
// Verify it's at column 0 by checking previous character
|
|
1150
|
-
const prev_char = this.#index > 0 ? this.#template.charCodeAt(this.#index - 1) : NEWLINE;
|
|
1151
|
-
if (prev_char === NEWLINE || this.#index === 0) {
|
|
1152
|
-
// Check if it's exactly the right number of backticks at line start
|
|
1153
|
-
let j = this.#index + backtick_count;
|
|
1154
|
-
// After closing fence, only whitespace and newline allowed
|
|
1155
|
-
while (j < this.#template.length && this.#template.charCodeAt(j) === SPACE) {
|
|
1156
|
-
j++;
|
|
1157
|
-
}
|
|
1158
|
-
if (j >= this.#template.length || this.#template.charCodeAt(j) === NEWLINE) {
|
|
1159
|
-
// Valid closing fence
|
|
1160
|
-
const content = this.#template.slice(content_start, this.#index);
|
|
1161
|
-
// Remove trailing newline if present (closing fence comes after a newline)
|
|
1162
|
-
const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
|
|
1163
|
-
// Consume closing fence
|
|
1164
|
-
this.#index += backtick_count;
|
|
1165
|
-
// Skip trailing whitespace on closing fence line
|
|
1166
|
-
while (this.#index < this.#template.length &&
|
|
1167
|
-
this.#template.charCodeAt(this.#index) === SPACE) {
|
|
1168
|
-
this.#index++;
|
|
1169
|
-
}
|
|
1170
|
-
// Don't consume the newline - let the main parse loop handle it
|
|
1171
|
-
return {
|
|
1172
|
-
type: 'Codeblock',
|
|
1173
|
-
lang,
|
|
1174
|
-
content: final_content,
|
|
1175
|
-
start,
|
|
1176
|
-
end: this.#index,
|
|
1177
|
-
};
|
|
1178
|
-
}
|
|
1179
|
-
}
|
|
1180
|
-
}
|
|
1181
|
-
this.#index++;
|
|
1182
|
-
}
|
|
1183
|
-
// Should not reach here if #match_code_block() validated correctly
|
|
1184
|
-
throw Error('Code block not properly closed');
|
|
1185
|
-
}
|
|
1186
|
-
}
|