@fuzdev/fuz_ui 0.204.0 → 0.205.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/dist/ApiModule.svelte +14 -1
  2. package/dist/ApiModule.svelte.d.ts.map +1 -1
  3. package/dist/DeclarationDetail.svelte +15 -4
  4. package/dist/DeclarationDetail.svelte.d.ts.map +1 -1
  5. package/package.json +12 -36
  6. package/dist/Mdz.svelte +0 -34
  7. package/dist/Mdz.svelte.d.ts +0 -11
  8. package/dist/Mdz.svelte.d.ts.map +0 -1
  9. package/dist/MdzNodeView.svelte +0 -107
  10. package/dist/MdzNodeView.svelte.d.ts +0 -9
  11. package/dist/MdzNodeView.svelte.d.ts.map +0 -1
  12. package/dist/MdzPrecompiled.svelte +0 -30
  13. package/dist/MdzPrecompiled.svelte.d.ts +0 -11
  14. package/dist/MdzPrecompiled.svelte.d.ts.map +0 -1
  15. package/dist/MdzRoot.svelte +0 -30
  16. package/dist/MdzRoot.svelte.d.ts +0 -12
  17. package/dist/MdzRoot.svelte.d.ts.map +0 -1
  18. package/dist/MdzStream.svelte +0 -32
  19. package/dist/MdzStream.svelte.d.ts +0 -12
  20. package/dist/MdzStream.svelte.d.ts.map +0 -1
  21. package/dist/MdzStreamNodeView.svelte +0 -106
  22. package/dist/MdzStreamNodeView.svelte.d.ts +0 -9
  23. package/dist/MdzStreamNodeView.svelte.d.ts.map +0 -1
  24. package/dist/mdz.d.ts +0 -112
  25. package/dist/mdz.d.ts.map +0 -1
  26. package/dist/mdz.js +0 -1186
  27. package/dist/mdz_components.d.ts +0 -63
  28. package/dist/mdz_components.d.ts.map +0 -1
  29. package/dist/mdz_components.js +0 -32
  30. package/dist/mdz_helpers.d.ts +0 -164
  31. package/dist/mdz_helpers.d.ts.map +0 -1
  32. package/dist/mdz_helpers.js +0 -424
  33. package/dist/mdz_lexer.d.ts +0 -93
  34. package/dist/mdz_lexer.d.ts.map +0 -1
  35. package/dist/mdz_lexer.js +0 -732
  36. package/dist/mdz_opcodes.d.ts +0 -174
  37. package/dist/mdz_opcodes.d.ts.map +0 -1
  38. package/dist/mdz_opcodes.js +0 -14
  39. package/dist/mdz_opcodes_to_nodes.d.ts +0 -20
  40. package/dist/mdz_opcodes_to_nodes.d.ts.map +0 -1
  41. package/dist/mdz_opcodes_to_nodes.js +0 -332
  42. package/dist/mdz_stream_parser.d.ts +0 -80
  43. package/dist/mdz_stream_parser.d.ts.map +0 -1
  44. package/dist/mdz_stream_parser.js +0 -354
  45. package/dist/mdz_stream_parser_block.d.ts +0 -76
  46. package/dist/mdz_stream_parser_block.d.ts.map +0 -1
  47. package/dist/mdz_stream_parser_block.js +0 -458
  48. package/dist/mdz_stream_parser_inline.d.ts +0 -52
  49. package/dist/mdz_stream_parser_inline.d.ts.map +0 -1
  50. package/dist/mdz_stream_parser_inline.js +0 -378
  51. package/dist/mdz_stream_parser_link.d.ts +0 -22
  52. package/dist/mdz_stream_parser_link.d.ts.map +0 -1
  53. package/dist/mdz_stream_parser_link.js +0 -215
  54. package/dist/mdz_stream_parser_state.d.ts +0 -187
  55. package/dist/mdz_stream_parser_state.d.ts.map +0 -1
  56. package/dist/mdz_stream_parser_state.js +0 -398
  57. package/dist/mdz_stream_parser_text.d.ts +0 -14
  58. package/dist/mdz_stream_parser_text.d.ts.map +0 -1
  59. package/dist/mdz_stream_parser_text.js +0 -50
  60. package/dist/mdz_stream_parser_url.d.ts +0 -60
  61. package/dist/mdz_stream_parser_url.d.ts.map +0 -1
  62. package/dist/mdz_stream_parser_url.js +0 -307
  63. package/dist/mdz_stream_state.svelte.d.ts +0 -44
  64. package/dist/mdz_stream_state.svelte.d.ts.map +0 -1
  65. package/dist/mdz_stream_state.svelte.js +0 -357
  66. package/dist/mdz_to_svelte.d.ts +0 -41
  67. package/dist/mdz_to_svelte.d.ts.map +0 -1
  68. package/dist/mdz_to_svelte.js +0 -100
  69. package/dist/mdz_token_parser.d.ts +0 -14
  70. package/dist/mdz_token_parser.d.ts.map +0 -1
  71. package/dist/mdz_token_parser.js +0 -344
  72. package/dist/svelte_preprocess_mdz.d.ts +0 -65
  73. package/dist/svelte_preprocess_mdz.d.ts.map +0 -1
  74. package/dist/svelte_preprocess_mdz.js +0 -529
  75. package/dist/tsdoc_mdz.d.ts +0 -45
  76. package/dist/tsdoc_mdz.d.ts.map +0 -1
  77. package/dist/tsdoc_mdz.js +0 -88
  78. package/src/lib/mdz.ts +0 -1532
  79. package/src/lib/mdz_components.ts +0 -64
  80. package/src/lib/mdz_helpers.ts +0 -449
  81. package/src/lib/mdz_lexer.ts +0 -1012
  82. package/src/lib/mdz_opcodes.ts +0 -205
  83. package/src/lib/mdz_opcodes_to_nodes.ts +0 -375
  84. package/src/lib/mdz_stream_parser.ts +0 -415
  85. package/src/lib/mdz_stream_parser_block.ts +0 -532
  86. package/src/lib/mdz_stream_parser_inline.ts +0 -414
  87. package/src/lib/mdz_stream_parser_link.ts +0 -271
  88. package/src/lib/mdz_stream_parser_state.ts +0 -539
  89. package/src/lib/mdz_stream_parser_text.ts +0 -77
  90. package/src/lib/mdz_stream_parser_url.ts +0 -365
  91. package/src/lib/mdz_stream_state.svelte.ts +0 -387
  92. package/src/lib/mdz_to_svelte.ts +0 -141
  93. package/src/lib/mdz_token_parser.ts +0 -434
  94. package/src/lib/svelte_preprocess_mdz.ts +0 -742
  95. package/src/lib/tsdoc_mdz.ts +0 -97
package/dist/mdz_lexer.js DELETED
@@ -1,732 +0,0 @@
1
- /**
2
- * mdz lexer — tokenizes input into a flat `MdzToken[]` stream.
3
- *
4
- * Phase 1 of the two-phase lexer+parser alternative to the single-pass parser
5
- * in `mdz.ts`. Phase 2 is in `mdz_token_parser.ts`.
6
- *
7
- * @module
8
- */
9
- import { mdz_is_url, mdz_is_safe_reference, is_letter, is_tag_name_char, is_word_char, is_valid_path_char, trim_trailing_punctuation, BACKTICK, ASTERISK, UNDERSCORE, TILDE, NEWLINE, HYPHEN, HASH, SPACE, TAB, LEFT_ANGLE, RIGHT_ANGLE, SLASH, PERIOD, LEFT_BRACKET, LEFT_PAREN, RIGHT_PAREN, RIGHT_BRACKET, A_UPPER, Z_UPPER, HR_HYPHEN_COUNT, MIN_CODEBLOCK_BACKTICKS, MAX_HEADING_LEVEL, match_url_prefix_case_insensitive, is_at_absolute_path, is_at_relative_path, } from './mdz_helpers.js';
10
- //
11
- // Lexer
12
- //
13
- export class MdzLexer {
14
- #text;
15
- #index = 0;
16
- #tokens = [];
17
- #max_search_index = Number.MAX_SAFE_INTEGER;
18
- constructor(text) {
19
- this.#text = text;
20
- }
21
- tokenize() {
22
- // Skip leading newlines
23
- this.#skip_newlines();
24
- while (this.#index < this.#text.length) {
25
- // Check for block elements at column 0
26
- if (this.#is_at_column_0()) {
27
- if (this.#tokenize_heading())
28
- continue;
29
- if (this.#tokenize_hr())
30
- continue;
31
- if (this.#tokenize_codeblock())
32
- continue;
33
- }
34
- // Check for paragraph break
35
- if (this.#is_at_paragraph_break()) {
36
- const start = this.#index;
37
- this.#skip_newlines();
38
- this.#tokens.push({ type: 'paragraph_break', start, end: this.#index });
39
- continue;
40
- }
41
- // Inline tokenization
42
- this.#tokenize_inline();
43
- }
44
- return this.#tokens;
45
- }
46
- #is_at_column_0() {
47
- return this.#index === 0 || this.#text.charCodeAt(this.#index - 1) === NEWLINE;
48
- }
49
- #is_at_paragraph_break() {
50
- return (this.#index + 1 < this.#text.length &&
51
- this.#text.charCodeAt(this.#index) === NEWLINE &&
52
- this.#text.charCodeAt(this.#index + 1) === NEWLINE);
53
- }
54
- #match(str) {
55
- return this.#text.startsWith(str, this.#index);
56
- }
57
- #skip_newlines() {
58
- while (this.#index < this.#text.length && this.#text.charCodeAt(this.#index) === NEWLINE) {
59
- this.#index++;
60
- }
61
- }
62
- // -- Block tokenizers --
63
- #tokenize_heading() {
64
- let i = this.#index;
65
- // Count hashes (must be 1-6)
66
- let hash_count = 0;
67
- while (i < this.#text.length &&
68
- this.#text.charCodeAt(i) === HASH &&
69
- hash_count <= MAX_HEADING_LEVEL) {
70
- hash_count++;
71
- i++;
72
- }
73
- if (hash_count === 0 || hash_count > MAX_HEADING_LEVEL)
74
- return false;
75
- // Must have space after hashes
76
- if (i >= this.#text.length || this.#text.charCodeAt(i) !== SPACE)
77
- return false;
78
- i++; // consume the space
79
- // Must have at least one non-whitespace character after the space
80
- let has_content = false;
81
- for (let j = i; j < this.#text.length; j++) {
82
- const char_code = this.#text.charCodeAt(j);
83
- if (char_code === NEWLINE)
84
- break;
85
- if (char_code !== SPACE && char_code !== TAB) {
86
- has_content = true;
87
- break;
88
- }
89
- }
90
- if (!has_content)
91
- return false;
92
- const start = this.#index;
93
- // Advance past "## " (hashes + space)
94
- this.#index = start + hash_count + 1;
95
- this.#tokens.push({
96
- type: 'heading_start',
97
- level: hash_count,
98
- start,
99
- end: this.#index, // end of "## " prefix
100
- });
101
- // Find end-of-line to bound nested tokenizers (prevents tag scanner from scanning past heading)
102
- let eol = this.#text.indexOf('\n', this.#index);
103
- if (eol === -1)
104
- eol = this.#text.length;
105
- const saved_max = this.#max_search_index;
106
- this.#max_search_index = eol;
107
- // Tokenize inline content until newline or EOF
108
- // tokenize_text may consume a newline as part of block-element lookahead,
109
- // so we check emitted text tokens for embedded newlines and trim.
110
- while (this.#index < eol) {
111
- const token_count_before = this.#tokens.length;
112
- this.#tokenize_inline();
113
- // Check if the last emitted token is text containing a newline
114
- if (this.#tokens.length > token_count_before) {
115
- const last_token = this.#tokens[this.#tokens.length - 1];
116
- if (last_token.type === 'text') {
117
- const newline_idx = last_token.content.indexOf('\n');
118
- if (newline_idx !== -1) {
119
- const trimmed = last_token.content.slice(0, newline_idx);
120
- if (trimmed) {
121
- last_token.content = trimmed;
122
- last_token.end = last_token.start + trimmed.length;
123
- }
124
- else {
125
- this.#tokens.pop();
126
- }
127
- this.#index = last_token.start + newline_idx;
128
- break;
129
- }
130
- }
131
- }
132
- }
133
- this.#max_search_index = saved_max;
134
- // Emit heading_end marker so the token parser knows where heading content stops
135
- this.#tokens.push({ type: 'heading_end', start: this.#index, end: this.#index });
136
- // Skip newlines after heading (like original parser's main loop)
137
- this.#skip_newlines();
138
- return true;
139
- }
140
- #tokenize_hr() {
141
- let i = this.#index;
142
- // Must have exactly three hyphens
143
- if (i + HR_HYPHEN_COUNT > this.#text.length ||
144
- this.#text.charCodeAt(i) !== HYPHEN ||
145
- this.#text.charCodeAt(i + 1) !== HYPHEN ||
146
- this.#text.charCodeAt(i + 2) !== HYPHEN) {
147
- return false;
148
- }
149
- i += HR_HYPHEN_COUNT;
150
- // After the three hyphens, only whitespace and newline (or EOF) allowed
151
- while (i < this.#text.length) {
152
- const char_code = this.#text.charCodeAt(i);
153
- if (char_code === NEWLINE)
154
- break;
155
- if (char_code !== SPACE)
156
- return false;
157
- i++;
158
- }
159
- const start = this.#index;
160
- this.#index = i; // consume up to but not including newline
161
- this.#tokens.push({ type: 'hr', start, end: this.#index });
162
- // Skip newlines after HR
163
- this.#skip_newlines();
164
- return true;
165
- }
166
- #tokenize_codeblock() {
167
- let i = this.#index;
168
- // Count backticks
169
- let backtick_count = 0;
170
- while (i < this.#text.length && this.#text.charCodeAt(i) === BACKTICK) {
171
- backtick_count++;
172
- i++;
173
- }
174
- if (backtick_count < MIN_CODEBLOCK_BACKTICKS)
175
- return false;
176
- // Skip optional language hint
177
- const lang_start = i;
178
- while (i < this.#text.length) {
179
- const char_code = this.#text.charCodeAt(i);
180
- if (char_code === SPACE || char_code === NEWLINE)
181
- break;
182
- i++;
183
- }
184
- const lang = i > lang_start ? this.#text.slice(lang_start, i) : null;
185
- // Skip trailing spaces on opening fence line
186
- while (i < this.#text.length && this.#text.charCodeAt(i) === SPACE) {
187
- i++;
188
- }
189
- // Must have newline after opening fence
190
- if (i >= this.#text.length || this.#text.charCodeAt(i) !== NEWLINE)
191
- return false;
192
- i++; // consume the newline
193
- const content_start = i;
194
- // Search for closing fence
195
- const closing_fence = '`'.repeat(backtick_count);
196
- while (i < this.#text.length) {
197
- if (this.#text.startsWith(closing_fence, i)) {
198
- const prev_char = i > 0 ? this.#text.charCodeAt(i - 1) : NEWLINE;
199
- if (prev_char === NEWLINE || i === 0) {
200
- const content = this.#text.slice(content_start, i);
201
- const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
202
- if (final_content.length === 0)
203
- return false; // Empty code block
204
- // Verify closing fence line
205
- let j = i + backtick_count;
206
- while (j < this.#text.length && this.#text.charCodeAt(j) === SPACE) {
207
- j++;
208
- }
209
- if (j < this.#text.length && this.#text.charCodeAt(j) !== NEWLINE) {
210
- return false;
211
- }
212
- const start = this.#index;
213
- this.#index = j; // advance past closing fence and trailing spaces
214
- this.#tokens.push({
215
- type: 'codeblock',
216
- lang,
217
- content: final_content,
218
- start,
219
- end: this.#index,
220
- });
221
- // Skip newlines after codeblock
222
- this.#skip_newlines();
223
- return true;
224
- }
225
- }
226
- i++;
227
- }
228
- return false; // No closing fence found
229
- }
230
- // -- Inline tokenizers --
231
- #tokenize_inline() {
232
- const char_code = this.#text.charCodeAt(this.#index);
233
- switch (char_code) {
234
- case BACKTICK:
235
- this.#tokenize_code();
236
- return;
237
- case ASTERISK:
238
- this.#tokenize_bold();
239
- return;
240
- case UNDERSCORE:
241
- this.#tokenize_single_delimiter('_', 'italic');
242
- return;
243
- case TILDE:
244
- this.#tokenize_single_delimiter('~', 'strikethrough');
245
- return;
246
- case LEFT_BRACKET:
247
- this.#tokenize_markdown_link();
248
- return;
249
- case LEFT_ANGLE:
250
- this.#tokenize_tag();
251
- return;
252
- default:
253
- this.#tokenize_text();
254
- return;
255
- }
256
- }
257
- #tokenize_code() {
258
- const start = this.#index;
259
- this.#index++; // consume `
260
- // Find closing backtick, stop at newline or search boundary
261
- let content_end = -1;
262
- const search_limit = Math.min(this.#max_search_index, this.#text.length);
263
- for (let i = this.#index; i < search_limit; i++) {
264
- const char_code = this.#text.charCodeAt(i);
265
- if (char_code === BACKTICK) {
266
- content_end = i;
267
- break;
268
- }
269
- if (char_code === NEWLINE)
270
- break;
271
- }
272
- if (content_end === -1) {
273
- // Unclosed backtick
274
- this.#emit_text('`', start);
275
- return;
276
- }
277
- const content = this.#text.slice(this.#index, content_end);
278
- // Empty inline code
279
- if (content.length === 0) {
280
- this.#index = content_end + 1;
281
- this.#emit_text('``', start);
282
- return;
283
- }
284
- this.#index = content_end + 1;
285
- this.#tokens.push({ type: 'code', content, start, end: this.#index });
286
- }
287
- #tokenize_bold() {
288
- const start = this.#index;
289
- // Check for **
290
- if (!this.#match('**')) {
291
- // Single asterisk - text
292
- this.#index++;
293
- this.#emit_text('*', start);
294
- return;
295
- }
296
- // Find closing ** within current search boundary
297
- const search_end = Math.min(this.#max_search_index, this.#text.length);
298
- let close_index = this.#text.indexOf('**', start + 2);
299
- if (close_index !== -1 && close_index >= search_end) {
300
- close_index = -1;
301
- }
302
- if (close_index === -1) {
303
- // Unclosed **
304
- this.#index += 2;
305
- this.#emit_text('**', start);
306
- return;
307
- }
308
- // Check for paragraph break between open and close
309
- if (this.#has_paragraph_break_between(start + 2, close_index)) {
310
- this.#index += 2;
311
- this.#emit_text('**', start);
312
- return;
313
- }
314
- // Emit bold_open, tokenize children up to close, emit bold_close
315
- this.#index += 2;
316
- const open_token_index = this.#tokens.length;
317
- this.#tokens.push({ type: 'bold_open', start, end: this.#index });
318
- // Set search boundary for nested parsers
319
- const saved_max = this.#max_search_index;
320
- this.#max_search_index = close_index;
321
- // Tokenize children inline content up to close_index
322
- while (this.#index < close_index) {
323
- if (this.#is_at_paragraph_break())
324
- break;
325
- this.#tokenize_inline();
326
- }
327
- this.#max_search_index = saved_max;
328
- if (this.#index === close_index && this.#match('**')) {
329
- // Check if empty
330
- const has_children = open_token_index < this.#tokens.length - 1;
331
- if (!has_children) {
332
- // Empty bold - convert to text
333
- this.#tokens.splice(open_token_index, 1);
334
- this.#index = start;
335
- this.#index += 4;
336
- this.#emit_text('****', start);
337
- return;
338
- }
339
- this.#tokens.push({ type: 'bold_close', start: close_index, end: close_index + 2 });
340
- this.#index = close_index + 2;
341
- }
342
- else {
343
- // Didn't reach closing - convert opening to text
344
- this.#tokens[open_token_index] = { type: 'text', content: '**', start, end: start + 2 };
345
- }
346
- }
347
- #tokenize_single_delimiter(delimiter, kind) {
348
- const start = this.#index;
349
- // Check opening word boundary
350
- if (!this.#is_at_word_boundary(this.#index, true, false)) {
351
- this.#index++;
352
- this.#emit_text(delimiter, start);
353
- return;
354
- }
355
- // Find closing delimiter within search boundary
356
- const search_end = Math.min(this.#max_search_index, this.#text.length);
357
- let close_index = this.#text.indexOf(delimiter, start + 1);
358
- if (close_index !== -1 && close_index >= search_end) {
359
- close_index = -1;
360
- }
361
- if (close_index === -1) {
362
- this.#index++;
363
- this.#emit_text(delimiter, start);
364
- return;
365
- }
366
- // Check closing word boundary
367
- if (!this.#is_at_word_boundary(close_index + 1, false, true)) {
368
- this.#index++;
369
- this.#emit_text(delimiter, start);
370
- return;
371
- }
372
- // Check for paragraph break between
373
- if (this.#has_paragraph_break_between(start + 1, close_index)) {
374
- this.#index++;
375
- this.#emit_text(delimiter, start);
376
- return;
377
- }
378
- // Emit open token
379
- this.#index++;
380
- const open_type = kind === 'italic' ? 'italic_open' : 'strikethrough_open';
381
- const close_type = kind === 'italic' ? 'italic_close' : 'strikethrough_close';
382
- const open_token_index = this.#tokens.length;
383
- this.#tokens.push({ type: open_type, start, end: this.#index });
384
- // Set search boundary for nested parsers
385
- const saved_max = this.#max_search_index;
386
- this.#max_search_index = close_index;
387
- // Tokenize children up to close_index
388
- while (this.#index < close_index) {
389
- if (this.#is_at_paragraph_break())
390
- break;
391
- this.#tokenize_inline();
392
- }
393
- this.#max_search_index = saved_max;
394
- if (this.#index === close_index && this.#match(delimiter)) {
395
- // Check if empty
396
- const has_children = open_token_index < this.#tokens.length - 1;
397
- if (!has_children) {
398
- // Empty - convert to text
399
- this.#tokens.splice(open_token_index, 1);
400
- this.#index = start;
401
- this.#index += 2;
402
- this.#emit_text(delimiter + delimiter, start);
403
- return;
404
- }
405
- this.#tokens.push({
406
- type: close_type,
407
- start: close_index,
408
- end: close_index + 1,
409
- });
410
- this.#index = close_index + 1;
411
- }
412
- else {
413
- // Convert opening to text
414
- this.#tokens[open_token_index] = {
415
- type: 'text',
416
- content: delimiter,
417
- start,
418
- end: start + 1,
419
- };
420
- }
421
- }
422
- #tokenize_markdown_link() {
423
- const start = this.#index;
424
- // Consume [
425
- if (this.#text.charCodeAt(this.#index) !== LEFT_BRACKET) {
426
- this.#index++;
427
- this.#emit_text(this.#text[start], start);
428
- return;
429
- }
430
- this.#index++;
431
- // Emit link_text_open
432
- this.#tokens.push({ type: 'link_text_open', start, end: this.#index });
433
- // Tokenize children until ]
434
- while (this.#index < this.#text.length) {
435
- if (this.#text.charCodeAt(this.#index) === RIGHT_BRACKET)
436
- break;
437
- if (this.#is_at_paragraph_break())
438
- break;
439
- // Stop at ] and ) as delimiters
440
- if (this.#text.charCodeAt(this.#index) === RIGHT_PAREN)
441
- break;
442
- this.#tokenize_inline();
443
- }
444
- // Check for ]
445
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== RIGHT_BRACKET) {
446
- // Revert - remove link_text_open and all children tokens added
447
- this.#revert_tokens_from_link_open(start);
448
- return;
449
- }
450
- const bracket_close_start = this.#index;
451
- this.#index++; // consume ]
452
- this.#tokens.push({
453
- type: 'link_text_close',
454
- start: bracket_close_start,
455
- end: this.#index,
456
- });
457
- // Check for (
458
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== LEFT_PAREN) {
459
- this.#revert_tokens_from_link_open(start);
460
- return;
461
- }
462
- this.#index++; // consume (
463
- // Find closing )
464
- const close_paren = this.#text.indexOf(')', this.#index);
465
- if (close_paren === -1) {
466
- this.#revert_tokens_from_link_open(start);
467
- return;
468
- }
469
- const reference = this.#text.slice(this.#index, close_paren);
470
- // Validate reference
471
- if (!reference.trim()) {
472
- this.#revert_tokens_from_link_open(start);
473
- return;
474
- }
475
- // Validate all characters
476
- for (let i = 0; i < reference.length; i++) {
477
- const char_code = reference.charCodeAt(i);
478
- if (!is_valid_path_char(char_code)) {
479
- this.#revert_tokens_from_link_open(start);
480
- return;
481
- }
482
- }
483
- // Reject unsafe protocols (javascript:, data:, etc.) — `is_valid_path_char`
484
- // permits `:` so we need an explicit filter here.
485
- if (!mdz_is_safe_reference(reference)) {
486
- this.#revert_tokens_from_link_open(start);
487
- return;
488
- }
489
- this.#index = close_paren + 1;
490
- const link_type = mdz_is_url(reference) ? 'external' : 'internal';
491
- this.#tokens.push({
492
- type: 'link_ref',
493
- reference,
494
- link_type,
495
- start: bracket_close_start + 1, // after ]
496
- end: this.#index,
497
- });
498
- }
499
- #revert_tokens_from_link_open(start) {
500
- // Find and remove link_text_open and all tokens after it
501
- let open_idx = -1;
502
- for (let i = this.#tokens.length - 1; i >= 0; i--) {
503
- if (this.#tokens[i].type === 'link_text_open' && this.#tokens[i].start === start) {
504
- open_idx = i;
505
- break;
506
- }
507
- }
508
- if (open_idx !== -1) {
509
- this.#tokens.splice(open_idx);
510
- }
511
- this.#index = start + 1;
512
- this.#emit_text('[', start);
513
- }
514
- #tokenize_tag() {
515
- const start = this.#index;
516
- this.#index++; // consume <
517
- // Tag name must start with a letter
518
- if (this.#index >= this.#text.length || !is_letter(this.#text.charCodeAt(this.#index))) {
519
- this.#emit_text('<', start);
520
- return;
521
- }
522
- // Collect tag name
523
- const tag_name_start = this.#index;
524
- while (this.#index < this.#text.length &&
525
- is_tag_name_char(this.#text.charCodeAt(this.#index))) {
526
- this.#index++;
527
- }
528
- const tag_name = this.#text.slice(tag_name_start, this.#index);
529
- if (tag_name.length === 0) {
530
- this.#emit_text('<', start);
531
- return;
532
- }
533
- const first_char_code = tag_name.charCodeAt(0);
534
- const is_component = first_char_code >= A_UPPER && first_char_code <= Z_UPPER;
535
- // Skip whitespace
536
- while (this.#index < this.#text.length && this.#text.charCodeAt(this.#index) === SPACE) {
537
- this.#index++;
538
- }
539
- // Check for self-closing />
540
- if (this.#index + 1 < this.#text.length &&
541
- this.#text.charCodeAt(this.#index) === SLASH &&
542
- this.#text.charCodeAt(this.#index + 1) === RIGHT_ANGLE) {
543
- this.#index += 2;
544
- this.#tokens.push({
545
- type: 'tag_self_close',
546
- name: tag_name,
547
- is_component,
548
- start,
549
- end: this.#index,
550
- });
551
- return;
552
- }
553
- // Check for >
554
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== RIGHT_ANGLE) {
555
- this.#index = start + 1;
556
- this.#emit_text('<', start);
557
- return;
558
- }
559
- this.#index++; // consume >
560
- // Check for closing tag existence before committing —
561
- // must exist within search boundary and before any paragraph break
562
- const closing_tag = `</${tag_name}>`;
563
- const search_limit = Math.min(this.#max_search_index, this.#text.length);
564
- const closing_tag_pos = this.#text.indexOf(closing_tag, this.#index);
565
- if (closing_tag_pos === -1 || closing_tag_pos >= search_limit) {
566
- this.#index = start + 1;
567
- this.#emit_text('<', start);
568
- return;
569
- }
570
- if (this.#has_paragraph_break_between(this.#index, closing_tag_pos)) {
571
- this.#index = start + 1;
572
- this.#emit_text('<', start);
573
- return;
574
- }
575
- // Emit tag_open
576
- this.#tokens.push({ type: 'tag_open', name: tag_name, is_component, start, end: this.#index });
577
- // Tokenize children until closing tag
578
- while (this.#index < this.#text.length) {
579
- if (this.#match(closing_tag)) {
580
- const close_start = this.#index;
581
- this.#index += closing_tag.length;
582
- this.#tokens.push({
583
- type: 'tag_close',
584
- name: tag_name,
585
- start: close_start,
586
- end: this.#index,
587
- });
588
- return;
589
- }
590
- this.#tokenize_inline();
591
- }
592
- // Shouldn't reach here since we verified closing tag exists
593
- // But if we do, the tag_open token stays and parser will handle it
594
- }
595
- #tokenize_text() {
596
- const start = this.#index;
597
- // Check for URL or internal path at current position
598
- if (this.#is_at_url()) {
599
- this.#tokenize_auto_link_url();
600
- return;
601
- }
602
- if (is_at_absolute_path(this.#text, this.#index)) {
603
- this.#tokenize_auto_link_internal();
604
- return;
605
- }
606
- if (is_at_relative_path(this.#text, this.#index)) {
607
- this.#tokenize_auto_link_internal();
608
- return;
609
- }
610
- while (this.#index < this.#text.length) {
611
- const char_code = this.#text.charCodeAt(this.#index);
612
- // Stop at special characters
613
- if (char_code === BACKTICK ||
614
- char_code === ASTERISK ||
615
- char_code === UNDERSCORE ||
616
- char_code === TILDE ||
617
- char_code === LEFT_BRACKET ||
618
- char_code === RIGHT_BRACKET ||
619
- char_code === RIGHT_PAREN ||
620
- char_code === LEFT_ANGLE) {
621
- break;
622
- }
623
- // Check for paragraph break
624
- if (this.#is_at_paragraph_break())
625
- break;
626
- // When next line could start a block element, consume the newline and stop
627
- if (char_code === NEWLINE) {
628
- const next_i = this.#index + 1;
629
- if (next_i < this.#text.length) {
630
- const next_char = this.#text.charCodeAt(next_i);
631
- if (next_char === HASH || next_char === HYPHEN || next_char === BACKTICK) {
632
- this.#index++; // consume the newline
633
- break;
634
- }
635
- }
636
- }
637
- // Check for URL or internal path mid-text (char code guard avoids startsWith on every char)
638
- if (((char_code === 104 /* h */ || char_code === 72) /* H */ && this.#is_at_url()) ||
639
- (char_code === SLASH && is_at_absolute_path(this.#text, this.#index)) ||
640
- (char_code === PERIOD && is_at_relative_path(this.#text, this.#index))) {
641
- break;
642
- }
643
- this.#index++;
644
- }
645
- // Ensure we always consume at least one character
646
- if (this.#index === start && this.#index < this.#text.length) {
647
- this.#index++;
648
- }
649
- const content = this.#text.slice(start, this.#index);
650
- this.#emit_text(content, start);
651
- }
652
- // -- Auto-link tokenizers --
653
- #tokenize_auto_link_url() {
654
- const start = this.#index;
655
- // Consume protocol (case-insensitive match; original casing preserved in reference)
656
- this.#index += match_url_prefix_case_insensitive(this.#text, this.#index);
657
- // Collect URL characters
658
- while (this.#index < this.#text.length) {
659
- const char_code = this.#text.charCodeAt(this.#index);
660
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
661
- break;
662
- }
663
- this.#index++;
664
- }
665
- let reference = this.#text.slice(start, this.#index);
666
- reference = trim_trailing_punctuation(reference);
667
- this.#index = start + reference.length;
668
- this.#tokens.push({
669
- type: 'autolink',
670
- reference,
671
- link_type: 'external',
672
- start,
673
- end: this.#index,
674
- });
675
- }
676
- #tokenize_auto_link_internal() {
677
- const start = this.#index;
678
- // Collect path characters
679
- while (this.#index < this.#text.length) {
680
- const char_code = this.#text.charCodeAt(this.#index);
681
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
682
- break;
683
- }
684
- this.#index++;
685
- }
686
- let reference = this.#text.slice(start, this.#index);
687
- reference = trim_trailing_punctuation(reference);
688
- this.#index = start + reference.length;
689
- this.#tokens.push({
690
- type: 'autolink',
691
- reference,
692
- link_type: 'internal',
693
- start,
694
- end: this.#index,
695
- });
696
- }
697
- // -- Helper methods --
698
- #emit_text(content, start) {
699
- this.#tokens.push({ type: 'text', content, start, end: start + content.length });
700
- }
701
- #has_paragraph_break_between(from, to) {
702
- for (let i = from; i < to - 1; i++) {
703
- if (this.#text.charCodeAt(i) === NEWLINE && this.#text.charCodeAt(i + 1) === NEWLINE) {
704
- return true;
705
- }
706
- }
707
- return false;
708
- }
709
- #is_at_url() {
710
- // Scheme matching is case-insensitive (RFC 3986).
711
- const prefix_len = match_url_prefix_case_insensitive(this.#text, this.#index);
712
- if (prefix_len === 0)
713
- return false;
714
- if (this.#index + prefix_len >= this.#text.length)
715
- return false;
716
- const next_char = this.#text.charCodeAt(this.#index + prefix_len);
717
- return next_char !== SPACE && next_char !== NEWLINE;
718
- }
719
- #is_at_word_boundary(index, check_before, check_after) {
720
- if (check_before && index > 0) {
721
- const prev = this.#text.charCodeAt(index - 1);
722
- if (is_word_char(prev))
723
- return false;
724
- }
725
- if (check_after && index < this.#text.length) {
726
- const next = this.#text.charCodeAt(index);
727
- if (is_word_char(next))
728
- return false;
729
- }
730
- return true;
731
- }
732
- }