@fuzdev/fuz_ui 0.204.0 → 0.205.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/dist/ApiModule.svelte +14 -1
  2. package/dist/ApiModule.svelte.d.ts.map +1 -1
  3. package/dist/DeclarationDetail.svelte +15 -4
  4. package/dist/DeclarationDetail.svelte.d.ts.map +1 -1
  5. package/package.json +12 -36
  6. package/dist/Mdz.svelte +0 -34
  7. package/dist/Mdz.svelte.d.ts +0 -11
  8. package/dist/Mdz.svelte.d.ts.map +0 -1
  9. package/dist/MdzNodeView.svelte +0 -107
  10. package/dist/MdzNodeView.svelte.d.ts +0 -9
  11. package/dist/MdzNodeView.svelte.d.ts.map +0 -1
  12. package/dist/MdzPrecompiled.svelte +0 -30
  13. package/dist/MdzPrecompiled.svelte.d.ts +0 -11
  14. package/dist/MdzPrecompiled.svelte.d.ts.map +0 -1
  15. package/dist/MdzRoot.svelte +0 -30
  16. package/dist/MdzRoot.svelte.d.ts +0 -12
  17. package/dist/MdzRoot.svelte.d.ts.map +0 -1
  18. package/dist/MdzStream.svelte +0 -32
  19. package/dist/MdzStream.svelte.d.ts +0 -12
  20. package/dist/MdzStream.svelte.d.ts.map +0 -1
  21. package/dist/MdzStreamNodeView.svelte +0 -106
  22. package/dist/MdzStreamNodeView.svelte.d.ts +0 -9
  23. package/dist/MdzStreamNodeView.svelte.d.ts.map +0 -1
  24. package/dist/mdz.d.ts +0 -112
  25. package/dist/mdz.d.ts.map +0 -1
  26. package/dist/mdz.js +0 -1186
  27. package/dist/mdz_components.d.ts +0 -63
  28. package/dist/mdz_components.d.ts.map +0 -1
  29. package/dist/mdz_components.js +0 -32
  30. package/dist/mdz_helpers.d.ts +0 -164
  31. package/dist/mdz_helpers.d.ts.map +0 -1
  32. package/dist/mdz_helpers.js +0 -424
  33. package/dist/mdz_lexer.d.ts +0 -93
  34. package/dist/mdz_lexer.d.ts.map +0 -1
  35. package/dist/mdz_lexer.js +0 -732
  36. package/dist/mdz_opcodes.d.ts +0 -174
  37. package/dist/mdz_opcodes.d.ts.map +0 -1
  38. package/dist/mdz_opcodes.js +0 -14
  39. package/dist/mdz_opcodes_to_nodes.d.ts +0 -20
  40. package/dist/mdz_opcodes_to_nodes.d.ts.map +0 -1
  41. package/dist/mdz_opcodes_to_nodes.js +0 -332
  42. package/dist/mdz_stream_parser.d.ts +0 -80
  43. package/dist/mdz_stream_parser.d.ts.map +0 -1
  44. package/dist/mdz_stream_parser.js +0 -354
  45. package/dist/mdz_stream_parser_block.d.ts +0 -76
  46. package/dist/mdz_stream_parser_block.d.ts.map +0 -1
  47. package/dist/mdz_stream_parser_block.js +0 -458
  48. package/dist/mdz_stream_parser_inline.d.ts +0 -52
  49. package/dist/mdz_stream_parser_inline.d.ts.map +0 -1
  50. package/dist/mdz_stream_parser_inline.js +0 -378
  51. package/dist/mdz_stream_parser_link.d.ts +0 -22
  52. package/dist/mdz_stream_parser_link.d.ts.map +0 -1
  53. package/dist/mdz_stream_parser_link.js +0 -215
  54. package/dist/mdz_stream_parser_state.d.ts +0 -187
  55. package/dist/mdz_stream_parser_state.d.ts.map +0 -1
  56. package/dist/mdz_stream_parser_state.js +0 -398
  57. package/dist/mdz_stream_parser_text.d.ts +0 -14
  58. package/dist/mdz_stream_parser_text.d.ts.map +0 -1
  59. package/dist/mdz_stream_parser_text.js +0 -50
  60. package/dist/mdz_stream_parser_url.d.ts +0 -60
  61. package/dist/mdz_stream_parser_url.d.ts.map +0 -1
  62. package/dist/mdz_stream_parser_url.js +0 -307
  63. package/dist/mdz_stream_state.svelte.d.ts +0 -44
  64. package/dist/mdz_stream_state.svelte.d.ts.map +0 -1
  65. package/dist/mdz_stream_state.svelte.js +0 -357
  66. package/dist/mdz_to_svelte.d.ts +0 -41
  67. package/dist/mdz_to_svelte.d.ts.map +0 -1
  68. package/dist/mdz_to_svelte.js +0 -100
  69. package/dist/mdz_token_parser.d.ts +0 -14
  70. package/dist/mdz_token_parser.d.ts.map +0 -1
  71. package/dist/mdz_token_parser.js +0 -344
  72. package/dist/svelte_preprocess_mdz.d.ts +0 -65
  73. package/dist/svelte_preprocess_mdz.d.ts.map +0 -1
  74. package/dist/svelte_preprocess_mdz.js +0 -529
  75. package/dist/tsdoc_mdz.d.ts +0 -45
  76. package/dist/tsdoc_mdz.d.ts.map +0 -1
  77. package/dist/tsdoc_mdz.js +0 -88
  78. package/src/lib/mdz.ts +0 -1532
  79. package/src/lib/mdz_components.ts +0 -64
  80. package/src/lib/mdz_helpers.ts +0 -449
  81. package/src/lib/mdz_lexer.ts +0 -1012
  82. package/src/lib/mdz_opcodes.ts +0 -205
  83. package/src/lib/mdz_opcodes_to_nodes.ts +0 -375
  84. package/src/lib/mdz_stream_parser.ts +0 -415
  85. package/src/lib/mdz_stream_parser_block.ts +0 -532
  86. package/src/lib/mdz_stream_parser_inline.ts +0 -414
  87. package/src/lib/mdz_stream_parser_link.ts +0 -271
  88. package/src/lib/mdz_stream_parser_state.ts +0 -539
  89. package/src/lib/mdz_stream_parser_text.ts +0 -77
  90. package/src/lib/mdz_stream_parser_url.ts +0 -365
  91. package/src/lib/mdz_stream_state.svelte.ts +0 -387
  92. package/src/lib/mdz_to_svelte.ts +0 -141
  93. package/src/lib/mdz_token_parser.ts +0 -434
  94. package/src/lib/svelte_preprocess_mdz.ts +0 -742
  95. package/src/lib/tsdoc_mdz.ts +0 -97
@@ -1,1012 +0,0 @@
1
- /**
2
- * mdz lexer — tokenizes input into a flat `MdzToken[]` stream.
3
- *
4
- * Phase 1 of the two-phase lexer+parser alternative to the single-pass parser
5
- * in `mdz.ts`. Phase 2 is in `mdz_token_parser.ts`.
6
- *
7
- * @module
8
- */
9
-
10
- import {
11
- mdz_is_url,
12
- mdz_is_safe_reference,
13
- is_letter,
14
- is_tag_name_char,
15
- is_word_char,
16
- is_valid_path_char,
17
- trim_trailing_punctuation,
18
- BACKTICK,
19
- ASTERISK,
20
- UNDERSCORE,
21
- TILDE,
22
- NEWLINE,
23
- HYPHEN,
24
- HASH,
25
- SPACE,
26
- TAB,
27
- LEFT_ANGLE,
28
- RIGHT_ANGLE,
29
- SLASH,
30
- PERIOD,
31
- LEFT_BRACKET,
32
- LEFT_PAREN,
33
- RIGHT_PAREN,
34
- RIGHT_BRACKET,
35
- A_UPPER,
36
- Z_UPPER,
37
- HR_HYPHEN_COUNT,
38
- MIN_CODEBLOCK_BACKTICKS,
39
- MAX_HEADING_LEVEL,
40
- match_url_prefix_case_insensitive,
41
- is_at_absolute_path,
42
- is_at_relative_path,
43
- } from './mdz_helpers.js';
44
-
45
- //
46
- // Token types
47
- //
48
-
49
- export interface MdzTokenBase {
50
- start: number;
51
- end: number;
52
- }
53
-
54
- export type MdzToken =
55
- | MdzTokenText
56
- | MdzTokenCode
57
- | MdzTokenCodeblock
58
- | MdzTokenBoldOpen
59
- | MdzTokenBoldClose
60
- | MdzTokenItalicOpen
61
- | MdzTokenItalicClose
62
- | MdzTokenStrikethroughOpen
63
- | MdzTokenStrikethroughClose
64
- | MdzTokenLinkTextOpen
65
- | MdzTokenLinkTextClose
66
- | MdzTokenLinkRef
67
- | MdzTokenAutolink
68
- | MdzTokenHeadingStart
69
- | MdzTokenHr
70
- | MdzTokenTagOpen
71
- | MdzTokenTagSelfClose
72
- | MdzTokenTagClose
73
- | MdzTokenHeadingEnd
74
- | MdzTokenParagraphBreak;
75
-
76
- export interface MdzTokenText extends MdzTokenBase {
77
- type: 'text';
78
- content: string;
79
- }
80
-
81
- export interface MdzTokenCode extends MdzTokenBase {
82
- type: 'code';
83
- content: string;
84
- }
85
-
86
- export interface MdzTokenCodeblock extends MdzTokenBase {
87
- type: 'codeblock';
88
- lang: string | null;
89
- content: string;
90
- }
91
-
92
- export interface MdzTokenBoldOpen extends MdzTokenBase {
93
- type: 'bold_open';
94
- }
95
-
96
- export interface MdzTokenBoldClose extends MdzTokenBase {
97
- type: 'bold_close';
98
- }
99
-
100
- export interface MdzTokenItalicOpen extends MdzTokenBase {
101
- type: 'italic_open';
102
- }
103
-
104
- export interface MdzTokenItalicClose extends MdzTokenBase {
105
- type: 'italic_close';
106
- }
107
-
108
- export interface MdzTokenStrikethroughOpen extends MdzTokenBase {
109
- type: 'strikethrough_open';
110
- }
111
-
112
- export interface MdzTokenStrikethroughClose extends MdzTokenBase {
113
- type: 'strikethrough_close';
114
- }
115
-
116
- export interface MdzTokenLinkTextOpen extends MdzTokenBase {
117
- type: 'link_text_open';
118
- }
119
-
120
- export interface MdzTokenLinkTextClose extends MdzTokenBase {
121
- type: 'link_text_close';
122
- }
123
-
124
- export interface MdzTokenLinkRef extends MdzTokenBase {
125
- type: 'link_ref';
126
- reference: string;
127
- link_type: 'external' | 'internal';
128
- }
129
-
130
- export interface MdzTokenAutolink extends MdzTokenBase {
131
- type: 'autolink';
132
- reference: string;
133
- link_type: 'external' | 'internal';
134
- }
135
-
136
- export interface MdzTokenHeadingStart extends MdzTokenBase {
137
- type: 'heading_start';
138
- level: 1 | 2 | 3 | 4 | 5 | 6;
139
- }
140
-
141
- export interface MdzTokenHr extends MdzTokenBase {
142
- type: 'hr';
143
- }
144
-
145
- export interface MdzTokenTagOpen extends MdzTokenBase {
146
- type: 'tag_open';
147
- name: string;
148
- is_component: boolean;
149
- }
150
-
151
- export interface MdzTokenTagSelfClose extends MdzTokenBase {
152
- type: 'tag_self_close';
153
- name: string;
154
- is_component: boolean;
155
- }
156
-
157
- export interface MdzTokenTagClose extends MdzTokenBase {
158
- type: 'tag_close';
159
- name: string;
160
- }
161
-
162
- export interface MdzTokenHeadingEnd extends MdzTokenBase {
163
- type: 'heading_end';
164
- }
165
-
166
- export interface MdzTokenParagraphBreak extends MdzTokenBase {
167
- type: 'paragraph_break';
168
- }
169
-
170
- //
171
- // Lexer
172
- //
173
-
174
- export class MdzLexer {
175
- #text: string;
176
- #index: number = 0;
177
- #tokens: Array<MdzToken> = [];
178
- #max_search_index: number = Number.MAX_SAFE_INTEGER;
179
-
180
- constructor(text: string) {
181
- this.#text = text;
182
- }
183
-
184
- tokenize(): Array<MdzToken> {
185
- // Skip leading newlines
186
- this.#skip_newlines();
187
-
188
- while (this.#index < this.#text.length) {
189
- // Check for block elements at column 0
190
- if (this.#is_at_column_0()) {
191
- if (this.#tokenize_heading()) continue;
192
- if (this.#tokenize_hr()) continue;
193
- if (this.#tokenize_codeblock()) continue;
194
- }
195
-
196
- // Check for paragraph break
197
- if (this.#is_at_paragraph_break()) {
198
- const start = this.#index;
199
- this.#skip_newlines();
200
- this.#tokens.push({type: 'paragraph_break', start, end: this.#index});
201
- continue;
202
- }
203
-
204
- // Inline tokenization
205
- this.#tokenize_inline();
206
- }
207
-
208
- return this.#tokens;
209
- }
210
-
211
- #is_at_column_0(): boolean {
212
- return this.#index === 0 || this.#text.charCodeAt(this.#index - 1) === NEWLINE;
213
- }
214
-
215
- #is_at_paragraph_break(): boolean {
216
- return (
217
- this.#index + 1 < this.#text.length &&
218
- this.#text.charCodeAt(this.#index) === NEWLINE &&
219
- this.#text.charCodeAt(this.#index + 1) === NEWLINE
220
- );
221
- }
222
-
223
- #match(str: string): boolean {
224
- return this.#text.startsWith(str, this.#index);
225
- }
226
-
227
- #skip_newlines(): void {
228
- while (this.#index < this.#text.length && this.#text.charCodeAt(this.#index) === NEWLINE) {
229
- this.#index++;
230
- }
231
- }
232
-
233
- // -- Block tokenizers --
234
-
235
- #tokenize_heading(): boolean {
236
- let i = this.#index;
237
-
238
- // Count hashes (must be 1-6)
239
- let hash_count = 0;
240
- while (
241
- i < this.#text.length &&
242
- this.#text.charCodeAt(i) === HASH &&
243
- hash_count <= MAX_HEADING_LEVEL
244
- ) {
245
- hash_count++;
246
- i++;
247
- }
248
-
249
- if (hash_count === 0 || hash_count > MAX_HEADING_LEVEL) return false;
250
-
251
- // Must have space after hashes
252
- if (i >= this.#text.length || this.#text.charCodeAt(i) !== SPACE) return false;
253
- i++; // consume the space
254
-
255
- // Must have at least one non-whitespace character after the space
256
- let has_content = false;
257
- for (let j = i; j < this.#text.length; j++) {
258
- const char_code = this.#text.charCodeAt(j);
259
- if (char_code === NEWLINE) break;
260
- if (char_code !== SPACE && char_code !== TAB) {
261
- has_content = true;
262
- break;
263
- }
264
- }
265
-
266
- if (!has_content) return false;
267
-
268
- const start = this.#index;
269
- // Advance past "## " (hashes + space)
270
- this.#index = start + hash_count + 1;
271
-
272
- this.#tokens.push({
273
- type: 'heading_start',
274
- level: hash_count as 1 | 2 | 3 | 4 | 5 | 6,
275
- start,
276
- end: this.#index, // end of "## " prefix
277
- });
278
-
279
- // Find end-of-line to bound nested tokenizers (prevents tag scanner from scanning past heading)
280
- let eol = this.#text.indexOf('\n', this.#index);
281
- if (eol === -1) eol = this.#text.length;
282
-
283
- const saved_max = this.#max_search_index;
284
- this.#max_search_index = eol;
285
-
286
- // Tokenize inline content until newline or EOF
287
- // tokenize_text may consume a newline as part of block-element lookahead,
288
- // so we check emitted text tokens for embedded newlines and trim.
289
- while (this.#index < eol) {
290
- const token_count_before = this.#tokens.length;
291
- this.#tokenize_inline();
292
-
293
- // Check if the last emitted token is text containing a newline
294
- if (this.#tokens.length > token_count_before) {
295
- const last_token = this.#tokens[this.#tokens.length - 1]!;
296
- if (last_token.type === 'text') {
297
- const newline_idx = last_token.content.indexOf('\n');
298
- if (newline_idx !== -1) {
299
- const trimmed = last_token.content.slice(0, newline_idx);
300
- if (trimmed) {
301
- last_token.content = trimmed;
302
- last_token.end = last_token.start + trimmed.length;
303
- } else {
304
- this.#tokens.pop();
305
- }
306
- this.#index = last_token.start + newline_idx;
307
- break;
308
- }
309
- }
310
- }
311
- }
312
-
313
- this.#max_search_index = saved_max;
314
-
315
- // Emit heading_end marker so the token parser knows where heading content stops
316
- this.#tokens.push({type: 'heading_end', start: this.#index, end: this.#index});
317
-
318
- // Skip newlines after heading (like original parser's main loop)
319
- this.#skip_newlines();
320
-
321
- return true;
322
- }
323
-
324
- #tokenize_hr(): boolean {
325
- let i = this.#index;
326
-
327
- // Must have exactly three hyphens
328
- if (
329
- i + HR_HYPHEN_COUNT > this.#text.length ||
330
- this.#text.charCodeAt(i) !== HYPHEN ||
331
- this.#text.charCodeAt(i + 1) !== HYPHEN ||
332
- this.#text.charCodeAt(i + 2) !== HYPHEN
333
- ) {
334
- return false;
335
- }
336
- i += HR_HYPHEN_COUNT;
337
-
338
- // After the three hyphens, only whitespace and newline (or EOF) allowed
339
- while (i < this.#text.length) {
340
- const char_code = this.#text.charCodeAt(i);
341
- if (char_code === NEWLINE) break;
342
- if (char_code !== SPACE) return false;
343
- i++;
344
- }
345
-
346
- const start = this.#index;
347
- this.#index = i; // consume up to but not including newline
348
-
349
- this.#tokens.push({type: 'hr', start, end: this.#index});
350
-
351
- // Skip newlines after HR
352
- this.#skip_newlines();
353
-
354
- return true;
355
- }
356
-
357
- #tokenize_codeblock(): boolean {
358
- let i = this.#index;
359
-
360
- // Count backticks
361
- let backtick_count = 0;
362
- while (i < this.#text.length && this.#text.charCodeAt(i) === BACKTICK) {
363
- backtick_count++;
364
- i++;
365
- }
366
-
367
- if (backtick_count < MIN_CODEBLOCK_BACKTICKS) return false;
368
-
369
- // Skip optional language hint
370
- const lang_start = i;
371
- while (i < this.#text.length) {
372
- const char_code = this.#text.charCodeAt(i);
373
- if (char_code === SPACE || char_code === NEWLINE) break;
374
- i++;
375
- }
376
- const lang = i > lang_start ? this.#text.slice(lang_start, i) : null;
377
-
378
- // Skip trailing spaces on opening fence line
379
- while (i < this.#text.length && this.#text.charCodeAt(i) === SPACE) {
380
- i++;
381
- }
382
-
383
- // Must have newline after opening fence
384
- if (i >= this.#text.length || this.#text.charCodeAt(i) !== NEWLINE) return false;
385
- i++; // consume the newline
386
-
387
- const content_start = i;
388
-
389
- // Search for closing fence
390
- const closing_fence = '`'.repeat(backtick_count);
391
- while (i < this.#text.length) {
392
- if (this.#text.startsWith(closing_fence, i)) {
393
- const prev_char = i > 0 ? this.#text.charCodeAt(i - 1) : NEWLINE;
394
- if (prev_char === NEWLINE || i === 0) {
395
- const content = this.#text.slice(content_start, i);
396
- const final_content = content.endsWith('\n') ? content.slice(0, -1) : content;
397
- if (final_content.length === 0) return false; // Empty code block
398
-
399
- // Verify closing fence line
400
- let j = i + backtick_count;
401
- while (j < this.#text.length && this.#text.charCodeAt(j) === SPACE) {
402
- j++;
403
- }
404
- if (j < this.#text.length && this.#text.charCodeAt(j) !== NEWLINE) {
405
- return false;
406
- }
407
-
408
- const start = this.#index;
409
- this.#index = j; // advance past closing fence and trailing spaces
410
-
411
- this.#tokens.push({
412
- type: 'codeblock',
413
- lang,
414
- content: final_content,
415
- start,
416
- end: this.#index,
417
- });
418
-
419
- // Skip newlines after codeblock
420
- this.#skip_newlines();
421
-
422
- return true;
423
- }
424
- }
425
- i++;
426
- }
427
-
428
- return false; // No closing fence found
429
- }
430
-
431
- // -- Inline tokenizers --
432
-
433
- #tokenize_inline(): void {
434
- const char_code = this.#text.charCodeAt(this.#index);
435
-
436
- switch (char_code) {
437
- case BACKTICK:
438
- this.#tokenize_code();
439
- return;
440
- case ASTERISK:
441
- this.#tokenize_bold();
442
- return;
443
- case UNDERSCORE:
444
- this.#tokenize_single_delimiter('_', 'italic');
445
- return;
446
- case TILDE:
447
- this.#tokenize_single_delimiter('~', 'strikethrough');
448
- return;
449
- case LEFT_BRACKET:
450
- this.#tokenize_markdown_link();
451
- return;
452
- case LEFT_ANGLE:
453
- this.#tokenize_tag();
454
- return;
455
- default:
456
- this.#tokenize_text();
457
- return;
458
- }
459
- }
460
-
461
- #tokenize_code(): void {
462
- const start = this.#index;
463
- this.#index++; // consume `
464
-
465
- // Find closing backtick, stop at newline or search boundary
466
- let content_end = -1;
467
- const search_limit = Math.min(this.#max_search_index, this.#text.length);
468
- for (let i = this.#index; i < search_limit; i++) {
469
- const char_code = this.#text.charCodeAt(i);
470
- if (char_code === BACKTICK) {
471
- content_end = i;
472
- break;
473
- }
474
- if (char_code === NEWLINE) break;
475
- }
476
-
477
- if (content_end === -1) {
478
- // Unclosed backtick
479
- this.#emit_text('`', start);
480
- return;
481
- }
482
-
483
- const content = this.#text.slice(this.#index, content_end);
484
-
485
- // Empty inline code
486
- if (content.length === 0) {
487
- this.#index = content_end + 1;
488
- this.#emit_text('``', start);
489
- return;
490
- }
491
-
492
- this.#index = content_end + 1;
493
- this.#tokens.push({type: 'code', content, start, end: this.#index});
494
- }
495
-
496
- #tokenize_bold(): void {
497
- const start = this.#index;
498
-
499
- // Check for **
500
- if (!this.#match('**')) {
501
- // Single asterisk - text
502
- this.#index++;
503
- this.#emit_text('*', start);
504
- return;
505
- }
506
-
507
- // Find closing ** within current search boundary
508
- const search_end = Math.min(this.#max_search_index, this.#text.length);
509
- let close_index = this.#text.indexOf('**', start + 2);
510
- if (close_index !== -1 && close_index >= search_end) {
511
- close_index = -1;
512
- }
513
- if (close_index === -1) {
514
- // Unclosed **
515
- this.#index += 2;
516
- this.#emit_text('**', start);
517
- return;
518
- }
519
-
520
- // Check for paragraph break between open and close
521
- if (this.#has_paragraph_break_between(start + 2, close_index)) {
522
- this.#index += 2;
523
- this.#emit_text('**', start);
524
- return;
525
- }
526
-
527
- // Emit bold_open, tokenize children up to close, emit bold_close
528
- this.#index += 2;
529
- const open_token_index = this.#tokens.length;
530
- this.#tokens.push({type: 'bold_open', start, end: this.#index});
531
-
532
- // Set search boundary for nested parsers
533
- const saved_max = this.#max_search_index;
534
- this.#max_search_index = close_index;
535
-
536
- // Tokenize children inline content up to close_index
537
- while (this.#index < close_index) {
538
- if (this.#is_at_paragraph_break()) break;
539
- this.#tokenize_inline();
540
- }
541
-
542
- this.#max_search_index = saved_max;
543
-
544
- if (this.#index === close_index && this.#match('**')) {
545
- // Check if empty
546
- const has_children = open_token_index < this.#tokens.length - 1;
547
-
548
- if (!has_children) {
549
- // Empty bold - convert to text
550
- this.#tokens.splice(open_token_index, 1);
551
- this.#index = start;
552
- this.#index += 4;
553
- this.#emit_text('****', start);
554
- return;
555
- }
556
-
557
- this.#tokens.push({type: 'bold_close', start: close_index, end: close_index + 2});
558
- this.#index = close_index + 2;
559
- } else {
560
- // Didn't reach closing - convert opening to text
561
- this.#tokens[open_token_index] = {type: 'text', content: '**', start, end: start + 2};
562
- }
563
- }
564
-
565
- #tokenize_single_delimiter(delimiter: '_' | '~', kind: 'italic' | 'strikethrough'): void {
566
- const start = this.#index;
567
-
568
- // Check opening word boundary
569
- if (!this.#is_at_word_boundary(this.#index, true, false)) {
570
- this.#index++;
571
- this.#emit_text(delimiter, start);
572
- return;
573
- }
574
-
575
- // Find closing delimiter within search boundary
576
- const search_end = Math.min(this.#max_search_index, this.#text.length);
577
- let close_index = this.#text.indexOf(delimiter, start + 1);
578
- if (close_index !== -1 && close_index >= search_end) {
579
- close_index = -1;
580
- }
581
- if (close_index === -1) {
582
- this.#index++;
583
- this.#emit_text(delimiter, start);
584
- return;
585
- }
586
-
587
- // Check closing word boundary
588
- if (!this.#is_at_word_boundary(close_index + 1, false, true)) {
589
- this.#index++;
590
- this.#emit_text(delimiter, start);
591
- return;
592
- }
593
-
594
- // Check for paragraph break between
595
- if (this.#has_paragraph_break_between(start + 1, close_index)) {
596
- this.#index++;
597
- this.#emit_text(delimiter, start);
598
- return;
599
- }
600
-
601
- // Emit open token
602
- this.#index++;
603
- const open_type = kind === 'italic' ? 'italic_open' : 'strikethrough_open';
604
- const close_type = kind === 'italic' ? 'italic_close' : 'strikethrough_close';
605
- const open_token_index = this.#tokens.length;
606
- this.#tokens.push({type: open_type, start, end: this.#index});
607
-
608
- // Set search boundary for nested parsers
609
- const saved_max = this.#max_search_index;
610
- this.#max_search_index = close_index;
611
-
612
- // Tokenize children up to close_index
613
- while (this.#index < close_index) {
614
- if (this.#is_at_paragraph_break()) break;
615
- this.#tokenize_inline();
616
- }
617
-
618
- this.#max_search_index = saved_max;
619
-
620
- if (this.#index === close_index && this.#match(delimiter)) {
621
- // Check if empty
622
- const has_children = open_token_index < this.#tokens.length - 1;
623
-
624
- if (!has_children) {
625
- // Empty - convert to text
626
- this.#tokens.splice(open_token_index, 1);
627
- this.#index = start;
628
- this.#index += 2;
629
- this.#emit_text(delimiter + delimiter, start);
630
- return;
631
- }
632
-
633
- this.#tokens.push({
634
- type: close_type,
635
- start: close_index,
636
- end: close_index + 1,
637
- });
638
- this.#index = close_index + 1;
639
- } else {
640
- // Convert opening to text
641
- this.#tokens[open_token_index] = {
642
- type: 'text',
643
- content: delimiter,
644
- start,
645
- end: start + 1,
646
- };
647
- }
648
- }
649
-
650
- #tokenize_markdown_link(): void {
651
- const start = this.#index;
652
-
653
- // Consume [
654
- if (this.#text.charCodeAt(this.#index) !== LEFT_BRACKET) {
655
- this.#index++;
656
- this.#emit_text(this.#text[start]!, start);
657
- return;
658
- }
659
- this.#index++;
660
-
661
- // Emit link_text_open
662
- this.#tokens.push({type: 'link_text_open', start, end: this.#index});
663
-
664
- // Tokenize children until ]
665
- while (this.#index < this.#text.length) {
666
- if (this.#text.charCodeAt(this.#index) === RIGHT_BRACKET) break;
667
- if (this.#is_at_paragraph_break()) break;
668
- // Stop at ] and ) as delimiters
669
- if (this.#text.charCodeAt(this.#index) === RIGHT_PAREN) break;
670
- this.#tokenize_inline();
671
- }
672
-
673
- // Check for ]
674
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== RIGHT_BRACKET) {
675
- // Revert - remove link_text_open and all children tokens added
676
- this.#revert_tokens_from_link_open(start);
677
- return;
678
- }
679
-
680
- const bracket_close_start = this.#index;
681
- this.#index++; // consume ]
682
- this.#tokens.push({
683
- type: 'link_text_close',
684
- start: bracket_close_start,
685
- end: this.#index,
686
- });
687
-
688
- // Check for (
689
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== LEFT_PAREN) {
690
- this.#revert_tokens_from_link_open(start);
691
- return;
692
- }
693
- this.#index++; // consume (
694
-
695
- // Find closing )
696
- const close_paren = this.#text.indexOf(')', this.#index);
697
- if (close_paren === -1) {
698
- this.#revert_tokens_from_link_open(start);
699
- return;
700
- }
701
-
702
- const reference = this.#text.slice(this.#index, close_paren);
703
-
704
- // Validate reference
705
- if (!reference.trim()) {
706
- this.#revert_tokens_from_link_open(start);
707
- return;
708
- }
709
-
710
- // Validate all characters
711
- for (let i = 0; i < reference.length; i++) {
712
- const char_code = reference.charCodeAt(i);
713
- if (!is_valid_path_char(char_code)) {
714
- this.#revert_tokens_from_link_open(start);
715
- return;
716
- }
717
- }
718
-
719
- // Reject unsafe protocols (javascript:, data:, etc.) — `is_valid_path_char`
720
- // permits `:` so we need an explicit filter here.
721
- if (!mdz_is_safe_reference(reference)) {
722
- this.#revert_tokens_from_link_open(start);
723
- return;
724
- }
725
-
726
- this.#index = close_paren + 1;
727
-
728
- const link_type = mdz_is_url(reference) ? 'external' : 'internal';
729
-
730
- this.#tokens.push({
731
- type: 'link_ref',
732
- reference,
733
- link_type,
734
- start: bracket_close_start + 1, // after ]
735
- end: this.#index,
736
- });
737
- }
738
-
739
- #revert_tokens_from_link_open(start: number): void {
740
- // Find and remove link_text_open and all tokens after it
741
- let open_idx = -1;
742
- for (let i = this.#tokens.length - 1; i >= 0; i--) {
743
- if (this.#tokens[i]!.type === 'link_text_open' && this.#tokens[i]!.start === start) {
744
- open_idx = i;
745
- break;
746
- }
747
- }
748
- if (open_idx !== -1) {
749
- this.#tokens.splice(open_idx);
750
- }
751
- this.#index = start + 1;
752
- this.#emit_text('[', start);
753
- }
754
-
755
- #tokenize_tag(): void {
756
- const start = this.#index;
757
- this.#index++; // consume <
758
-
759
- // Tag name must start with a letter
760
- if (this.#index >= this.#text.length || !is_letter(this.#text.charCodeAt(this.#index))) {
761
- this.#emit_text('<', start);
762
- return;
763
- }
764
-
765
- // Collect tag name
766
- const tag_name_start = this.#index;
767
- while (
768
- this.#index < this.#text.length &&
769
- is_tag_name_char(this.#text.charCodeAt(this.#index))
770
- ) {
771
- this.#index++;
772
- }
773
- const tag_name = this.#text.slice(tag_name_start, this.#index);
774
-
775
- if (tag_name.length === 0) {
776
- this.#emit_text('<', start);
777
- return;
778
- }
779
-
780
- const first_char_code = tag_name.charCodeAt(0);
781
- const is_component = first_char_code >= A_UPPER && first_char_code <= Z_UPPER;
782
-
783
- // Skip whitespace
784
- while (this.#index < this.#text.length && this.#text.charCodeAt(this.#index) === SPACE) {
785
- this.#index++;
786
- }
787
-
788
- // Check for self-closing />
789
- if (
790
- this.#index + 1 < this.#text.length &&
791
- this.#text.charCodeAt(this.#index) === SLASH &&
792
- this.#text.charCodeAt(this.#index + 1) === RIGHT_ANGLE
793
- ) {
794
- this.#index += 2;
795
- this.#tokens.push({
796
- type: 'tag_self_close',
797
- name: tag_name,
798
- is_component,
799
- start,
800
- end: this.#index,
801
- });
802
- return;
803
- }
804
-
805
- // Check for >
806
- if (this.#index >= this.#text.length || this.#text.charCodeAt(this.#index) !== RIGHT_ANGLE) {
807
- this.#index = start + 1;
808
- this.#emit_text('<', start);
809
- return;
810
- }
811
- this.#index++; // consume >
812
-
813
- // Check for closing tag existence before committing —
814
- // must exist within search boundary and before any paragraph break
815
- const closing_tag = `</${tag_name}>`;
816
- const search_limit = Math.min(this.#max_search_index, this.#text.length);
817
- const closing_tag_pos = this.#text.indexOf(closing_tag, this.#index);
818
- if (closing_tag_pos === -1 || closing_tag_pos >= search_limit) {
819
- this.#index = start + 1;
820
- this.#emit_text('<', start);
821
- return;
822
- }
823
- if (this.#has_paragraph_break_between(this.#index, closing_tag_pos)) {
824
- this.#index = start + 1;
825
- this.#emit_text('<', start);
826
- return;
827
- }
828
-
829
- // Emit tag_open
830
- this.#tokens.push({type: 'tag_open', name: tag_name, is_component, start, end: this.#index});
831
-
832
- // Tokenize children until closing tag
833
- while (this.#index < this.#text.length) {
834
- if (this.#match(closing_tag)) {
835
- const close_start = this.#index;
836
- this.#index += closing_tag.length;
837
- this.#tokens.push({
838
- type: 'tag_close',
839
- name: tag_name,
840
- start: close_start,
841
- end: this.#index,
842
- });
843
- return;
844
- }
845
- this.#tokenize_inline();
846
- }
847
-
848
- // Shouldn't reach here since we verified closing tag exists
849
- // But if we do, the tag_open token stays and parser will handle it
850
- }
851
-
852
- #tokenize_text(): void {
853
- const start = this.#index;
854
-
855
- // Check for URL or internal path at current position
856
- if (this.#is_at_url()) {
857
- this.#tokenize_auto_link_url();
858
- return;
859
- }
860
- if (is_at_absolute_path(this.#text, this.#index)) {
861
- this.#tokenize_auto_link_internal();
862
- return;
863
- }
864
- if (is_at_relative_path(this.#text, this.#index)) {
865
- this.#tokenize_auto_link_internal();
866
- return;
867
- }
868
-
869
- while (this.#index < this.#text.length) {
870
- const char_code = this.#text.charCodeAt(this.#index);
871
-
872
- // Stop at special characters
873
- if (
874
- char_code === BACKTICK ||
875
- char_code === ASTERISK ||
876
- char_code === UNDERSCORE ||
877
- char_code === TILDE ||
878
- char_code === LEFT_BRACKET ||
879
- char_code === RIGHT_BRACKET ||
880
- char_code === RIGHT_PAREN ||
881
- char_code === LEFT_ANGLE
882
- ) {
883
- break;
884
- }
885
-
886
- // Check for paragraph break
887
- if (this.#is_at_paragraph_break()) break;
888
-
889
- // When next line could start a block element, consume the newline and stop
890
- if (char_code === NEWLINE) {
891
- const next_i = this.#index + 1;
892
- if (next_i < this.#text.length) {
893
- const next_char = this.#text.charCodeAt(next_i);
894
- if (next_char === HASH || next_char === HYPHEN || next_char === BACKTICK) {
895
- this.#index++; // consume the newline
896
- break;
897
- }
898
- }
899
- }
900
-
901
- // Check for URL or internal path mid-text (char code guard avoids startsWith on every char)
902
- if (
903
- ((char_code === 104 /* h */ || char_code === 72) /* H */ && this.#is_at_url()) ||
904
- (char_code === SLASH && is_at_absolute_path(this.#text, this.#index)) ||
905
- (char_code === PERIOD && is_at_relative_path(this.#text, this.#index))
906
- ) {
907
- break;
908
- }
909
-
910
- this.#index++;
911
- }
912
-
913
- // Ensure we always consume at least one character
914
- if (this.#index === start && this.#index < this.#text.length) {
915
- this.#index++;
916
- }
917
-
918
- const content = this.#text.slice(start, this.#index);
919
- this.#emit_text(content, start);
920
- }
921
-
922
- // -- Auto-link tokenizers --
923
-
924
- #tokenize_auto_link_url(): void {
925
- const start = this.#index;
926
-
927
- // Consume protocol (case-insensitive match; original casing preserved in reference)
928
- this.#index += match_url_prefix_case_insensitive(this.#text, this.#index);
929
-
930
- // Collect URL characters
931
- while (this.#index < this.#text.length) {
932
- const char_code = this.#text.charCodeAt(this.#index);
933
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
934
- break;
935
- }
936
- this.#index++;
937
- }
938
-
939
- let reference = this.#text.slice(start, this.#index);
940
- reference = trim_trailing_punctuation(reference);
941
- this.#index = start + reference.length;
942
-
943
- this.#tokens.push({
944
- type: 'autolink',
945
- reference,
946
- link_type: 'external',
947
- start,
948
- end: this.#index,
949
- });
950
- }
951
-
952
- #tokenize_auto_link_internal(): void {
953
- const start = this.#index;
954
-
955
- // Collect path characters
956
- while (this.#index < this.#text.length) {
957
- const char_code = this.#text.charCodeAt(this.#index);
958
- if (char_code === SPACE || char_code === NEWLINE || !is_valid_path_char(char_code)) {
959
- break;
960
- }
961
- this.#index++;
962
- }
963
-
964
- let reference = this.#text.slice(start, this.#index);
965
- reference = trim_trailing_punctuation(reference);
966
- this.#index = start + reference.length;
967
-
968
- this.#tokens.push({
969
- type: 'autolink',
970
- reference,
971
- link_type: 'internal',
972
- start,
973
- end: this.#index,
974
- });
975
- }
976
-
977
- // -- Helper methods --
978
-
979
- #emit_text(content: string, start: number): void {
980
- this.#tokens.push({type: 'text', content, start, end: start + content.length});
981
- }
982
-
983
- #has_paragraph_break_between(from: number, to: number): boolean {
984
- for (let i = from; i < to - 1; i++) {
985
- if (this.#text.charCodeAt(i) === NEWLINE && this.#text.charCodeAt(i + 1) === NEWLINE) {
986
- return true;
987
- }
988
- }
989
- return false;
990
- }
991
-
992
- #is_at_url(): boolean {
993
- // Scheme matching is case-insensitive (RFC 3986).
994
- const prefix_len = match_url_prefix_case_insensitive(this.#text, this.#index);
995
- if (prefix_len === 0) return false;
996
- if (this.#index + prefix_len >= this.#text.length) return false;
997
- const next_char = this.#text.charCodeAt(this.#index + prefix_len);
998
- return next_char !== SPACE && next_char !== NEWLINE;
999
- }
1000
-
1001
- #is_at_word_boundary(index: number, check_before: boolean, check_after: boolean): boolean {
1002
- if (check_before && index > 0) {
1003
- const prev = this.#text.charCodeAt(index - 1);
1004
- if (is_word_char(prev)) return false;
1005
- }
1006
- if (check_after && index < this.#text.length) {
1007
- const next = this.#text.charCodeAt(index);
1008
- if (is_word_char(next)) return false;
1009
- }
1010
- return true;
1011
- }
1012
- }