@fuzdev/fuz_ui 0.198.1 → 0.200.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ApiDeclarationList.svelte +2 -1
- package/dist/ApiDeclarationList.svelte.d.ts.map +1 -1
- package/dist/ApiIndex.svelte +2 -2
- package/dist/ApiModule.svelte +3 -3
- package/dist/DeclarationDetail.svelte +145 -90
- package/dist/DeclarationDetail.svelte.d.ts.map +1 -1
- package/dist/LibraryDetail.svelte +14 -16
- package/dist/LibraryDetail.svelte.d.ts.map +1 -1
- package/dist/LibrarySummary.svelte +12 -12
- package/dist/Mdz.svelte +2 -2
- package/dist/MdzNodeView.svelte +8 -6
- package/dist/MdzNodeView.svelte.d.ts.map +1 -1
- package/dist/MdzRoot.svelte +30 -0
- package/dist/MdzRoot.svelte.d.ts +12 -0
- package/dist/MdzRoot.svelte.d.ts.map +1 -0
- package/dist/MdzStream.svelte +32 -0
- package/dist/MdzStream.svelte.d.ts +12 -0
- package/dist/MdzStream.svelte.d.ts.map +1 -0
- package/dist/MdzStreamNodeView.svelte +106 -0
- package/dist/MdzStreamNodeView.svelte.d.ts +9 -0
- package/dist/MdzStreamNodeView.svelte.d.ts.map +1 -0
- package/dist/ProjectLinks.svelte +1 -1
- package/dist/api_search.svelte.d.ts.map +1 -1
- package/dist/api_search.svelte.js +4 -2
- package/dist/declaration.svelte.d.ts +11 -0
- package/dist/declaration.svelte.d.ts.map +1 -1
- package/dist/declaration.svelte.js +15 -3
- package/dist/library.svelte.d.ts +9 -70
- package/dist/library.svelte.d.ts.map +1 -1
- package/dist/library.svelte.js +24 -18
- package/dist/library_helpers.d.ts +3 -3
- package/dist/library_helpers.js +3 -3
- package/dist/mdz.d.ts.map +1 -1
- package/dist/mdz.js +37 -29
- package/dist/mdz_components.d.ts +37 -23
- package/dist/mdz_components.d.ts.map +1 -1
- package/dist/mdz_components.js +16 -4
- package/dist/mdz_helpers.d.ts +46 -14
- package/dist/mdz_helpers.d.ts.map +1 -1
- package/dist/mdz_helpers.js +189 -56
- package/dist/mdz_lexer.d.ts.map +1 -1
- package/dist/mdz_lexer.js +18 -22
- package/dist/mdz_opcodes.d.ts +174 -0
- package/dist/mdz_opcodes.d.ts.map +1 -0
- package/dist/mdz_opcodes.js +14 -0
- package/dist/mdz_opcodes_to_nodes.d.ts +20 -0
- package/dist/mdz_opcodes_to_nodes.d.ts.map +1 -0
- package/dist/mdz_opcodes_to_nodes.js +332 -0
- package/dist/mdz_stream_parser.d.ts +80 -0
- package/dist/mdz_stream_parser.d.ts.map +1 -0
- package/dist/mdz_stream_parser.js +354 -0
- package/dist/mdz_stream_parser_block.d.ts +76 -0
- package/dist/mdz_stream_parser_block.d.ts.map +1 -0
- package/dist/mdz_stream_parser_block.js +458 -0
- package/dist/mdz_stream_parser_inline.d.ts +52 -0
- package/dist/mdz_stream_parser_inline.d.ts.map +1 -0
- package/dist/mdz_stream_parser_inline.js +378 -0
- package/dist/mdz_stream_parser_link.d.ts +22 -0
- package/dist/mdz_stream_parser_link.d.ts.map +1 -0
- package/dist/mdz_stream_parser_link.js +215 -0
- package/dist/mdz_stream_parser_state.d.ts +187 -0
- package/dist/mdz_stream_parser_state.d.ts.map +1 -0
- package/dist/mdz_stream_parser_state.js +398 -0
- package/dist/mdz_stream_parser_text.d.ts +14 -0
- package/dist/mdz_stream_parser_text.d.ts.map +1 -0
- package/dist/mdz_stream_parser_text.js +50 -0
- package/dist/mdz_stream_parser_url.d.ts +60 -0
- package/dist/mdz_stream_parser_url.d.ts.map +1 -0
- package/dist/mdz_stream_parser_url.js +307 -0
- package/dist/mdz_stream_state.svelte.d.ts +44 -0
- package/dist/mdz_stream_state.svelte.d.ts.map +1 -0
- package/dist/mdz_stream_state.svelte.js +357 -0
- package/dist/mdz_token_parser.d.ts.map +1 -1
- package/dist/mdz_token_parser.js +8 -39
- package/dist/module.svelte.d.ts +2 -0
- package/dist/module.svelte.d.ts.map +1 -1
- package/dist/module.svelte.js +3 -1
- package/dist/site.svelte.d.ts +13 -0
- package/dist/site.svelte.d.ts.map +1 -1
- package/dist/site.svelte.js +8 -2
- package/dist/tsdoc_mdz.d.ts +2 -2
- package/dist/tsdoc_mdz.js +2 -2
- package/dist/vite_plugin_pkg_json.d.ts +63 -0
- package/dist/vite_plugin_pkg_json.d.ts.map +1 -0
- package/dist/vite_plugin_pkg_json.js +126 -0
- package/package.json +8 -7
- package/src/lib/api_search.svelte.ts +4 -2
- package/src/lib/declaration.svelte.ts +18 -3
- package/src/lib/library.svelte.ts +37 -20
- package/src/lib/library_helpers.ts +3 -3
- package/src/lib/mdz.ts +38 -29
- package/src/lib/mdz_components.ts +40 -19
- package/src/lib/mdz_helpers.ts +199 -56
- package/src/lib/mdz_lexer.ts +18 -20
- package/src/lib/mdz_opcodes.ts +205 -0
- package/src/lib/mdz_opcodes_to_nodes.ts +375 -0
- package/src/lib/mdz_stream_parser.ts +415 -0
- package/src/lib/mdz_stream_parser_block.ts +532 -0
- package/src/lib/mdz_stream_parser_inline.ts +414 -0
- package/src/lib/mdz_stream_parser_link.ts +271 -0
- package/src/lib/mdz_stream_parser_state.ts +539 -0
- package/src/lib/mdz_stream_parser_text.ts +77 -0
- package/src/lib/mdz_stream_parser_url.ts +365 -0
- package/src/lib/mdz_stream_state.svelte.ts +387 -0
- package/src/lib/mdz_token_parser.ts +13 -40
- package/src/lib/module.svelte.ts +5 -1
- package/src/lib/site.svelte.ts +17 -2
- package/src/lib/tsdoc_mdz.ts +2 -2
- package/src/lib/vite_plugin_pkg_json.ts +142 -0
- package/dist/package_helpers.d.ts +0 -150
- package/dist/package_helpers.d.ts.map +0 -1
- package/dist/package_helpers.js +0 -179
- package/src/lib/package_helpers.ts +0 -186
package/src/lib/mdz_helpers.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* @module
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
-
import type {MdzNode, MdzComponentNode, MdzElementNode} from './mdz.js';
|
|
10
|
+
import type {MdzNode, MdzTextNode, MdzComponentNode, MdzElementNode} from './mdz.js';
|
|
11
11
|
import {slugify} from '@fuzdev/fuz_util/path.js';
|
|
12
12
|
|
|
13
13
|
// Character codes for performance
|
|
@@ -54,6 +54,54 @@ export const MIN_CODEBLOCK_BACKTICKS = 3; // Code blocks require minimum 3 backt
|
|
|
54
54
|
export const MAX_HEADING_LEVEL = 6; // Headings support levels 1-6
|
|
55
55
|
export const HTTPS_PREFIX_LENGTH = 8; // Length of "https://"
|
|
56
56
|
export const HTTP_PREFIX_LENGTH = 7; // Length of "http://"
|
|
57
|
+
export const H_LOWER = 104; // h
|
|
58
|
+
export const H_UPPER = 72; // H
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Lowercase an ASCII letter code; passes other chars through.
|
|
62
|
+
* Mirrors the `[A-Z] → [a-z]` shift without touching non-ASCII.
|
|
63
|
+
*/
|
|
64
|
+
export const ascii_to_lower = (char_code: number): number =>
|
|
65
|
+
char_code >= A_UPPER && char_code <= Z_UPPER ? char_code + 32 : char_code;
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Case-insensitive URL scheme prefix check. Returns the prefix length
|
|
69
|
+
* (8 for `https://`, 7 for `http://`) or 0 if neither matches at `pos`.
|
|
70
|
+
*
|
|
71
|
+
* URI schemes are case-insensitive per RFC 3986 §3.1, and mobile keyboards
|
|
72
|
+
* routinely auto-capitalize the first letter of a sentence, producing
|
|
73
|
+
* `Https://` etc. The original case is preserved in the emitted reference —
|
|
74
|
+
* browsers normalize schemes themselves.
|
|
75
|
+
*/
|
|
76
|
+
export const match_url_prefix_case_insensitive = (s: string, pos: number): 0 | 7 | 8 => {
|
|
77
|
+
if (pos + HTTP_PREFIX_LENGTH > s.length) return 0;
|
|
78
|
+
// h | H
|
|
79
|
+
if (ascii_to_lower(s.charCodeAt(pos)) !== H_LOWER) return 0;
|
|
80
|
+
// 'ttps://' (7 chars) or 'ttp://' (6 chars)
|
|
81
|
+
if (
|
|
82
|
+
ascii_to_lower(s.charCodeAt(pos + 1)) === 116 /* t */ &&
|
|
83
|
+
ascii_to_lower(s.charCodeAt(pos + 2)) === 116 /* t */ &&
|
|
84
|
+
ascii_to_lower(s.charCodeAt(pos + 3)) === 112 /* p */
|
|
85
|
+
) {
|
|
86
|
+
if (
|
|
87
|
+
pos + HTTPS_PREFIX_LENGTH <= s.length &&
|
|
88
|
+
ascii_to_lower(s.charCodeAt(pos + 4)) === 115 /* s */ &&
|
|
89
|
+
s.charCodeAt(pos + 5) === COLON &&
|
|
90
|
+
s.charCodeAt(pos + 6) === SLASH &&
|
|
91
|
+
s.charCodeAt(pos + 7) === SLASH
|
|
92
|
+
) {
|
|
93
|
+
return HTTPS_PREFIX_LENGTH;
|
|
94
|
+
}
|
|
95
|
+
if (
|
|
96
|
+
s.charCodeAt(pos + 4) === COLON &&
|
|
97
|
+
s.charCodeAt(pos + 5) === SLASH &&
|
|
98
|
+
s.charCodeAt(pos + 6) === SLASH
|
|
99
|
+
) {
|
|
100
|
+
return HTTP_PREFIX_LENGTH;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return 0;
|
|
104
|
+
};
|
|
57
105
|
|
|
58
106
|
/**
|
|
59
107
|
* Check if character code is a letter (A-Z, a-z).
|
|
@@ -74,24 +122,21 @@ export const is_tag_name_char = (char_code: number): boolean =>
|
|
|
74
122
|
* Check if character is part of a word for word boundary detection.
|
|
75
123
|
* Used to prevent intraword emphasis with `_` and `~` delimiters.
|
|
76
124
|
*
|
|
77
|
-
*
|
|
78
|
-
*
|
|
125
|
+
* Only alphanumeric characters (A-Z, a-z, 0-9) are word characters.
|
|
126
|
+
* Formatting delimiters (`*`, `_`, `~`) fall outside all three ranges,
|
|
127
|
+
* so they're naturally excluded without explicit checks.
|
|
79
128
|
*
|
|
80
129
|
* This prevents false positives with snake_case identifiers while allowing
|
|
81
130
|
* adjacent formatting like `**bold**_italic_`.
|
|
82
131
|
*/
|
|
83
|
-
export const is_word_char = (char_code: number): boolean =>
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
(char_code >= A_LOWER && char_code <= Z_LOWER) ||
|
|
88
|
-
(char_code >= ZERO && char_code <= NINE)
|
|
89
|
-
);
|
|
90
|
-
};
|
|
132
|
+
export const is_word_char = (char_code: number): boolean =>
|
|
133
|
+
(char_code >= A_UPPER && char_code <= Z_UPPER) ||
|
|
134
|
+
(char_code >= A_LOWER && char_code <= Z_LOWER) ||
|
|
135
|
+
(char_code >= ZERO && char_code <= NINE);
|
|
91
136
|
|
|
92
137
|
/**
|
|
93
|
-
*
|
|
94
|
-
*
|
|
138
|
+
* Lookup table for valid URI path characters per RFC 3986.
|
|
139
|
+
* Replaces a 16-comparison chain with a single array access.
|
|
95
140
|
*
|
|
96
141
|
* Valid characters:
|
|
97
142
|
* - unreserved: A-Z a-z 0-9 - . _ ~
|
|
@@ -100,44 +145,60 @@ export const is_word_char = (char_code: number): boolean => {
|
|
|
100
145
|
* - separators: / ? #
|
|
101
146
|
* - percent-encoding: %
|
|
102
147
|
*/
|
|
148
|
+
const PATH_CHAR_TABLE: Uint8Array = (() => {
|
|
149
|
+
const t = new Uint8Array(128);
|
|
150
|
+
for (let i = A_UPPER; i <= Z_UPPER; i++) t[i] = 1;
|
|
151
|
+
for (let i = A_LOWER; i <= Z_LOWER; i++) t[i] = 1;
|
|
152
|
+
for (let i = ZERO; i <= NINE; i++) t[i] = 1;
|
|
153
|
+
// unreserved: - . _ ~
|
|
154
|
+
t[HYPHEN] = 1;
|
|
155
|
+
t[PERIOD] = 1;
|
|
156
|
+
t[UNDERSCORE] = 1;
|
|
157
|
+
t[TILDE] = 1;
|
|
158
|
+
// sub-delims: ! $ & ' ( ) * + , ; =
|
|
159
|
+
t[EXCLAMATION] = 1;
|
|
160
|
+
t[DOLLAR] = 1;
|
|
161
|
+
t[AMPERSAND] = 1;
|
|
162
|
+
t[APOSTROPHE] = 1;
|
|
163
|
+
t[LEFT_PAREN] = 1;
|
|
164
|
+
t[RIGHT_PAREN] = 1;
|
|
165
|
+
t[ASTERISK] = 1;
|
|
166
|
+
t[PLUS] = 1;
|
|
167
|
+
t[COMMA] = 1;
|
|
168
|
+
t[SEMICOLON] = 1;
|
|
169
|
+
t[EQUALS] = 1;
|
|
170
|
+
// path allowed: : @
|
|
171
|
+
t[COLON] = 1;
|
|
172
|
+
t[AT] = 1;
|
|
173
|
+
// separators: / ? #
|
|
174
|
+
t[SLASH] = 1;
|
|
175
|
+
t[QUESTION] = 1;
|
|
176
|
+
t[HASH] = 1;
|
|
177
|
+
// percent-encoding: %
|
|
178
|
+
t[PERCENT] = 1;
|
|
179
|
+
return t;
|
|
180
|
+
})();
|
|
181
|
+
|
|
103
182
|
export const is_valid_path_char = (char_code: number): boolean =>
|
|
104
|
-
|
|
105
|
-
(char_code >= A_LOWER && char_code <= Z_LOWER) ||
|
|
106
|
-
(char_code >= ZERO && char_code <= NINE) ||
|
|
107
|
-
char_code === HYPHEN ||
|
|
108
|
-
char_code === PERIOD ||
|
|
109
|
-
char_code === UNDERSCORE ||
|
|
110
|
-
char_code === TILDE ||
|
|
111
|
-
char_code === EXCLAMATION ||
|
|
112
|
-
char_code === DOLLAR ||
|
|
113
|
-
char_code === AMPERSAND ||
|
|
114
|
-
char_code === APOSTROPHE ||
|
|
115
|
-
char_code === LEFT_PAREN ||
|
|
116
|
-
char_code === RIGHT_PAREN ||
|
|
117
|
-
char_code === ASTERISK ||
|
|
118
|
-
char_code === PLUS ||
|
|
119
|
-
char_code === COMMA ||
|
|
120
|
-
char_code === SEMICOLON ||
|
|
121
|
-
char_code === EQUALS ||
|
|
122
|
-
char_code === COLON ||
|
|
123
|
-
char_code === AT ||
|
|
124
|
-
char_code === SLASH ||
|
|
125
|
-
char_code === QUESTION ||
|
|
126
|
-
char_code === HASH ||
|
|
127
|
-
char_code === PERCENT;
|
|
183
|
+
char_code < 128 && PATH_CHAR_TABLE[char_code] === 1;
|
|
128
184
|
|
|
129
185
|
/**
|
|
130
186
|
* Trim trailing punctuation from URL/path per RFC 3986 and GFM rules.
|
|
131
187
|
* - Trims simple trailing: .,;:!?]
|
|
132
188
|
* - Balanced logic for () only (valid in path components)
|
|
133
|
-
*
|
|
189
|
+
*
|
|
190
|
+
* Note on `]`: the mdz parsers (`mdz.ts`, `mdz_stream_parser_url.ts`,
|
|
191
|
+
* `mdz_lexer.ts`) scan URL/path chars through `is_valid_path_char`, which
|
|
192
|
+
* already rejects `]` — so it can never reach this function via parser flow.
|
|
193
|
+
* The `]` branch here is for external callers using this helper directly on
|
|
194
|
+
* arbitrary URL-ish input (it is part of the published `@fuzdev/fuz_ui`
|
|
195
|
+
* surface, see `mdz_helpers.test.ts` for direct coverage).
|
|
134
196
|
*
|
|
135
197
|
* Optimized to avoid O(n²) string slicing - tracks end index and slices once at the end.
|
|
136
198
|
*/
|
|
137
199
|
export const trim_trailing_punctuation = (url: string): string => {
|
|
138
200
|
let end = url.length;
|
|
139
201
|
|
|
140
|
-
// Trim simple trailing punctuation (] as fallback - whitelist should prevent it)
|
|
141
202
|
while (end > 0) {
|
|
142
203
|
const last_char = url.charCodeAt(end - 1);
|
|
143
204
|
if (
|
|
@@ -156,23 +217,26 @@ export const trim_trailing_punctuation = (url: string): string => {
|
|
|
156
217
|
}
|
|
157
218
|
|
|
158
219
|
// Handle balanced parentheses ONLY (parens are valid in URI path components)
|
|
159
|
-
//
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
220
|
+
// Only count parens when the trimmed URL ends with ')' — otherwise the
|
|
221
|
+
// trailing-paren loop below can't trim anything.
|
|
222
|
+
if (end > 0 && url.charCodeAt(end - 1) === RIGHT_PAREN) {
|
|
223
|
+
let open_count = 0;
|
|
224
|
+
let close_count = 0;
|
|
225
|
+
for (let i = 0; i < end; i++) {
|
|
226
|
+
const char = url.charCodeAt(i);
|
|
227
|
+
if (char === LEFT_PAREN) open_count++;
|
|
228
|
+
if (char === RIGHT_PAREN) close_count++;
|
|
229
|
+
}
|
|
167
230
|
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
231
|
+
// Trim unmatched trailing closing parens
|
|
232
|
+
while (end > 0 && close_count > open_count) {
|
|
233
|
+
const last_char = url.charCodeAt(end - 1);
|
|
234
|
+
if (last_char === RIGHT_PAREN) {
|
|
235
|
+
end--;
|
|
236
|
+
close_count--;
|
|
237
|
+
} else {
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
176
240
|
}
|
|
177
241
|
}
|
|
178
242
|
|
|
@@ -246,14 +310,30 @@ export const mdz_text_content = (nodes: Array<MdzNode>): string =>
|
|
|
246
310
|
export const mdz_heading_id = (nodes: Array<MdzNode>): string =>
|
|
247
311
|
slugify(mdz_text_content(nodes), false);
|
|
248
312
|
|
|
313
|
+
/**
|
|
314
|
+
* Generates a lowercase slug id for a heading from plain text content.
|
|
315
|
+
* Used by the streaming parser which tracks text content directly
|
|
316
|
+
* rather than building `MdzNode[]` trees.
|
|
317
|
+
*/
|
|
318
|
+
export const mdz_heading_id_from_text = (text: string): string => slugify(text, false);
|
|
319
|
+
|
|
249
320
|
/**
|
|
250
321
|
* Check if a string is a URL (`https://` or `http://`).
|
|
251
322
|
* Requires at least one valid character after the protocol.
|
|
252
323
|
* Rejects whitespace and characters that can't start a valid hostname.
|
|
253
324
|
*/
|
|
254
|
-
const URL_PATTERN = /^https?:\/\/[^\s)\]}<>.,:/?#!]
|
|
325
|
+
const URL_PATTERN = /^https?:\/\/[^\s)\]}<>.,:/?#!]/i;
|
|
255
326
|
export const mdz_is_url = (s: string): boolean => URL_PATTERN.test(s);
|
|
256
327
|
|
|
328
|
+
/**
|
|
329
|
+
* Check if a link reference is safe to use as an `href` attribute.
|
|
330
|
+
* References without a colon are always safe (paths, fragments, queries).
|
|
331
|
+
* References with a colon must use `http(s)://` — rejects `javascript:`, `data:`, etc.
|
|
332
|
+
*/
|
|
333
|
+
const SAFE_PROTOCOL_PATTERN = /^https?:\/\//i;
|
|
334
|
+
export const mdz_is_safe_reference = (reference: string): boolean =>
|
|
335
|
+
!reference.includes(':') || SAFE_PROTOCOL_PATTERN.test(reference);
|
|
336
|
+
|
|
257
337
|
/**
|
|
258
338
|
* Resolves a relative path (`./` or `../`) against a base path.
|
|
259
339
|
* The base is treated as a directory regardless of trailing slash
|
|
@@ -284,6 +364,69 @@ export const resolve_relative_path = (reference: string, base: string): string =
|
|
|
284
364
|
return segments.join('/');
|
|
285
365
|
};
|
|
286
366
|
|
|
367
|
+
/**
|
|
368
|
+
* Push a node into a children array, coalescing with the previous Text node.
|
|
369
|
+
* Mutates `prev.content` and `prev.end` when both are Text, avoiding array growth
|
|
370
|
+
* and an extra allocation. Callers must own `dest` and not retain references to
|
|
371
|
+
* the prior last element across the call.
|
|
372
|
+
*/
|
|
373
|
+
export const mdz_push_merging_text = (dest: Array<MdzNode>, node: MdzNode): void => {
|
|
374
|
+
if (node.type === 'Text') {
|
|
375
|
+
const last = dest[dest.length - 1];
|
|
376
|
+
if (last?.type === 'Text') {
|
|
377
|
+
last.content += node.content;
|
|
378
|
+
last.end = node.end;
|
|
379
|
+
return;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
dest.push(node);
|
|
383
|
+
};
|
|
384
|
+
|
|
385
|
+
/**
|
|
386
|
+
* Return a new array with adjacent Text nodes merged into single nodes.
|
|
387
|
+
* Fast path: returns the original array when no merging is needed.
|
|
388
|
+
*/
|
|
389
|
+
export const mdz_merge_adjacent_text = (nodes: Array<MdzNode>): Array<MdzNode> => {
|
|
390
|
+
if (nodes.length <= 1) return nodes;
|
|
391
|
+
|
|
392
|
+
let needs_merge = false;
|
|
393
|
+
for (let i = 1; i < nodes.length; i++) {
|
|
394
|
+
if (nodes[i - 1]!.type === 'Text' && nodes[i]!.type === 'Text') {
|
|
395
|
+
needs_merge = true;
|
|
396
|
+
break;
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
if (!needs_merge) return nodes;
|
|
400
|
+
|
|
401
|
+
const merged: Array<MdzNode> = [];
|
|
402
|
+
let pending: MdzTextNode | null = null;
|
|
403
|
+
|
|
404
|
+
for (const node of nodes) {
|
|
405
|
+
if (node.type === 'Text') {
|
|
406
|
+
if (pending) {
|
|
407
|
+
pending = {
|
|
408
|
+
type: 'Text',
|
|
409
|
+
content: pending.content + node.content,
|
|
410
|
+
start: pending.start,
|
|
411
|
+
end: node.end,
|
|
412
|
+
};
|
|
413
|
+
} else {
|
|
414
|
+
pending = {...node};
|
|
415
|
+
}
|
|
416
|
+
} else {
|
|
417
|
+
if (pending) {
|
|
418
|
+
merged.push(pending);
|
|
419
|
+
pending = null;
|
|
420
|
+
}
|
|
421
|
+
merged.push(node);
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
if (pending) merged.push(pending);
|
|
426
|
+
|
|
427
|
+
return merged;
|
|
428
|
+
};
|
|
429
|
+
|
|
287
430
|
export const extract_single_tag = (
|
|
288
431
|
nodes: Array<MdzNode>,
|
|
289
432
|
): MdzComponentNode | MdzElementNode | null => {
|
package/src/lib/mdz_lexer.ts
CHANGED
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
import {
|
|
11
11
|
mdz_is_url,
|
|
12
|
+
mdz_is_safe_reference,
|
|
12
13
|
is_letter,
|
|
13
14
|
is_tag_name_char,
|
|
14
15
|
is_word_char,
|
|
@@ -36,8 +37,7 @@ import {
|
|
|
36
37
|
HR_HYPHEN_COUNT,
|
|
37
38
|
MIN_CODEBLOCK_BACKTICKS,
|
|
38
39
|
MAX_HEADING_LEVEL,
|
|
39
|
-
|
|
40
|
-
HTTP_PREFIX_LENGTH,
|
|
40
|
+
match_url_prefix_case_insensitive,
|
|
41
41
|
is_at_absolute_path,
|
|
42
42
|
is_at_relative_path,
|
|
43
43
|
} from './mdz_helpers.js';
|
|
@@ -716,6 +716,13 @@ export class MdzLexer {
|
|
|
716
716
|
}
|
|
717
717
|
}
|
|
718
718
|
|
|
719
|
+
// Reject unsafe protocols (javascript:, data:, etc.) — `is_valid_path_char`
|
|
720
|
+
// permits `:` so we need an explicit filter here.
|
|
721
|
+
if (!mdz_is_safe_reference(reference)) {
|
|
722
|
+
this.#revert_tokens_from_link_open(start);
|
|
723
|
+
return;
|
|
724
|
+
}
|
|
725
|
+
|
|
719
726
|
this.#index = close_paren + 1;
|
|
720
727
|
|
|
721
728
|
const link_type = mdz_is_url(reference) ? 'external' : 'internal';
|
|
@@ -893,7 +900,7 @@ export class MdzLexer {
|
|
|
893
900
|
|
|
894
901
|
// Check for URL or internal path mid-text (char code guard avoids startsWith on every char)
|
|
895
902
|
if (
|
|
896
|
-
(char_code === 104 /* h */ && this.#is_at_url()) ||
|
|
903
|
+
((char_code === 104 /* h */ || char_code === 72) /* H */ && this.#is_at_url()) ||
|
|
897
904
|
(char_code === SLASH && is_at_absolute_path(this.#text, this.#index)) ||
|
|
898
905
|
(char_code === PERIOD && is_at_relative_path(this.#text, this.#index))
|
|
899
906
|
) {
|
|
@@ -917,12 +924,8 @@ export class MdzLexer {
|
|
|
917
924
|
#tokenize_auto_link_url(): void {
|
|
918
925
|
const start = this.#index;
|
|
919
926
|
|
|
920
|
-
// Consume protocol
|
|
921
|
-
|
|
922
|
-
this.#index += HTTPS_PREFIX_LENGTH;
|
|
923
|
-
} else if (this.#match('http://')) {
|
|
924
|
-
this.#index += HTTP_PREFIX_LENGTH;
|
|
925
|
-
}
|
|
927
|
+
// Consume protocol (case-insensitive match; original casing preserved in reference)
|
|
928
|
+
this.#index += match_url_prefix_case_insensitive(this.#text, this.#index);
|
|
926
929
|
|
|
927
930
|
// Collect URL characters
|
|
928
931
|
while (this.#index < this.#text.length) {
|
|
@@ -987,17 +990,12 @@ export class MdzLexer {
|
|
|
987
990
|
}
|
|
988
991
|
|
|
989
992
|
#is_at_url(): boolean {
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
if (this.#index + HTTP_PREFIX_LENGTH >= this.#text.length) return false;
|
|
997
|
-
const next_char = this.#text.charCodeAt(this.#index + HTTP_PREFIX_LENGTH);
|
|
998
|
-
return next_char !== SPACE && next_char !== NEWLINE;
|
|
999
|
-
}
|
|
1000
|
-
return false;
|
|
993
|
+
// Scheme matching is case-insensitive (RFC 3986).
|
|
994
|
+
const prefix_len = match_url_prefix_case_insensitive(this.#text, this.#index);
|
|
995
|
+
if (prefix_len === 0) return false;
|
|
996
|
+
if (this.#index + prefix_len >= this.#text.length) return false;
|
|
997
|
+
const next_char = this.#text.charCodeAt(this.#index + prefix_len);
|
|
998
|
+
return next_char !== SPACE && next_char !== NEWLINE;
|
|
1001
999
|
}
|
|
1002
1000
|
|
|
1003
1001
|
#is_at_word_boundary(index: number, check_before: boolean, check_after: boolean): boolean {
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Opcode types for the mdz streaming parser.
|
|
3
|
+
*
|
|
4
|
+
* Opcodes are serializable rendering instructions emitted by `MdzStreamParser`.
|
|
5
|
+
* They tell a renderer what to do next — open a container, append text, close it,
|
|
6
|
+
* or revert an optimistic assumption. Target-agnostic: works for HTML, Svelte, PDF, etc.
|
|
7
|
+
*
|
|
8
|
+
* The parser makes optimistic assumptions about ambiguous syntax (e.g., `**` is probably bold)
|
|
9
|
+
* and emits `revert` opcodes to correct when wrong. This enables true streaming rendering
|
|
10
|
+
* without ever re-parsing.
|
|
11
|
+
*
|
|
12
|
+
* @module
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Unique monotonic identifier for each node created by the parser.
|
|
17
|
+
* IDs are never reused within a parser instance.
|
|
18
|
+
*/
|
|
19
|
+
export type MdzNodeId = number;
|
|
20
|
+
|
|
21
|
+
/** Node types that can be opened as containers. */
|
|
22
|
+
export type MdzContainerNodeType =
|
|
23
|
+
| 'Paragraph'
|
|
24
|
+
| 'Bold'
|
|
25
|
+
| 'Italic'
|
|
26
|
+
| 'Strikethrough'
|
|
27
|
+
| 'Link'
|
|
28
|
+
| 'Heading'
|
|
29
|
+
| 'Element'
|
|
30
|
+
| 'Component'
|
|
31
|
+
| 'Codeblock'
|
|
32
|
+
| 'Code';
|
|
33
|
+
|
|
34
|
+
/** Node types for self-contained leaf elements. */
|
|
35
|
+
export type MdzVoidNodeType = 'Hr';
|
|
36
|
+
|
|
37
|
+
/** Discriminant for leaf text nodes. */
|
|
38
|
+
export type MdzTextNodeType = 'Text' | 'Code';
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Open a container node. The renderer starts a new element/wrapper.
|
|
42
|
+
* Children are subsequent opcodes until the matching `close`.
|
|
43
|
+
*/
|
|
44
|
+
export interface MdzOpcodeOpen {
|
|
45
|
+
type: 'open';
|
|
46
|
+
id: MdzNodeId;
|
|
47
|
+
node_type: MdzContainerNodeType;
|
|
48
|
+
/** Byte offset in the full input where the opening delimiter begins. */
|
|
49
|
+
start: number;
|
|
50
|
+
/** Heading level (1-6). Present when `node_type` is `'Heading'`. */
|
|
51
|
+
level?: 1 | 2 | 3 | 4 | 5 | 6;
|
|
52
|
+
/** Tag name. Present when `node_type` is `'Element'` or `'Component'`. */
|
|
53
|
+
name?: string;
|
|
54
|
+
/** Language hint. Present when `node_type` is `'Codeblock'`. */
|
|
55
|
+
lang?: string | null;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Close a previously opened container node.
|
|
60
|
+
* Carries deferred metadata that wasn't known at open time.
|
|
61
|
+
*/
|
|
62
|
+
export interface MdzOpcodeClose {
|
|
63
|
+
type: 'close';
|
|
64
|
+
id: MdzNodeId;
|
|
65
|
+
/** Byte offset in the full input immediately after the closing delimiter. */
|
|
66
|
+
end: number;
|
|
67
|
+
/** Link URL/path, resolved when `](url)` completes. */
|
|
68
|
+
reference?: string;
|
|
69
|
+
/** Link type, resolved alongside `reference`. */
|
|
70
|
+
link_type?: 'external' | 'internal';
|
|
71
|
+
/** Heading slug, computed from full heading content. */
|
|
72
|
+
heading_id?: string;
|
|
73
|
+
/**
|
|
74
|
+
* If true, consumer drops this node and its descendants from the tree.
|
|
75
|
+
* Used for whitespace-only paragraphs that match nothing in `mdz_parse`'s
|
|
76
|
+
* output — the streaming parser emits open/text speculatively, then
|
|
77
|
+
* retroactively drops the empty wrapper at close.
|
|
78
|
+
*/
|
|
79
|
+
discard?: boolean;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Create a leaf text or code node.
|
|
84
|
+
* The parent is implicit — the innermost open container on the renderer's stack.
|
|
85
|
+
*/
|
|
86
|
+
export interface MdzOpcodeText {
|
|
87
|
+
type: 'text';
|
|
88
|
+
id: MdzNodeId;
|
|
89
|
+
content: string;
|
|
90
|
+
text_type: MdzTextNodeType;
|
|
91
|
+
/** Byte offset where this node begins (for Code, the opening backtick). */
|
|
92
|
+
start: number;
|
|
93
|
+
/** Byte offset immediately after this node ends (for Code, after the closing backtick). */
|
|
94
|
+
end: number;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Append content to an existing text node.
|
|
99
|
+
* Streaming optimization — avoids creating a new node per chunk
|
|
100
|
+
* during plain text runs.
|
|
101
|
+
*/
|
|
102
|
+
export interface MdzOpcodeAppendText {
|
|
103
|
+
type: 'append_text';
|
|
104
|
+
id: MdzNodeId;
|
|
105
|
+
content: string;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Trim `count` characters from the end of an existing text node.
|
|
110
|
+
* If trimming empties the node, the consumer removes it from its parent.
|
|
111
|
+
*
|
|
112
|
+
* Used by paragraph/codeblock close to drop the trailing newline that
|
|
113
|
+
* separates inline content from the block boundary. Emitted instead of
|
|
114
|
+
* retroactively mutating the prior `text`/`append_text` opcode, so the
|
|
115
|
+
* opcode stream is append-only.
|
|
116
|
+
*/
|
|
117
|
+
export interface MdzOpcodeTrimText {
|
|
118
|
+
type: 'trim_text';
|
|
119
|
+
id: MdzNodeId;
|
|
120
|
+
count: number;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Create a self-contained leaf node (e.g., horizontal rule).
|
|
125
|
+
* Inserted as a child of the innermost open container, or at root level.
|
|
126
|
+
*/
|
|
127
|
+
export interface MdzOpcodeVoid {
|
|
128
|
+
type: 'void';
|
|
129
|
+
id: MdzNodeId;
|
|
130
|
+
node_type: MdzVoidNodeType;
|
|
131
|
+
/** Byte offset in the full input where this element begins. */
|
|
132
|
+
start: number;
|
|
133
|
+
/** Byte offset immediately after this element ends. */
|
|
134
|
+
end: number;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Undo an optimistic open. Removes the container wrapper,
|
|
139
|
+
* inserts `replacement_text` as literal text at the container's position,
|
|
140
|
+
* and re-parents the container's children to the grandparent.
|
|
141
|
+
*
|
|
142
|
+
* When `wrap_node_type` and `wrap_id` are set, the replacement text and
|
|
143
|
+
* re-parented children are wrapped in a new container of the given type
|
|
144
|
+
* instead of being placed directly at the grandparent level. The wrapper
|
|
145
|
+
* is pushed onto the consumer's stack (open for future content). This is
|
|
146
|
+
* used for block-level reverts (e.g., codeblock → paragraph) where the
|
|
147
|
+
* grandparent is root and content needs a container.
|
|
148
|
+
*/
|
|
149
|
+
export interface MdzOpcodeRevert {
|
|
150
|
+
type: 'revert';
|
|
151
|
+
id: MdzNodeId;
|
|
152
|
+
/** The delimiter text to emit as literal content (e.g., `"**"`, `"["`, `"<Tag>"`). */
|
|
153
|
+
replacement_text: string;
|
|
154
|
+
/** Byte offset of the original opening delimiter in the full input. */
|
|
155
|
+
start: number;
|
|
156
|
+
/** Wrap replacement text and re-parented children in a new container of this type. */
|
|
157
|
+
wrap_node_type?: MdzContainerNodeType;
|
|
158
|
+
/** ID for the wrapper node. Required when `wrap_node_type` is set. */
|
|
159
|
+
wrap_id?: MdzNodeId;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Retroactively wrap an existing text node in a container.
|
|
164
|
+
* Used for text-first auto-links: URL/path text streams as plain text,
|
|
165
|
+
* then gets wrapped in a Link when the URL boundary is found.
|
|
166
|
+
*
|
|
167
|
+
* When `trim_end` is set, trailing characters (punctuation) are trimmed
|
|
168
|
+
* from the target text node and placed in a new sibling Text node after
|
|
169
|
+
* the Link wrapper, identified by `trim_id`.
|
|
170
|
+
*/
|
|
171
|
+
export interface MdzOpcodeWrap {
|
|
172
|
+
type: 'wrap';
|
|
173
|
+
/** ID for the new Link container node. */
|
|
174
|
+
id: MdzNodeId;
|
|
175
|
+
/** Container type to wrap in (always `'Link'` for now). */
|
|
176
|
+
node_type: 'Link';
|
|
177
|
+
/** ID of the existing text node to wrap. */
|
|
178
|
+
target_id: MdzNodeId;
|
|
179
|
+
/** Resolved URL or path reference. */
|
|
180
|
+
reference: string;
|
|
181
|
+
/** Whether the link is external (URL) or internal (path). */
|
|
182
|
+
link_type: 'external' | 'internal';
|
|
183
|
+
/** Byte offset where the URL/path begins. */
|
|
184
|
+
start: number;
|
|
185
|
+
/** Byte offset immediately after the URL/path (before any trimmed punctuation). */
|
|
186
|
+
end: number;
|
|
187
|
+
/** Number of trailing chars to trim from target and place after the link. */
|
|
188
|
+
trim_end?: number;
|
|
189
|
+
/** ID for the trimmed-text sibling node. Required when `trim_end` > 0. */
|
|
190
|
+
trim_id?: MdzNodeId;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/** All node types that can appear in the mdz tree. */
|
|
194
|
+
export type MdzNodeType = MdzContainerNodeType | MdzVoidNodeType | MdzTextNodeType;
|
|
195
|
+
|
|
196
|
+
/** Discriminated union of all mdz opcodes. */
|
|
197
|
+
export type MdzOpcode =
|
|
198
|
+
| MdzOpcodeOpen
|
|
199
|
+
| MdzOpcodeClose
|
|
200
|
+
| MdzOpcodeText
|
|
201
|
+
| MdzOpcodeAppendText
|
|
202
|
+
| MdzOpcodeTrimText
|
|
203
|
+
| MdzOpcodeVoid
|
|
204
|
+
| MdzOpcodeRevert
|
|
205
|
+
| MdzOpcodeWrap;
|