@localess/richtext 4.0.0-dev.20260905085556 → 4.0.0-dev.20260906124410
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/SKILL.md +35 -0
- package/dist/html-parser/index.d.ts +29 -0
- package/dist/html-parser/index.js +491 -0
- package/dist/html-parser/index.mjs +489 -0
- package/dist/html-parser/to-model.d.ts +11 -0
- package/dist/html-parser/tokenizer.d.ts +32 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +11 -44
- package/dist/index.mjs +2 -38
- package/dist/markdown-parser/block.d.ts +11 -0
- package/dist/markdown-parser/index.d.ts +32 -0
- package/dist/markdown-parser/index.js +442 -0
- package/dist/markdown-parser/index.mjs +440 -0
- package/dist/markdown-parser/inline.d.ts +13 -0
- package/dist/parse-common-C-dAtFji.mjs +100 -0
- package/dist/parse-common-DEYhfKnK.js +135 -0
- package/dist/parse-common.d.ts +71 -0
- package/package.json +15 -2
|
@@ -0,0 +1,489 @@
|
|
|
1
|
+
import { n as UnsupportedTracker, o as sanitizeUrl, r as emptyDocument, t as RichTextParseError } from "../parse-common-C-dAtFji.mjs";
|
|
2
|
+
//#region src/html-parser/to-model.ts
|
|
3
|
+
/** Tag → mark, the inverse of `MARK_RENDER_MAP`. `b`/`i`/`del`/`s` are accepted as aliases. */
|
|
4
|
+
var MARK_TAGS = {
|
|
5
|
+
strong: "bold",
|
|
6
|
+
b: "bold",
|
|
7
|
+
em: "italic",
|
|
8
|
+
i: "italic",
|
|
9
|
+
s: "strike",
|
|
10
|
+
strike: "strike",
|
|
11
|
+
del: "strike",
|
|
12
|
+
u: "underline",
|
|
13
|
+
code: "code",
|
|
14
|
+
a: "link"
|
|
15
|
+
};
|
|
16
|
+
var HEADING_TAGS = {
|
|
17
|
+
h1: 1,
|
|
18
|
+
h2: 2,
|
|
19
|
+
h3: 3,
|
|
20
|
+
h4: 4,
|
|
21
|
+
h5: 5,
|
|
22
|
+
h6: 6
|
|
23
|
+
};
|
|
24
|
+
/** Block tags the model represents directly. */
|
|
25
|
+
var BLOCK_TAGS = /* @__PURE__ */ new Set([
|
|
26
|
+
"p",
|
|
27
|
+
"ul",
|
|
28
|
+
"ol",
|
|
29
|
+
"li",
|
|
30
|
+
"pre",
|
|
31
|
+
...Object.keys(HEADING_TAGS)
|
|
32
|
+
]);
|
|
33
|
+
/** Tags that carry no meaning of their own — their children are simply kept. */
|
|
34
|
+
var TRANSPARENT_TAGS = /* @__PURE__ */ new Set([
|
|
35
|
+
"html",
|
|
36
|
+
"body",
|
|
37
|
+
"head",
|
|
38
|
+
"div",
|
|
39
|
+
"section",
|
|
40
|
+
"article",
|
|
41
|
+
"main",
|
|
42
|
+
"span",
|
|
43
|
+
"font",
|
|
44
|
+
"tbody",
|
|
45
|
+
"thead"
|
|
46
|
+
]);
|
|
47
|
+
/** Tags dropped entirely, content and all, regardless of policy. */
|
|
48
|
+
var DROPPED_TAGS = /* @__PURE__ */ new Set([
|
|
49
|
+
"script",
|
|
50
|
+
"style",
|
|
51
|
+
"title",
|
|
52
|
+
"meta",
|
|
53
|
+
"link",
|
|
54
|
+
"base",
|
|
55
|
+
"noscript"
|
|
56
|
+
]);
|
|
57
|
+
function linkMark(attrs) {
|
|
58
|
+
return {
|
|
59
|
+
type: "link",
|
|
60
|
+
attrs: {
|
|
61
|
+
href: sanitizeUrl(attrs.href ?? ""),
|
|
62
|
+
target: attrs.target ?? null,
|
|
63
|
+
rel: attrs.rel ?? null,
|
|
64
|
+
class: attrs.class ?? null
|
|
65
|
+
}
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
function codeBlockLanguage(attrs) {
|
|
69
|
+
const className = attrs.class ?? "";
|
|
70
|
+
const match = /(?:^|\s)language-([^\s]+)/.exec(className);
|
|
71
|
+
return match ? match[1] : null;
|
|
72
|
+
}
|
|
73
|
+
/** Collapses HTML whitespace the way a browser would for non-preformatted text. */
|
|
74
|
+
function collapseWhitespace(text) {
|
|
75
|
+
return text.replace(/[\t\n\r ]+/g, " ");
|
|
76
|
+
}
|
|
77
|
+
function isBlockNode(node) {
|
|
78
|
+
return node.type !== "text";
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Folds a token stream into model nodes.
|
|
82
|
+
*
|
|
83
|
+
* Text is only kept where the model can hold it — inside a block node. Text
|
|
84
|
+
* found at the top level is wrapped in a paragraph, matching what the editor
|
|
85
|
+
* would produce.
|
|
86
|
+
*/
|
|
87
|
+
function tokensToNodes(tokens, tracker) {
|
|
88
|
+
const root = {
|
|
89
|
+
tag: "",
|
|
90
|
+
node: null,
|
|
91
|
+
children: [],
|
|
92
|
+
marks: [],
|
|
93
|
+
preformatted: false
|
|
94
|
+
};
|
|
95
|
+
const stack = [root];
|
|
96
|
+
const top = () => stack[stack.length - 1];
|
|
97
|
+
/** Blocks skipped wholesale; text inside them is discarded. */
|
|
98
|
+
let skipDepth = 0;
|
|
99
|
+
let skipTag = "";
|
|
100
|
+
const appendText = (text) => {
|
|
101
|
+
const frame = top();
|
|
102
|
+
if (text === "") return;
|
|
103
|
+
const node = {
|
|
104
|
+
type: "text",
|
|
105
|
+
text,
|
|
106
|
+
...frame.marks.length ? { marks: [...frame.marks] } : {}
|
|
107
|
+
};
|
|
108
|
+
if (frame === root) {
|
|
109
|
+
const last = root.children[root.children.length - 1];
|
|
110
|
+
if (last && last.type === "paragraph") (last.content ??= []).push(node);
|
|
111
|
+
else root.children.push({
|
|
112
|
+
type: "paragraph",
|
|
113
|
+
content: [node]
|
|
114
|
+
});
|
|
115
|
+
return;
|
|
116
|
+
}
|
|
117
|
+
frame.children.push(node);
|
|
118
|
+
};
|
|
119
|
+
const closeFrame = () => {
|
|
120
|
+
const frame = stack.pop();
|
|
121
|
+
const parent = top();
|
|
122
|
+
if (frame.node) {
|
|
123
|
+
if (frame.children.length) frame.node.content = frame.children;
|
|
124
|
+
parent.children.push(frame.node);
|
|
125
|
+
} else parent.children.push(...frame.children);
|
|
126
|
+
};
|
|
127
|
+
for (const token of tokens) {
|
|
128
|
+
if (skipDepth > 0) {
|
|
129
|
+
if (token.kind === "open" && token.name === skipTag && !token.selfClosing) skipDepth++;
|
|
130
|
+
else if (token.kind === "close" && token.name === skipTag) skipDepth--;
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
if (token.kind === "text") {
|
|
134
|
+
const frame = top();
|
|
135
|
+
const text = frame.preformatted ? token.text : collapseWhitespace(token.text);
|
|
136
|
+
if (!frame.preformatted && text.trim() === "" && (frame === root || frame.children.every(isBlockNode))) continue;
|
|
137
|
+
appendText(text);
|
|
138
|
+
continue;
|
|
139
|
+
}
|
|
140
|
+
if (token.kind === "close") {
|
|
141
|
+
const depth = stack.findIndex((frame) => frame.tag === token.name);
|
|
142
|
+
if (depth <= 0) continue;
|
|
143
|
+
while (stack.length > depth) closeFrame();
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
const { name, attrs, selfClosing } = token;
|
|
147
|
+
if (DROPPED_TAGS.has(name)) {
|
|
148
|
+
if (!selfClosing) {
|
|
149
|
+
skipDepth = 1;
|
|
150
|
+
skipTag = name;
|
|
151
|
+
}
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
if (name === "br") {
|
|
155
|
+
appendText(" ");
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
const markType = MARK_TAGS[name];
|
|
159
|
+
if (markType) {
|
|
160
|
+
const frame = top();
|
|
161
|
+
if (name === "code" && frame.tag === "pre" && frame.node?.type === "codeBlock") {
|
|
162
|
+
const language = codeBlockLanguage(attrs);
|
|
163
|
+
if (language) frame.node.attrs = { language };
|
|
164
|
+
if (selfClosing) continue;
|
|
165
|
+
stack.push({
|
|
166
|
+
tag: name,
|
|
167
|
+
node: null,
|
|
168
|
+
children: [],
|
|
169
|
+
marks: [...frame.marks],
|
|
170
|
+
preformatted: true
|
|
171
|
+
});
|
|
172
|
+
continue;
|
|
173
|
+
}
|
|
174
|
+
const mark = markType === "link" ? linkMark(attrs) : { type: markType };
|
|
175
|
+
if (selfClosing) continue;
|
|
176
|
+
stack.push({
|
|
177
|
+
tag: name,
|
|
178
|
+
node: null,
|
|
179
|
+
children: [],
|
|
180
|
+
marks: [...frame.marks, mark],
|
|
181
|
+
preformatted: frame.preformatted
|
|
182
|
+
});
|
|
183
|
+
continue;
|
|
184
|
+
}
|
|
185
|
+
if (TRANSPARENT_TAGS.has(name)) {
|
|
186
|
+
if (selfClosing) continue;
|
|
187
|
+
const frame = top();
|
|
188
|
+
stack.push({
|
|
189
|
+
tag: name,
|
|
190
|
+
node: null,
|
|
191
|
+
children: [],
|
|
192
|
+
marks: [...frame.marks],
|
|
193
|
+
preformatted: frame.preformatted
|
|
194
|
+
});
|
|
195
|
+
continue;
|
|
196
|
+
}
|
|
197
|
+
if (BLOCK_TAGS.has(name)) {
|
|
198
|
+
const frame = top();
|
|
199
|
+
let node;
|
|
200
|
+
let preformatted = frame.preformatted;
|
|
201
|
+
if (name === "p") node = { type: "paragraph" };
|
|
202
|
+
else if (name in HEADING_TAGS) node = {
|
|
203
|
+
type: "heading",
|
|
204
|
+
attrs: { level: HEADING_TAGS[name] }
|
|
205
|
+
};
|
|
206
|
+
else if (name === "ul") node = { type: "bulletList" };
|
|
207
|
+
else if (name === "ol") {
|
|
208
|
+
const start = Number.parseInt(attrs.start ?? "", 10);
|
|
209
|
+
node = Number.isFinite(start) && start !== 1 ? {
|
|
210
|
+
type: "orderedList",
|
|
211
|
+
attrs: { start }
|
|
212
|
+
} : { type: "orderedList" };
|
|
213
|
+
} else if (name === "li") node = { type: "listItem" };
|
|
214
|
+
else {
|
|
215
|
+
node = { type: "codeBlock" };
|
|
216
|
+
preformatted = true;
|
|
217
|
+
}
|
|
218
|
+
if (selfClosing) {
|
|
219
|
+
frame.children.push(node);
|
|
220
|
+
continue;
|
|
221
|
+
}
|
|
222
|
+
stack.push({
|
|
223
|
+
tag: name,
|
|
224
|
+
node,
|
|
225
|
+
children: [],
|
|
226
|
+
marks: [...frame.marks],
|
|
227
|
+
preformatted
|
|
228
|
+
});
|
|
229
|
+
continue;
|
|
230
|
+
}
|
|
231
|
+
const action = tracker.record(name);
|
|
232
|
+
if (selfClosing) continue;
|
|
233
|
+
if (action === "skip") {
|
|
234
|
+
skipDepth = 1;
|
|
235
|
+
skipTag = name;
|
|
236
|
+
continue;
|
|
237
|
+
}
|
|
238
|
+
const frame = top();
|
|
239
|
+
stack.push({
|
|
240
|
+
tag: name,
|
|
241
|
+
node: null,
|
|
242
|
+
children: [],
|
|
243
|
+
marks: [...frame.marks],
|
|
244
|
+
preformatted: frame.preformatted
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
while (stack.length > 1) closeFrame();
|
|
248
|
+
return finalize(root.children);
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* Post-pass matching the model's shape rules:
|
|
252
|
+
* `<pre>` wraps a `<code>` in the renderer, so the parser lifts that `code`
|
|
253
|
+
* mark back onto the `codeBlock`'s `language`, and list items always hold
|
|
254
|
+
* blocks rather than bare text.
|
|
255
|
+
*/
|
|
256
|
+
function finalize(nodes) {
|
|
257
|
+
return nodes.map((node) => {
|
|
258
|
+
if (node.type === "text") return node;
|
|
259
|
+
const content = node.content;
|
|
260
|
+
if (!content) return node;
|
|
261
|
+
if (node.type === "codeBlock") {
|
|
262
|
+
const text = flattenText(content);
|
|
263
|
+
const next = {
|
|
264
|
+
type: "codeBlock",
|
|
265
|
+
...node.attrs ? { attrs: node.attrs } : {}
|
|
266
|
+
};
|
|
267
|
+
if (text !== "") next.content = [{
|
|
268
|
+
type: "text",
|
|
269
|
+
text
|
|
270
|
+
}];
|
|
271
|
+
return next;
|
|
272
|
+
}
|
|
273
|
+
const finalized = finalize(content);
|
|
274
|
+
if (node.type === "listItem" && finalized.some((child) => child.type === "text")) {
|
|
275
|
+
const wrapped = [];
|
|
276
|
+
let run = [];
|
|
277
|
+
for (const child of finalized) if (child.type === "text") run.push(child);
|
|
278
|
+
else {
|
|
279
|
+
if (run.length) {
|
|
280
|
+
wrapped.push({
|
|
281
|
+
type: "paragraph",
|
|
282
|
+
content: run
|
|
283
|
+
});
|
|
284
|
+
run = [];
|
|
285
|
+
}
|
|
286
|
+
wrapped.push(child);
|
|
287
|
+
}
|
|
288
|
+
if (run.length) wrapped.push({
|
|
289
|
+
type: "paragraph",
|
|
290
|
+
content: run
|
|
291
|
+
});
|
|
292
|
+
return {
|
|
293
|
+
...node,
|
|
294
|
+
content: wrapped
|
|
295
|
+
};
|
|
296
|
+
}
|
|
297
|
+
return {
|
|
298
|
+
...node,
|
|
299
|
+
content: finalized
|
|
300
|
+
};
|
|
301
|
+
});
|
|
302
|
+
}
|
|
303
|
+
function flattenText(nodes) {
|
|
304
|
+
let out = "";
|
|
305
|
+
for (const node of nodes) if (node.type === "text") out += node.text;
|
|
306
|
+
else out += flattenText(node.content ?? []);
|
|
307
|
+
return out;
|
|
308
|
+
}
|
|
309
|
+
//#endregion
|
|
310
|
+
//#region src/html-parser/tokenizer.ts
|
|
311
|
+
/** Elements that never have a closing tag. */
|
|
312
|
+
var VOID_ELEMENTS = /* @__PURE__ */ new Set([
|
|
313
|
+
"area",
|
|
314
|
+
"base",
|
|
315
|
+
"br",
|
|
316
|
+
"col",
|
|
317
|
+
"embed",
|
|
318
|
+
"hr",
|
|
319
|
+
"img",
|
|
320
|
+
"input",
|
|
321
|
+
"link",
|
|
322
|
+
"meta",
|
|
323
|
+
"param",
|
|
324
|
+
"source",
|
|
325
|
+
"track",
|
|
326
|
+
"wbr"
|
|
327
|
+
]);
|
|
328
|
+
/** Elements whose content is raw text, not markup. */
|
|
329
|
+
var RAW_TEXT_ELEMENTS = /* @__PURE__ */ new Set(["script", "style"]);
|
|
330
|
+
var NAMED_ENTITIES = {
|
|
331
|
+
amp: "&",
|
|
332
|
+
lt: "<",
|
|
333
|
+
gt: ">",
|
|
334
|
+
quot: "\"",
|
|
335
|
+
apos: "'",
|
|
336
|
+
nbsp: "\xA0"
|
|
337
|
+
};
|
|
338
|
+
/**
|
|
339
|
+
* Decodes the named and numeric character references the renderer can emit,
|
|
340
|
+
* plus the few that appear in hand-written HTML. Unknown references are left
|
|
341
|
+
* verbatim rather than dropped, so no text is silently lost.
|
|
342
|
+
*/
|
|
343
|
+
function decodeEntities(text) {
|
|
344
|
+
if (!text.includes("&")) return text;
|
|
345
|
+
return text.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (match, body) => {
|
|
346
|
+
if (body[0] === "#") {
|
|
347
|
+
const codePoint = body[1] === "x" || body[1] === "X" ? Number.parseInt(body.slice(2), 16) : Number.parseInt(body.slice(1), 10);
|
|
348
|
+
if (!Number.isFinite(codePoint) || codePoint < 0 || codePoint > 1114111) return match;
|
|
349
|
+
try {
|
|
350
|
+
return String.fromCodePoint(codePoint);
|
|
351
|
+
} catch {
|
|
352
|
+
return match;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
return NAMED_ENTITIES[body.toLowerCase()] ?? match;
|
|
356
|
+
});
|
|
357
|
+
}
|
|
358
|
+
function parseAttributes(source) {
|
|
359
|
+
const attrs = {};
|
|
360
|
+
const pattern = /([^\s=/>]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
|
|
361
|
+
let match;
|
|
362
|
+
while ((match = pattern.exec(source)) !== null) {
|
|
363
|
+
const name = match[1].toLowerCase();
|
|
364
|
+
const value = match[2] ?? match[3] ?? match[4] ?? "";
|
|
365
|
+
if (!(name in attrs)) attrs[name] = decodeEntities(value);
|
|
366
|
+
}
|
|
367
|
+
return attrs;
|
|
368
|
+
}
|
|
369
|
+
/**
|
|
370
|
+
* Tokenizes HTML into a flat token stream.
|
|
371
|
+
*
|
|
372
|
+
* A deliberate **subset** tokenizer, not an HTML5-conformant one: it handles
|
|
373
|
+
* the tags the Localess model can represent, treats everything else as a
|
|
374
|
+
* generic element for the caller's unsupported policy, and performs no error
|
|
375
|
+
* recovery beyond never throwing. Chosen over branching on `DOMParser` so the
|
|
376
|
+
* result is identical in browsers, Node, and edge runtimes, and so the package
|
|
377
|
+
* keeps its zero-dependency guarantee.
|
|
378
|
+
*
|
|
379
|
+
* A stray `<` that does not begin a valid tag is emitted as text.
|
|
380
|
+
*/
|
|
381
|
+
function tokenizeHtml(html) {
|
|
382
|
+
const tokens = [];
|
|
383
|
+
let index = 0;
|
|
384
|
+
let pendingText = "";
|
|
385
|
+
const flushText = () => {
|
|
386
|
+
if (pendingText === "") return;
|
|
387
|
+
tokens.push({
|
|
388
|
+
kind: "text",
|
|
389
|
+
text: decodeEntities(pendingText)
|
|
390
|
+
});
|
|
391
|
+
pendingText = "";
|
|
392
|
+
};
|
|
393
|
+
while (index < html.length) {
|
|
394
|
+
const next = html.indexOf("<", index);
|
|
395
|
+
if (next === -1) {
|
|
396
|
+
pendingText += html.slice(index);
|
|
397
|
+
break;
|
|
398
|
+
}
|
|
399
|
+
pendingText += html.slice(index, next);
|
|
400
|
+
const rest = html.slice(next);
|
|
401
|
+
if (rest.startsWith("<!--")) {
|
|
402
|
+
const end = html.indexOf("-->", next + 4);
|
|
403
|
+
index = end === -1 ? html.length : end + 3;
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
if (rest.startsWith("<!") || rest.startsWith("<?")) {
|
|
407
|
+
const end = html.indexOf(">", next + 1);
|
|
408
|
+
index = end === -1 ? html.length : end + 1;
|
|
409
|
+
continue;
|
|
410
|
+
}
|
|
411
|
+
const closeMatch = /^<\/\s*([a-zA-Z][^\s>]*)\s*>/.exec(rest);
|
|
412
|
+
if (closeMatch) {
|
|
413
|
+
flushText();
|
|
414
|
+
tokens.push({
|
|
415
|
+
kind: "close",
|
|
416
|
+
name: closeMatch[1].toLowerCase()
|
|
417
|
+
});
|
|
418
|
+
index = next + closeMatch[0].length;
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
const openMatch = /^<([a-zA-Z][^\s/>]*)((?:[^>"']|"[^"]*"|'[^']*')*)>/.exec(rest);
|
|
422
|
+
if (openMatch) {
|
|
423
|
+
flushText();
|
|
424
|
+
const name = openMatch[1].toLowerCase();
|
|
425
|
+
const raw = openMatch[2] ?? "";
|
|
426
|
+
const selfClosing = /\/\s*$/.test(raw) || VOID_ELEMENTS.has(name);
|
|
427
|
+
tokens.push({
|
|
428
|
+
kind: "open",
|
|
429
|
+
name,
|
|
430
|
+
attrs: parseAttributes(raw),
|
|
431
|
+
selfClosing
|
|
432
|
+
});
|
|
433
|
+
index = next + openMatch[0].length;
|
|
434
|
+
if (RAW_TEXT_ELEMENTS.has(name) && !selfClosing) {
|
|
435
|
+
const closeTag = `</${name}`;
|
|
436
|
+
const end = html.toLowerCase().indexOf(closeTag, index);
|
|
437
|
+
index = end === -1 ? html.length : end;
|
|
438
|
+
}
|
|
439
|
+
continue;
|
|
440
|
+
}
|
|
441
|
+
pendingText += "<";
|
|
442
|
+
index = next + 1;
|
|
443
|
+
}
|
|
444
|
+
flushText();
|
|
445
|
+
return tokens;
|
|
446
|
+
}
|
|
447
|
+
//#endregion
|
|
448
|
+
//#region src/html-parser/index.ts
|
|
449
|
+
/**
|
|
450
|
+
* Parses an HTML string into a Localess rich text document.
|
|
451
|
+
*
|
|
452
|
+
* The inverse of `renderRichTextToHtml`, and deliberately a **subset** parser:
|
|
453
|
+
* it understands the tags the Localess model can represent and routes
|
|
454
|
+
* everything else through {@link RichTextParseOptions.unsupported}. It is not
|
|
455
|
+
* HTML5-conformant and does not attempt full error recovery — the trade is
|
|
456
|
+
* identical behaviour across browsers, Node, and edge runtimes with no
|
|
457
|
+
* dependency.
|
|
458
|
+
*
|
|
459
|
+
* Supported: `p`, `h1`–`h6`, `ul`, `ol` (with `start`), `li`, `pre`/`code`
|
|
460
|
+
* blocks (with `language-*`), and the marks `strong`/`b`, `em`/`i`,
|
|
461
|
+
* `s`/`strike`/`del`, `u`, `code`, `a`. Structural wrappers such as `div` and
|
|
462
|
+
* `span` are transparent; `script` and `style` are always dropped.
|
|
463
|
+
*
|
|
464
|
+
* Link hrefs pass through the same allowlist the renderer applies, so
|
|
465
|
+
* `javascript:` and `data:` become `""`.
|
|
466
|
+
*
|
|
467
|
+
* Never throws for malformed input — except under `unsupported: 'throw'`.
|
|
468
|
+
*
|
|
469
|
+
* @example
|
|
470
|
+
* ```ts
|
|
471
|
+
* const { doc, unsupported } = parseHtmlToRichText('<p>Hello <strong>world</strong></p>');
|
|
472
|
+
* ```
|
|
473
|
+
*/
|
|
474
|
+
function parseHtmlToRichText(html, options = {}) {
|
|
475
|
+
const tracker = new UnsupportedTracker(options);
|
|
476
|
+
if (typeof html !== "string" || html === "") return {
|
|
477
|
+
doc: emptyDocument(),
|
|
478
|
+
unsupported: []
|
|
479
|
+
};
|
|
480
|
+
return {
|
|
481
|
+
doc: {
|
|
482
|
+
type: "doc",
|
|
483
|
+
content: tokensToNodes(tokenizeHtml(html), tracker)
|
|
484
|
+
},
|
|
485
|
+
unsupported: tracker.report()
|
|
486
|
+
};
|
|
487
|
+
}
|
|
488
|
+
//#endregion
|
|
489
|
+
export { RichTextParseError, parseHtmlToRichText };
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { LocalessRichTextNode } from '../model';
|
|
2
|
+
import { UnsupportedTracker } from '../parse-common';
|
|
3
|
+
import { HtmlToken } from './tokenizer';
|
|
4
|
+
/**
|
|
5
|
+
* Folds a token stream into model nodes.
|
|
6
|
+
*
|
|
7
|
+
* Text is only kept where the model can hold it — inside a block node. Text
|
|
8
|
+
* found at the top level is wrapped in a paragraph, matching what the editor
|
|
9
|
+
* would produce.
|
|
10
|
+
*/
|
|
11
|
+
export declare function tokensToNodes(tokens: HtmlToken[], tracker: UnsupportedTracker): LocalessRichTextNode[];
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/** A token produced by {@link tokenizeHtml}. */
|
|
2
|
+
export type HtmlToken = {
|
|
3
|
+
kind: 'open';
|
|
4
|
+
name: string;
|
|
5
|
+
attrs: Record<string, string>;
|
|
6
|
+
selfClosing: boolean;
|
|
7
|
+
} | {
|
|
8
|
+
kind: 'close';
|
|
9
|
+
name: string;
|
|
10
|
+
} | {
|
|
11
|
+
kind: 'text';
|
|
12
|
+
text: string;
|
|
13
|
+
};
|
|
14
|
+
/**
|
|
15
|
+
* Decodes the named and numeric character references the renderer can emit,
|
|
16
|
+
* plus the few that appear in hand-written HTML. Unknown references are left
|
|
17
|
+
* verbatim rather than dropped, so no text is silently lost.
|
|
18
|
+
*/
|
|
19
|
+
export declare function decodeEntities(text: string): string;
|
|
20
|
+
/**
|
|
21
|
+
* Tokenizes HTML into a flat token stream.
|
|
22
|
+
*
|
|
23
|
+
* A deliberate **subset** tokenizer, not an HTML5-conformant one: it handles
|
|
24
|
+
* the tags the Localess model can represent, treats everything else as a
|
|
25
|
+
* generic element for the caller's unsupported policy, and performs no error
|
|
26
|
+
* recovery beyond never throwing. Chosen over branching on `DOMParser` so the
|
|
27
|
+
* result is identical in browsers, Node, and edge runtimes, and so the package
|
|
28
|
+
* keeps its zero-dependency guarantee.
|
|
29
|
+
*
|
|
30
|
+
* A stray `<` that does not begin a valid tag is emitted as text.
|
|
31
|
+
*/
|
|
32
|
+
export declare function tokenizeHtml(html: string): HtmlToken[];
|
package/dist/index.d.ts
CHANGED
package/dist/index.js
CHANGED
|
@@ -1,41 +1,5 @@
|
|
|
1
1
|
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
-
|
|
3
|
-
var TEXT_ESCAPES = {
|
|
4
|
-
"&": "&",
|
|
5
|
-
"<": "<",
|
|
6
|
-
">": ">"
|
|
7
|
-
};
|
|
8
|
-
var ATTR_ESCAPES = {
|
|
9
|
-
...TEXT_ESCAPES,
|
|
10
|
-
"\"": """
|
|
11
|
-
};
|
|
12
|
-
/**
|
|
13
|
-
* Escapes text content for safe HTML output. The escape set (`& < >`) matches
|
|
14
|
-
* TipTap's `generateHTML` DOM serialization — parity-tested; do not widen it
|
|
15
|
-
* without updating the parity fixtures.
|
|
16
|
-
*/
|
|
17
|
-
function escapeHtml(text) {
|
|
18
|
-
return text.replace(/[&<>]/g, (ch) => TEXT_ESCAPES[ch]);
|
|
19
|
-
}
|
|
20
|
-
/** Escapes an attribute value for safe double-quoted HTML output (`& " < >`). */
|
|
21
|
-
function escapeAttr(value) {
|
|
22
|
-
return value.replace(/[&"<>]/g, (ch) => ATTR_ESCAPES[ch]);
|
|
23
|
-
}
|
|
24
|
-
var SAFE_SCHEME = /^(?:https?:|mailto:|tel:)/i;
|
|
25
|
-
var HAS_SCHEME = /^[a-z][a-z0-9+.-]*:/i;
|
|
26
|
-
/**
|
|
27
|
-
* Allowlist URL sanitizer for link hrefs: `http:`, `https:`, `mailto:`, `tel:`
|
|
28
|
-
* and scheme-less (relative/protocol-relative/fragment/query) URLs pass;
|
|
29
|
-
* everything else (e.g. `javascript:`, `data:`) becomes `''`.
|
|
30
|
-
*/
|
|
31
|
-
function sanitizeUrl(url) {
|
|
32
|
-
const trimmed = url.trim();
|
|
33
|
-
if (trimmed === "") return "";
|
|
34
|
-
if (SAFE_SCHEME.test(trimmed)) return trimmed;
|
|
35
|
-
if (!HAS_SCHEME.test(trimmed)) return trimmed;
|
|
36
|
-
return "";
|
|
37
|
-
}
|
|
38
|
-
//#endregion
|
|
2
|
+
const require_parse_common = require("./parse-common-DEYhfKnK.js");
|
|
39
3
|
//#region src/attrs.ts
|
|
40
4
|
/**
|
|
41
5
|
* Normalizes a node/mark's stored attrs into the attributes to emit, in the
|
|
@@ -60,7 +24,7 @@ function processAttrs(type, attrs, options = {}) {
|
|
|
60
24
|
case "link":
|
|
61
25
|
put("target", attrs.target);
|
|
62
26
|
put("rel", attrs.rel);
|
|
63
|
-
out[name("href")] = sanitizeUrl(String(attrs.href ?? ""));
|
|
27
|
+
out[name("href")] = require_parse_common.sanitizeUrl(String(attrs.href ?? ""));
|
|
64
28
|
put("class", attrs.class);
|
|
65
29
|
}
|
|
66
30
|
return out;
|
|
@@ -245,7 +209,7 @@ function renderNode(node, ctx) {
|
|
|
245
209
|
renderers: childRenderers,
|
|
246
210
|
warned: ctx.warned
|
|
247
211
|
};
|
|
248
|
-
const children = node.type === "text" ? escapeHtml(node.text ?? "") : renderNodes(node.content ?? [], childCtx);
|
|
212
|
+
const children = node.type === "text" ? require_parse_common.escapeHtml(node.text ?? "") : renderNodes(node.content ?? [], childCtx);
|
|
249
213
|
return custom({
|
|
250
214
|
...node,
|
|
251
215
|
children,
|
|
@@ -275,7 +239,7 @@ function renderSegments(segments, ctx) {
|
|
|
275
239
|
let out = "";
|
|
276
240
|
for (const segment of segments) {
|
|
277
241
|
if (segment.kind === "text") {
|
|
278
|
-
out += escapeHtml(segment.text);
|
|
242
|
+
out += require_parse_common.escapeHtml(segment.text);
|
|
279
243
|
continue;
|
|
280
244
|
}
|
|
281
245
|
const children = renderSegments(segment.children, ctx);
|
|
@@ -300,7 +264,7 @@ function renderSegments(segments, ctx) {
|
|
|
300
264
|
}
|
|
301
265
|
function wrapTag(tag, attrs, children) {
|
|
302
266
|
let open = `<${tag}`;
|
|
303
|
-
for (const [name, value] of Object.entries(attrs)) open += ` ${name}="${escapeAttr(String(value))}"`;
|
|
267
|
+
for (const [name, value] of Object.entries(attrs)) open += ` ${name}="${require_parse_common.escapeAttr(String(value))}"`;
|
|
304
268
|
return `${open}>${children}</${tag}>`;
|
|
305
269
|
}
|
|
306
270
|
function warnUnknown(ctx, type) {
|
|
@@ -312,12 +276,15 @@ function warnUnknown(ctx, type) {
|
|
|
312
276
|
//#endregion
|
|
313
277
|
exports.MARK_RENDER_MAP = MARK_RENDER_MAP;
|
|
314
278
|
exports.NODE_RENDER_MAP = NODE_RENDER_MAP;
|
|
279
|
+
exports.RichTextParseError = require_parse_common.RichTextParseError;
|
|
280
|
+
exports.UnsupportedTracker = require_parse_common.UnsupportedTracker;
|
|
315
281
|
exports.buildMarkTree = buildMarkTree;
|
|
316
|
-
exports.
|
|
317
|
-
exports.
|
|
282
|
+
exports.emptyDocument = require_parse_common.emptyDocument;
|
|
283
|
+
exports.escapeAttr = require_parse_common.escapeAttr;
|
|
284
|
+
exports.escapeHtml = require_parse_common.escapeHtml;
|
|
318
285
|
exports.marksEqual = marksEqual;
|
|
319
286
|
exports.normalizeInput = normalizeInput;
|
|
320
287
|
exports.processAttrs = processAttrs;
|
|
321
288
|
exports.renderRichTextToHtml = renderRichTextToHtml;
|
|
322
289
|
exports.resolveHeadingTag = resolveHeadingTag;
|
|
323
|
-
exports.sanitizeUrl = sanitizeUrl;
|
|
290
|
+
exports.sanitizeUrl = require_parse_common.sanitizeUrl;
|
package/dist/index.mjs
CHANGED
|
@@ -1,40 +1,4 @@
|
|
|
1
|
-
|
|
2
|
-
var TEXT_ESCAPES = {
|
|
3
|
-
"&": "&",
|
|
4
|
-
"<": "<",
|
|
5
|
-
">": ">"
|
|
6
|
-
};
|
|
7
|
-
var ATTR_ESCAPES = {
|
|
8
|
-
...TEXT_ESCAPES,
|
|
9
|
-
"\"": """
|
|
10
|
-
};
|
|
11
|
-
/**
|
|
12
|
-
* Escapes text content for safe HTML output. The escape set (`& < >`) matches
|
|
13
|
-
* TipTap's `generateHTML` DOM serialization — parity-tested; do not widen it
|
|
14
|
-
* without updating the parity fixtures.
|
|
15
|
-
*/
|
|
16
|
-
function escapeHtml(text) {
|
|
17
|
-
return text.replace(/[&<>]/g, (ch) => TEXT_ESCAPES[ch]);
|
|
18
|
-
}
|
|
19
|
-
/** Escapes an attribute value for safe double-quoted HTML output (`& " < >`). */
|
|
20
|
-
function escapeAttr(value) {
|
|
21
|
-
return value.replace(/[&"<>]/g, (ch) => ATTR_ESCAPES[ch]);
|
|
22
|
-
}
|
|
23
|
-
var SAFE_SCHEME = /^(?:https?:|mailto:|tel:)/i;
|
|
24
|
-
var HAS_SCHEME = /^[a-z][a-z0-9+.-]*:/i;
|
|
25
|
-
/**
|
|
26
|
-
* Allowlist URL sanitizer for link hrefs: `http:`, `https:`, `mailto:`, `tel:`
|
|
27
|
-
* and scheme-less (relative/protocol-relative/fragment/query) URLs pass;
|
|
28
|
-
* everything else (e.g. `javascript:`, `data:`) becomes `''`.
|
|
29
|
-
*/
|
|
30
|
-
function sanitizeUrl(url) {
|
|
31
|
-
const trimmed = url.trim();
|
|
32
|
-
if (trimmed === "") return "";
|
|
33
|
-
if (SAFE_SCHEME.test(trimmed)) return trimmed;
|
|
34
|
-
if (!HAS_SCHEME.test(trimmed)) return trimmed;
|
|
35
|
-
return "";
|
|
36
|
-
}
|
|
37
|
-
//#endregion
|
|
1
|
+
import { a as escapeHtml, i as escapeAttr, n as UnsupportedTracker, o as sanitizeUrl, r as emptyDocument, t as RichTextParseError } from "./parse-common-C-dAtFji.mjs";
|
|
38
2
|
//#region src/attrs.ts
|
|
39
3
|
/**
|
|
40
4
|
* Normalizes a node/mark's stored attrs into the attributes to emit, in the
|
|
@@ -309,4 +273,4 @@ function warnUnknown(ctx, type) {
|
|
|
309
273
|
console.warn(`[@localess/richtext] Unknown rich text element "${type}" was skipped. Provide a custom renderer to handle it.`);
|
|
310
274
|
}
|
|
311
275
|
//#endregion
|
|
312
|
-
export { MARK_RENDER_MAP, NODE_RENDER_MAP, buildMarkTree, escapeAttr, escapeHtml, marksEqual, normalizeInput, processAttrs, renderRichTextToHtml, resolveHeadingTag, sanitizeUrl };
|
|
276
|
+
export { MARK_RENDER_MAP, NODE_RENDER_MAP, RichTextParseError, UnsupportedTracker, buildMarkTree, emptyDocument, escapeAttr, escapeHtml, marksEqual, normalizeInput, processAttrs, renderRichTextToHtml, resolveHeadingTag, sanitizeUrl };
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { LocalessRichTextNode } from '../model';
|
|
2
|
+
import { UnsupportedTracker } from '../parse-common';
|
|
3
|
+
/**
|
|
4
|
+
* Parses Markdown block structure into model nodes.
|
|
5
|
+
*
|
|
6
|
+
* A documented **subset**, not CommonMark: ATX and setext headings,
|
|
7
|
+
* paragraphs, bullet and ordered lists, fenced and indented code blocks. Tables,
|
|
8
|
+
* images, blockquotes, thematic breaks, footnotes, and reference links have no
|
|
9
|
+
* representation in the model and go through the unsupported policy.
|
|
10
|
+
*/
|
|
11
|
+
export declare function parseBlocks(markdown: string, tracker: UnsupportedTracker): LocalessRichTextNode[];
|