@localess/richtext 4.0.0-dev.20260905085556 → 4.0.0-dev.20260906124410
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/SKILL.md +35 -0
- package/dist/html-parser/index.d.ts +29 -0
- package/dist/html-parser/index.js +491 -0
- package/dist/html-parser/index.mjs +489 -0
- package/dist/html-parser/to-model.d.ts +11 -0
- package/dist/html-parser/tokenizer.d.ts +32 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +11 -44
- package/dist/index.mjs +2 -38
- package/dist/markdown-parser/block.d.ts +11 -0
- package/dist/markdown-parser/index.d.ts +32 -0
- package/dist/markdown-parser/index.js +442 -0
- package/dist/markdown-parser/index.mjs +440 -0
- package/dist/markdown-parser/inline.d.ts +13 -0
- package/dist/parse-common-C-dAtFji.mjs +100 -0
- package/dist/parse-common-DEYhfKnK.js +135 -0
- package/dist/parse-common.d.ts +71 -0
- package/package.json +15 -2
package/SKILL.md
CHANGED
|
@@ -122,3 +122,38 @@ export type {
|
|
|
122
122
|
export { richTextFixtures }
|
|
123
123
|
export type { RichTextFixture }
|
|
124
124
|
```
|
|
125
|
+
|
|
126
|
+
## Parsing (`@localess/richtext/html-parser`, `/markdown-parser`)
|
|
127
|
+
|
|
128
|
+
```ts
|
|
129
|
+
import { parseHtmlToRichText } from '@localess/richtext/html-parser';
|
|
130
|
+
import { parseMarkdownToRichText } from '@localess/richtext/markdown-parser';
|
|
131
|
+
|
|
132
|
+
parseHtmlToRichText(html: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult
|
|
133
|
+
parseMarkdownToRichText(markdown: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
```ts
|
|
137
|
+
interface RichTextParseOptions { unsupported?: 'unwrap' | 'skip' | 'throw' } // default 'unwrap'
|
|
138
|
+
interface RichTextParseResult {
|
|
139
|
+
doc: LocalessRichTextDocument;
|
|
140
|
+
unsupported: { element: string; action: 'unwrapped' | 'skipped'; count: number }[];
|
|
141
|
+
}
|
|
142
|
+
class RichTextParseError extends Error { element: string }
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Both are **subset** parsers matched to the closed model — not HTML5-conformant, not CommonMark.
|
|
146
|
+
HTML covers `p`, `h1`–`h6`, `ul`, `ol` (`start`), `li`, `pre`/`code` (`language-*`) and the marks
|
|
147
|
+
`strong`/`b`, `em`/`i`, `s`/`strike`/`del`, `u`, `code`, `a`; `div`/`span` are transparent and
|
|
148
|
+
`script`/`style` are always dropped. Markdown covers ATX and setext headings, paragraphs, bullet and
|
|
149
|
+
ordered lists (nested, `start`), fenced and indented code blocks, `**bold**`, `*italic*`,
|
|
150
|
+
`~~strike~~`, `` `code` ``, `[text](href)` and backslash escapes.
|
|
151
|
+
|
|
152
|
+
Anything else goes through `unsupported` and is listed in the result — return the report to the
|
|
153
|
+
caller, do not assume a clean parse. One warning per element type per parse, silent in production.
|
|
154
|
+
|
|
155
|
+
Link hrefs pass the renderer's `sanitizeUrl` allowlist: `javascript:` and `data:` become `""`.
|
|
156
|
+
|
|
157
|
+
Never throws except under `unsupported: 'throw'`; `null`, `undefined`, and `''` give an empty doc.
|
|
158
|
+
|
|
159
|
+
`parseHtmlToRichText` is the exact inverse of `renderRichTextToHtml` over the whole supported model.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import { RichTextParseOptions, RichTextParseResult } from '../parse-common';
|
|
2
|
+
export type { RichTextParseOptions, RichTextParseResult, RichTextUnsupportedPolicy, RichTextUnsupportedReport } from '../parse-common';
|
|
3
|
+
export { RichTextParseError } from '../parse-common';
|
|
4
|
+
/**
|
|
5
|
+
* Parses an HTML string into a Localess rich text document.
|
|
6
|
+
*
|
|
7
|
+
* The inverse of `renderRichTextToHtml`, and deliberately a **subset** parser:
|
|
8
|
+
* it understands the tags the Localess model can represent and routes
|
|
9
|
+
* everything else through {@link RichTextParseOptions.unsupported}. It is not
|
|
10
|
+
* HTML5-conformant and does not attempt full error recovery — the trade is
|
|
11
|
+
* identical behaviour across browsers, Node, and edge runtimes with no
|
|
12
|
+
* dependency.
|
|
13
|
+
*
|
|
14
|
+
* Supported: `p`, `h1`–`h6`, `ul`, `ol` (with `start`), `li`, `pre`/`code`
|
|
15
|
+
* blocks (with `language-*`), and the marks `strong`/`b`, `em`/`i`,
|
|
16
|
+
* `s`/`strike`/`del`, `u`, `code`, `a`. Structural wrappers such as `div` and
|
|
17
|
+
* `span` are transparent; `script` and `style` are always dropped.
|
|
18
|
+
*
|
|
19
|
+
* Link hrefs pass through the same allowlist the renderer applies, so
|
|
20
|
+
* `javascript:` and `data:` become `""`.
|
|
21
|
+
*
|
|
22
|
+
* Never throws for malformed input — except under `unsupported: 'throw'`.
|
|
23
|
+
*
|
|
24
|
+
* @example
|
|
25
|
+
* ```ts
|
|
26
|
+
* const { doc, unsupported } = parseHtmlToRichText('<p>Hello <strong>world</strong></p>');
|
|
27
|
+
* ```
|
|
28
|
+
*/
|
|
29
|
+
export declare function parseHtmlToRichText(html: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult;
|
|
@@ -0,0 +1,491 @@
|
|
|
1
|
+
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
+
const require_parse_common = require("../parse-common-DEYhfKnK.js");
|
|
3
|
+
//#region src/html-parser/to-model.ts
|
|
4
|
+
/** Tag → mark, the inverse of `MARK_RENDER_MAP`. `b`/`i`/`del`/`s` are accepted as aliases. */
|
|
5
|
+
var MARK_TAGS = {
|
|
6
|
+
strong: "bold",
|
|
7
|
+
b: "bold",
|
|
8
|
+
em: "italic",
|
|
9
|
+
i: "italic",
|
|
10
|
+
s: "strike",
|
|
11
|
+
strike: "strike",
|
|
12
|
+
del: "strike",
|
|
13
|
+
u: "underline",
|
|
14
|
+
code: "code",
|
|
15
|
+
a: "link"
|
|
16
|
+
};
|
|
17
|
+
var HEADING_TAGS = {
|
|
18
|
+
h1: 1,
|
|
19
|
+
h2: 2,
|
|
20
|
+
h3: 3,
|
|
21
|
+
h4: 4,
|
|
22
|
+
h5: 5,
|
|
23
|
+
h6: 6
|
|
24
|
+
};
|
|
25
|
+
/** Block tags the model represents directly. */
|
|
26
|
+
var BLOCK_TAGS = /* @__PURE__ */ new Set([
|
|
27
|
+
"p",
|
|
28
|
+
"ul",
|
|
29
|
+
"ol",
|
|
30
|
+
"li",
|
|
31
|
+
"pre",
|
|
32
|
+
...Object.keys(HEADING_TAGS)
|
|
33
|
+
]);
|
|
34
|
+
/** Tags that carry no meaning of their own — their children are simply kept. */
|
|
35
|
+
var TRANSPARENT_TAGS = /* @__PURE__ */ new Set([
|
|
36
|
+
"html",
|
|
37
|
+
"body",
|
|
38
|
+
"head",
|
|
39
|
+
"div",
|
|
40
|
+
"section",
|
|
41
|
+
"article",
|
|
42
|
+
"main",
|
|
43
|
+
"span",
|
|
44
|
+
"font",
|
|
45
|
+
"tbody",
|
|
46
|
+
"thead"
|
|
47
|
+
]);
|
|
48
|
+
/** Tags dropped entirely, content and all, regardless of policy. */
|
|
49
|
+
var DROPPED_TAGS = /* @__PURE__ */ new Set([
|
|
50
|
+
"script",
|
|
51
|
+
"style",
|
|
52
|
+
"title",
|
|
53
|
+
"meta",
|
|
54
|
+
"link",
|
|
55
|
+
"base",
|
|
56
|
+
"noscript"
|
|
57
|
+
]);
|
|
58
|
+
function linkMark(attrs) {
|
|
59
|
+
return {
|
|
60
|
+
type: "link",
|
|
61
|
+
attrs: {
|
|
62
|
+
href: require_parse_common.sanitizeUrl(attrs.href ?? ""),
|
|
63
|
+
target: attrs.target ?? null,
|
|
64
|
+
rel: attrs.rel ?? null,
|
|
65
|
+
class: attrs.class ?? null
|
|
66
|
+
}
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
function codeBlockLanguage(attrs) {
|
|
70
|
+
const className = attrs.class ?? "";
|
|
71
|
+
const match = /(?:^|\s)language-([^\s]+)/.exec(className);
|
|
72
|
+
return match ? match[1] : null;
|
|
73
|
+
}
|
|
74
|
+
/** Collapses HTML whitespace the way a browser would for non-preformatted text. */
|
|
75
|
+
function collapseWhitespace(text) {
|
|
76
|
+
return text.replace(/[\t\n\r ]+/g, " ");
|
|
77
|
+
}
|
|
78
|
+
function isBlockNode(node) {
|
|
79
|
+
return node.type !== "text";
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Folds a token stream into model nodes.
|
|
83
|
+
*
|
|
84
|
+
* Text is only kept where the model can hold it — inside a block node. Text
|
|
85
|
+
* found at the top level is wrapped in a paragraph, matching what the editor
|
|
86
|
+
* would produce.
|
|
87
|
+
*/
|
|
88
|
+
function tokensToNodes(tokens, tracker) {
|
|
89
|
+
const root = {
|
|
90
|
+
tag: "",
|
|
91
|
+
node: null,
|
|
92
|
+
children: [],
|
|
93
|
+
marks: [],
|
|
94
|
+
preformatted: false
|
|
95
|
+
};
|
|
96
|
+
const stack = [root];
|
|
97
|
+
const top = () => stack[stack.length - 1];
|
|
98
|
+
/** Blocks skipped wholesale; text inside them is discarded. */
|
|
99
|
+
let skipDepth = 0;
|
|
100
|
+
let skipTag = "";
|
|
101
|
+
const appendText = (text) => {
|
|
102
|
+
const frame = top();
|
|
103
|
+
if (text === "") return;
|
|
104
|
+
const node = {
|
|
105
|
+
type: "text",
|
|
106
|
+
text,
|
|
107
|
+
...frame.marks.length ? { marks: [...frame.marks] } : {}
|
|
108
|
+
};
|
|
109
|
+
if (frame === root) {
|
|
110
|
+
const last = root.children[root.children.length - 1];
|
|
111
|
+
if (last && last.type === "paragraph") (last.content ??= []).push(node);
|
|
112
|
+
else root.children.push({
|
|
113
|
+
type: "paragraph",
|
|
114
|
+
content: [node]
|
|
115
|
+
});
|
|
116
|
+
return;
|
|
117
|
+
}
|
|
118
|
+
frame.children.push(node);
|
|
119
|
+
};
|
|
120
|
+
const closeFrame = () => {
|
|
121
|
+
const frame = stack.pop();
|
|
122
|
+
const parent = top();
|
|
123
|
+
if (frame.node) {
|
|
124
|
+
if (frame.children.length) frame.node.content = frame.children;
|
|
125
|
+
parent.children.push(frame.node);
|
|
126
|
+
} else parent.children.push(...frame.children);
|
|
127
|
+
};
|
|
128
|
+
for (const token of tokens) {
|
|
129
|
+
if (skipDepth > 0) {
|
|
130
|
+
if (token.kind === "open" && token.name === skipTag && !token.selfClosing) skipDepth++;
|
|
131
|
+
else if (token.kind === "close" && token.name === skipTag) skipDepth--;
|
|
132
|
+
continue;
|
|
133
|
+
}
|
|
134
|
+
if (token.kind === "text") {
|
|
135
|
+
const frame = top();
|
|
136
|
+
const text = frame.preformatted ? token.text : collapseWhitespace(token.text);
|
|
137
|
+
if (!frame.preformatted && text.trim() === "" && (frame === root || frame.children.every(isBlockNode))) continue;
|
|
138
|
+
appendText(text);
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
if (token.kind === "close") {
|
|
142
|
+
const depth = stack.findIndex((frame) => frame.tag === token.name);
|
|
143
|
+
if (depth <= 0) continue;
|
|
144
|
+
while (stack.length > depth) closeFrame();
|
|
145
|
+
continue;
|
|
146
|
+
}
|
|
147
|
+
const { name, attrs, selfClosing } = token;
|
|
148
|
+
if (DROPPED_TAGS.has(name)) {
|
|
149
|
+
if (!selfClosing) {
|
|
150
|
+
skipDepth = 1;
|
|
151
|
+
skipTag = name;
|
|
152
|
+
}
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
if (name === "br") {
|
|
156
|
+
appendText(" ");
|
|
157
|
+
continue;
|
|
158
|
+
}
|
|
159
|
+
const markType = MARK_TAGS[name];
|
|
160
|
+
if (markType) {
|
|
161
|
+
const frame = top();
|
|
162
|
+
if (name === "code" && frame.tag === "pre" && frame.node?.type === "codeBlock") {
|
|
163
|
+
const language = codeBlockLanguage(attrs);
|
|
164
|
+
if (language) frame.node.attrs = { language };
|
|
165
|
+
if (selfClosing) continue;
|
|
166
|
+
stack.push({
|
|
167
|
+
tag: name,
|
|
168
|
+
node: null,
|
|
169
|
+
children: [],
|
|
170
|
+
marks: [...frame.marks],
|
|
171
|
+
preformatted: true
|
|
172
|
+
});
|
|
173
|
+
continue;
|
|
174
|
+
}
|
|
175
|
+
const mark = markType === "link" ? linkMark(attrs) : { type: markType };
|
|
176
|
+
if (selfClosing) continue;
|
|
177
|
+
stack.push({
|
|
178
|
+
tag: name,
|
|
179
|
+
node: null,
|
|
180
|
+
children: [],
|
|
181
|
+
marks: [...frame.marks, mark],
|
|
182
|
+
preformatted: frame.preformatted
|
|
183
|
+
});
|
|
184
|
+
continue;
|
|
185
|
+
}
|
|
186
|
+
if (TRANSPARENT_TAGS.has(name)) {
|
|
187
|
+
if (selfClosing) continue;
|
|
188
|
+
const frame = top();
|
|
189
|
+
stack.push({
|
|
190
|
+
tag: name,
|
|
191
|
+
node: null,
|
|
192
|
+
children: [],
|
|
193
|
+
marks: [...frame.marks],
|
|
194
|
+
preformatted: frame.preformatted
|
|
195
|
+
});
|
|
196
|
+
continue;
|
|
197
|
+
}
|
|
198
|
+
if (BLOCK_TAGS.has(name)) {
|
|
199
|
+
const frame = top();
|
|
200
|
+
let node;
|
|
201
|
+
let preformatted = frame.preformatted;
|
|
202
|
+
if (name === "p") node = { type: "paragraph" };
|
|
203
|
+
else if (name in HEADING_TAGS) node = {
|
|
204
|
+
type: "heading",
|
|
205
|
+
attrs: { level: HEADING_TAGS[name] }
|
|
206
|
+
};
|
|
207
|
+
else if (name === "ul") node = { type: "bulletList" };
|
|
208
|
+
else if (name === "ol") {
|
|
209
|
+
const start = Number.parseInt(attrs.start ?? "", 10);
|
|
210
|
+
node = Number.isFinite(start) && start !== 1 ? {
|
|
211
|
+
type: "orderedList",
|
|
212
|
+
attrs: { start }
|
|
213
|
+
} : { type: "orderedList" };
|
|
214
|
+
} else if (name === "li") node = { type: "listItem" };
|
|
215
|
+
else {
|
|
216
|
+
node = { type: "codeBlock" };
|
|
217
|
+
preformatted = true;
|
|
218
|
+
}
|
|
219
|
+
if (selfClosing) {
|
|
220
|
+
frame.children.push(node);
|
|
221
|
+
continue;
|
|
222
|
+
}
|
|
223
|
+
stack.push({
|
|
224
|
+
tag: name,
|
|
225
|
+
node,
|
|
226
|
+
children: [],
|
|
227
|
+
marks: [...frame.marks],
|
|
228
|
+
preformatted
|
|
229
|
+
});
|
|
230
|
+
continue;
|
|
231
|
+
}
|
|
232
|
+
const action = tracker.record(name);
|
|
233
|
+
if (selfClosing) continue;
|
|
234
|
+
if (action === "skip") {
|
|
235
|
+
skipDepth = 1;
|
|
236
|
+
skipTag = name;
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
239
|
+
const frame = top();
|
|
240
|
+
stack.push({
|
|
241
|
+
tag: name,
|
|
242
|
+
node: null,
|
|
243
|
+
children: [],
|
|
244
|
+
marks: [...frame.marks],
|
|
245
|
+
preformatted: frame.preformatted
|
|
246
|
+
});
|
|
247
|
+
}
|
|
248
|
+
while (stack.length > 1) closeFrame();
|
|
249
|
+
return finalize(root.children);
|
|
250
|
+
}
|
|
251
|
+
/**
|
|
252
|
+
* Post-pass matching the model's shape rules:
|
|
253
|
+
* `<pre>` wraps a `<code>` in the renderer, so the parser lifts that `code`
|
|
254
|
+
* mark back onto the `codeBlock`'s `language`, and list items always hold
|
|
255
|
+
* blocks rather than bare text.
|
|
256
|
+
*/
|
|
257
|
+
function finalize(nodes) {
|
|
258
|
+
return nodes.map((node) => {
|
|
259
|
+
if (node.type === "text") return node;
|
|
260
|
+
const content = node.content;
|
|
261
|
+
if (!content) return node;
|
|
262
|
+
if (node.type === "codeBlock") {
|
|
263
|
+
const text = flattenText(content);
|
|
264
|
+
const next = {
|
|
265
|
+
type: "codeBlock",
|
|
266
|
+
...node.attrs ? { attrs: node.attrs } : {}
|
|
267
|
+
};
|
|
268
|
+
if (text !== "") next.content = [{
|
|
269
|
+
type: "text",
|
|
270
|
+
text
|
|
271
|
+
}];
|
|
272
|
+
return next;
|
|
273
|
+
}
|
|
274
|
+
const finalized = finalize(content);
|
|
275
|
+
if (node.type === "listItem" && finalized.some((child) => child.type === "text")) {
|
|
276
|
+
const wrapped = [];
|
|
277
|
+
let run = [];
|
|
278
|
+
for (const child of finalized) if (child.type === "text") run.push(child);
|
|
279
|
+
else {
|
|
280
|
+
if (run.length) {
|
|
281
|
+
wrapped.push({
|
|
282
|
+
type: "paragraph",
|
|
283
|
+
content: run
|
|
284
|
+
});
|
|
285
|
+
run = [];
|
|
286
|
+
}
|
|
287
|
+
wrapped.push(child);
|
|
288
|
+
}
|
|
289
|
+
if (run.length) wrapped.push({
|
|
290
|
+
type: "paragraph",
|
|
291
|
+
content: run
|
|
292
|
+
});
|
|
293
|
+
return {
|
|
294
|
+
...node,
|
|
295
|
+
content: wrapped
|
|
296
|
+
};
|
|
297
|
+
}
|
|
298
|
+
return {
|
|
299
|
+
...node,
|
|
300
|
+
content: finalized
|
|
301
|
+
};
|
|
302
|
+
});
|
|
303
|
+
}
|
|
304
|
+
function flattenText(nodes) {
|
|
305
|
+
let out = "";
|
|
306
|
+
for (const node of nodes) if (node.type === "text") out += node.text;
|
|
307
|
+
else out += flattenText(node.content ?? []);
|
|
308
|
+
return out;
|
|
309
|
+
}
|
|
310
|
+
//#endregion
|
|
311
|
+
//#region src/html-parser/tokenizer.ts
|
|
312
|
+
/** Elements that never have a closing tag. */
|
|
313
|
+
var VOID_ELEMENTS = /* @__PURE__ */ new Set([
|
|
314
|
+
"area",
|
|
315
|
+
"base",
|
|
316
|
+
"br",
|
|
317
|
+
"col",
|
|
318
|
+
"embed",
|
|
319
|
+
"hr",
|
|
320
|
+
"img",
|
|
321
|
+
"input",
|
|
322
|
+
"link",
|
|
323
|
+
"meta",
|
|
324
|
+
"param",
|
|
325
|
+
"source",
|
|
326
|
+
"track",
|
|
327
|
+
"wbr"
|
|
328
|
+
]);
|
|
329
|
+
/** Elements whose content is raw text, not markup. */
|
|
330
|
+
var RAW_TEXT_ELEMENTS = /* @__PURE__ */ new Set(["script", "style"]);
|
|
331
|
+
var NAMED_ENTITIES = {
|
|
332
|
+
amp: "&",
|
|
333
|
+
lt: "<",
|
|
334
|
+
gt: ">",
|
|
335
|
+
quot: "\"",
|
|
336
|
+
apos: "'",
|
|
337
|
+
nbsp: "\xA0"
|
|
338
|
+
};
|
|
339
|
+
/**
|
|
340
|
+
* Decodes the named and numeric character references the renderer can emit,
|
|
341
|
+
* plus the few that appear in hand-written HTML. Unknown references are left
|
|
342
|
+
* verbatim rather than dropped, so no text is silently lost.
|
|
343
|
+
*/
|
|
344
|
+
function decodeEntities(text) {
|
|
345
|
+
if (!text.includes("&")) return text;
|
|
346
|
+
return text.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (match, body) => {
|
|
347
|
+
if (body[0] === "#") {
|
|
348
|
+
const codePoint = body[1] === "x" || body[1] === "X" ? Number.parseInt(body.slice(2), 16) : Number.parseInt(body.slice(1), 10);
|
|
349
|
+
if (!Number.isFinite(codePoint) || codePoint < 0 || codePoint > 1114111) return match;
|
|
350
|
+
try {
|
|
351
|
+
return String.fromCodePoint(codePoint);
|
|
352
|
+
} catch {
|
|
353
|
+
return match;
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
return NAMED_ENTITIES[body.toLowerCase()] ?? match;
|
|
357
|
+
});
|
|
358
|
+
}
|
|
359
|
+
function parseAttributes(source) {
|
|
360
|
+
const attrs = {};
|
|
361
|
+
const pattern = /([^\s=/>]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
|
|
362
|
+
let match;
|
|
363
|
+
while ((match = pattern.exec(source)) !== null) {
|
|
364
|
+
const name = match[1].toLowerCase();
|
|
365
|
+
const value = match[2] ?? match[3] ?? match[4] ?? "";
|
|
366
|
+
if (!(name in attrs)) attrs[name] = decodeEntities(value);
|
|
367
|
+
}
|
|
368
|
+
return attrs;
|
|
369
|
+
}
|
|
370
|
+
/**
|
|
371
|
+
* Tokenizes HTML into a flat token stream.
|
|
372
|
+
*
|
|
373
|
+
* A deliberate **subset** tokenizer, not an HTML5-conformant one: it handles
|
|
374
|
+
* the tags the Localess model can represent, treats everything else as a
|
|
375
|
+
* generic element for the caller's unsupported policy, and performs no error
|
|
376
|
+
* recovery beyond never throwing. Chosen over branching on `DOMParser` so the
|
|
377
|
+
* result is identical in browsers, Node, and edge runtimes, and so the package
|
|
378
|
+
* keeps its zero-dependency guarantee.
|
|
379
|
+
*
|
|
380
|
+
* A stray `<` that does not begin a valid tag is emitted as text.
|
|
381
|
+
*/
|
|
382
|
+
function tokenizeHtml(html) {
|
|
383
|
+
const tokens = [];
|
|
384
|
+
let index = 0;
|
|
385
|
+
let pendingText = "";
|
|
386
|
+
const flushText = () => {
|
|
387
|
+
if (pendingText === "") return;
|
|
388
|
+
tokens.push({
|
|
389
|
+
kind: "text",
|
|
390
|
+
text: decodeEntities(pendingText)
|
|
391
|
+
});
|
|
392
|
+
pendingText = "";
|
|
393
|
+
};
|
|
394
|
+
while (index < html.length) {
|
|
395
|
+
const next = html.indexOf("<", index);
|
|
396
|
+
if (next === -1) {
|
|
397
|
+
pendingText += html.slice(index);
|
|
398
|
+
break;
|
|
399
|
+
}
|
|
400
|
+
pendingText += html.slice(index, next);
|
|
401
|
+
const rest = html.slice(next);
|
|
402
|
+
if (rest.startsWith("<!--")) {
|
|
403
|
+
const end = html.indexOf("-->", next + 4);
|
|
404
|
+
index = end === -1 ? html.length : end + 3;
|
|
405
|
+
continue;
|
|
406
|
+
}
|
|
407
|
+
if (rest.startsWith("<!") || rest.startsWith("<?")) {
|
|
408
|
+
const end = html.indexOf(">", next + 1);
|
|
409
|
+
index = end === -1 ? html.length : end + 1;
|
|
410
|
+
continue;
|
|
411
|
+
}
|
|
412
|
+
const closeMatch = /^<\/\s*([a-zA-Z][^\s>]*)\s*>/.exec(rest);
|
|
413
|
+
if (closeMatch) {
|
|
414
|
+
flushText();
|
|
415
|
+
tokens.push({
|
|
416
|
+
kind: "close",
|
|
417
|
+
name: closeMatch[1].toLowerCase()
|
|
418
|
+
});
|
|
419
|
+
index = next + closeMatch[0].length;
|
|
420
|
+
continue;
|
|
421
|
+
}
|
|
422
|
+
const openMatch = /^<([a-zA-Z][^\s/>]*)((?:[^>"']|"[^"]*"|'[^']*')*)>/.exec(rest);
|
|
423
|
+
if (openMatch) {
|
|
424
|
+
flushText();
|
|
425
|
+
const name = openMatch[1].toLowerCase();
|
|
426
|
+
const raw = openMatch[2] ?? "";
|
|
427
|
+
const selfClosing = /\/\s*$/.test(raw) || VOID_ELEMENTS.has(name);
|
|
428
|
+
tokens.push({
|
|
429
|
+
kind: "open",
|
|
430
|
+
name,
|
|
431
|
+
attrs: parseAttributes(raw),
|
|
432
|
+
selfClosing
|
|
433
|
+
});
|
|
434
|
+
index = next + openMatch[0].length;
|
|
435
|
+
if (RAW_TEXT_ELEMENTS.has(name) && !selfClosing) {
|
|
436
|
+
const closeTag = `</${name}`;
|
|
437
|
+
const end = html.toLowerCase().indexOf(closeTag, index);
|
|
438
|
+
index = end === -1 ? html.length : end;
|
|
439
|
+
}
|
|
440
|
+
continue;
|
|
441
|
+
}
|
|
442
|
+
pendingText += "<";
|
|
443
|
+
index = next + 1;
|
|
444
|
+
}
|
|
445
|
+
flushText();
|
|
446
|
+
return tokens;
|
|
447
|
+
}
|
|
448
|
+
//#endregion
|
|
449
|
+
//#region src/html-parser/index.ts
|
|
450
|
+
/**
|
|
451
|
+
* Parses an HTML string into a Localess rich text document.
|
|
452
|
+
*
|
|
453
|
+
* The inverse of `renderRichTextToHtml`, and deliberately a **subset** parser:
|
|
454
|
+
* it understands the tags the Localess model can represent and routes
|
|
455
|
+
* everything else through {@link RichTextParseOptions.unsupported}. It is not
|
|
456
|
+
* HTML5-conformant and does not attempt full error recovery — the trade is
|
|
457
|
+
* identical behaviour across browsers, Node, and edge runtimes with no
|
|
458
|
+
* dependency.
|
|
459
|
+
*
|
|
460
|
+
* Supported: `p`, `h1`–`h6`, `ul`, `ol` (with `start`), `li`, `pre`/`code`
|
|
461
|
+
* blocks (with `language-*`), and the marks `strong`/`b`, `em`/`i`,
|
|
462
|
+
* `s`/`strike`/`del`, `u`, `code`, `a`. Structural wrappers such as `div` and
|
|
463
|
+
* `span` are transparent; `script` and `style` are always dropped.
|
|
464
|
+
*
|
|
465
|
+
* Link hrefs pass through the same allowlist the renderer applies, so
|
|
466
|
+
* `javascript:` and `data:` become `""`.
|
|
467
|
+
*
|
|
468
|
+
* Never throws for malformed input — except under `unsupported: 'throw'`.
|
|
469
|
+
*
|
|
470
|
+
* @example
|
|
471
|
+
* ```ts
|
|
472
|
+
* const { doc, unsupported } = parseHtmlToRichText('<p>Hello <strong>world</strong></p>');
|
|
473
|
+
* ```
|
|
474
|
+
*/
|
|
475
|
+
function parseHtmlToRichText(html, options = {}) {
|
|
476
|
+
const tracker = new require_parse_common.UnsupportedTracker(options);
|
|
477
|
+
if (typeof html !== "string" || html === "") return {
|
|
478
|
+
doc: require_parse_common.emptyDocument(),
|
|
479
|
+
unsupported: []
|
|
480
|
+
};
|
|
481
|
+
return {
|
|
482
|
+
doc: {
|
|
483
|
+
type: "doc",
|
|
484
|
+
content: tokensToNodes(tokenizeHtml(html), tracker)
|
|
485
|
+
},
|
|
486
|
+
unsupported: tracker.report()
|
|
487
|
+
};
|
|
488
|
+
}
|
|
489
|
+
//#endregion
|
|
490
|
+
exports.RichTextParseError = require_parse_common.RichTextParseError;
|
|
491
|
+
exports.parseHtmlToRichText = parseHtmlToRichText;
|