@localess/richtext 4.0.0-dev.20260905092558 → 4.0.0-dev.20260906133314

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/SKILL.md CHANGED
@@ -122,3 +122,38 @@ export type {
122
122
  export { richTextFixtures }
123
123
  export type { RichTextFixture }
124
124
  ```
125
+
126
+ ## Parsing (`@localess/richtext/html-parser`, `/markdown-parser`)
127
+
128
+ ```ts
129
+ import { parseHtmlToRichText } from '@localess/richtext/html-parser';
130
+ import { parseMarkdownToRichText } from '@localess/richtext/markdown-parser';
131
+
132
+ parseHtmlToRichText(html: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult
133
+ parseMarkdownToRichText(markdown: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult
134
+ ```
135
+
136
+ ```ts
137
+ interface RichTextParseOptions { unsupported?: 'unwrap' | 'skip' | 'throw' } // default 'unwrap'
138
+ interface RichTextParseResult {
139
+ doc: LocalessRichTextDocument;
140
+ unsupported: { element: string; action: 'unwrapped' | 'skipped'; count: number }[];
141
+ }
142
+ class RichTextParseError extends Error { element: string }
143
+ ```
144
+
145
+ Both are **subset** parsers matched to the closed model — not HTML5-conformant, not CommonMark.
146
+ HTML covers `p`, `h1`–`h6`, `ul`, `ol` (`start`), `li`, `pre`/`code` (`language-*`) and the marks
147
+ `strong`/`b`, `em`/`i`, `s`/`strike`/`del`, `u`, `code`, `a`; `div`/`span` are transparent and
148
+ `script`/`style` are always dropped. Markdown covers ATX and setext headings, paragraphs, bullet and
149
+ ordered lists (nested, `start`), fenced and indented code blocks, `**bold**`, `*italic*`,
150
+ `~~strike~~`, `` `code` ``, `[text](href)` and backslash escapes.
151
+
152
+ Anything else goes through `unsupported` and is listed in the result — return the report to the
153
+ caller, do not assume a clean parse. One warning per element type per parse, silent in production.
154
+
155
+ Link hrefs pass the renderer's `sanitizeUrl` allowlist: `javascript:` and `data:` become `""`.
156
+
157
+ Never throws except under `unsupported: 'throw'`; `null`, `undefined`, and `''` give an empty doc.
158
+
159
+ `parseHtmlToRichText` is the exact inverse of `renderRichTextToHtml` over the whole supported model.
@@ -0,0 +1,29 @@
1
+ import { RichTextParseOptions, RichTextParseResult } from '../parse-common';
2
+ export type { RichTextParseOptions, RichTextParseResult, RichTextUnsupportedPolicy, RichTextUnsupportedReport } from '../parse-common';
3
+ export { RichTextParseError } from '../parse-common';
4
+ /**
5
+ * Parses an HTML string into a Localess rich text document.
6
+ *
7
+ * The inverse of `renderRichTextToHtml`, and deliberately a **subset** parser:
8
+ * it understands the tags the Localess model can represent and routes
9
+ * everything else through {@link RichTextParseOptions.unsupported}. It is not
10
+ * HTML5-conformant and does not attempt full error recovery — the trade is
11
+ * identical behaviour across browsers, Node, and edge runtimes with no
12
+ * dependency.
13
+ *
14
+ * Supported: `p`, `h1`–`h6`, `ul`, `ol` (with `start`), `li`, `pre`/`code`
15
+ * blocks (with `language-*`), and the marks `strong`/`b`, `em`/`i`,
16
+ * `s`/`strike`/`del`, `u`, `code`, `a`. Structural wrappers such as `div` and
17
+ * `span` are transparent; `script` and `style` are always dropped.
18
+ *
19
+ * Link hrefs pass through the same allowlist the renderer applies, so
20
+ * `javascript:` and `data:` become `""`.
21
+ *
22
+ * Never throws for malformed input — except under `unsupported: 'throw'`.
23
+ *
24
+ * @example
25
+ * ```ts
26
+ * const { doc, unsupported } = parseHtmlToRichText('<p>Hello <strong>world</strong></p>');
27
+ * ```
28
+ */
29
+ export declare function parseHtmlToRichText(html: string | null | undefined, options?: RichTextParseOptions): RichTextParseResult;
@@ -0,0 +1,491 @@
1
+ Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
+ const require_parse_common = require("../parse-common-DEYhfKnK.js");
3
+ //#region src/html-parser/to-model.ts
4
+ /** Tag → mark, the inverse of `MARK_RENDER_MAP`. `b`/`i`/`del`/`s` are accepted as aliases. */
5
+ var MARK_TAGS = {
6
+ strong: "bold",
7
+ b: "bold",
8
+ em: "italic",
9
+ i: "italic",
10
+ s: "strike",
11
+ strike: "strike",
12
+ del: "strike",
13
+ u: "underline",
14
+ code: "code",
15
+ a: "link"
16
+ };
17
+ var HEADING_TAGS = {
18
+ h1: 1,
19
+ h2: 2,
20
+ h3: 3,
21
+ h4: 4,
22
+ h5: 5,
23
+ h6: 6
24
+ };
25
+ /** Block tags the model represents directly. */
26
+ var BLOCK_TAGS = /* @__PURE__ */ new Set([
27
+ "p",
28
+ "ul",
29
+ "ol",
30
+ "li",
31
+ "pre",
32
+ ...Object.keys(HEADING_TAGS)
33
+ ]);
34
+ /** Tags that carry no meaning of their own — their children are simply kept. */
35
+ var TRANSPARENT_TAGS = /* @__PURE__ */ new Set([
36
+ "html",
37
+ "body",
38
+ "head",
39
+ "div",
40
+ "section",
41
+ "article",
42
+ "main",
43
+ "span",
44
+ "font",
45
+ "tbody",
46
+ "thead"
47
+ ]);
48
+ /** Tags dropped entirely, content and all, regardless of policy. */
49
+ var DROPPED_TAGS = /* @__PURE__ */ new Set([
50
+ "script",
51
+ "style",
52
+ "title",
53
+ "meta",
54
+ "link",
55
+ "base",
56
+ "noscript"
57
+ ]);
58
+ function linkMark(attrs) {
59
+ return {
60
+ type: "link",
61
+ attrs: {
62
+ href: require_parse_common.sanitizeUrl(attrs.href ?? ""),
63
+ target: attrs.target ?? null,
64
+ rel: attrs.rel ?? null,
65
+ class: attrs.class ?? null
66
+ }
67
+ };
68
+ }
69
+ function codeBlockLanguage(attrs) {
70
+ const className = attrs.class ?? "";
71
+ const match = /(?:^|\s)language-([^\s]+)/.exec(className);
72
+ return match ? match[1] : null;
73
+ }
74
+ /** Collapses HTML whitespace the way a browser would for non-preformatted text. */
75
+ function collapseWhitespace(text) {
76
+ return text.replace(/[\t\n\r ]+/g, " ");
77
+ }
78
+ function isBlockNode(node) {
79
+ return node.type !== "text";
80
+ }
81
+ /**
82
+ * Folds a token stream into model nodes.
83
+ *
84
+ * Text is only kept where the model can hold it — inside a block node. Text
85
+ * found at the top level is wrapped in a paragraph, matching what the editor
86
+ * would produce.
87
+ */
88
+ function tokensToNodes(tokens, tracker) {
89
+ const root = {
90
+ tag: "",
91
+ node: null,
92
+ children: [],
93
+ marks: [],
94
+ preformatted: false
95
+ };
96
+ const stack = [root];
97
+ const top = () => stack[stack.length - 1];
98
+ /** Blocks skipped wholesale; text inside them is discarded. */
99
+ let skipDepth = 0;
100
+ let skipTag = "";
101
+ const appendText = (text) => {
102
+ const frame = top();
103
+ if (text === "") return;
104
+ const node = {
105
+ type: "text",
106
+ text,
107
+ ...frame.marks.length ? { marks: [...frame.marks] } : {}
108
+ };
109
+ if (frame === root) {
110
+ const last = root.children[root.children.length - 1];
111
+ if (last && last.type === "paragraph") (last.content ??= []).push(node);
112
+ else root.children.push({
113
+ type: "paragraph",
114
+ content: [node]
115
+ });
116
+ return;
117
+ }
118
+ frame.children.push(node);
119
+ };
120
+ const closeFrame = () => {
121
+ const frame = stack.pop();
122
+ const parent = top();
123
+ if (frame.node) {
124
+ if (frame.children.length) frame.node.content = frame.children;
125
+ parent.children.push(frame.node);
126
+ } else parent.children.push(...frame.children);
127
+ };
128
+ for (const token of tokens) {
129
+ if (skipDepth > 0) {
130
+ if (token.kind === "open" && token.name === skipTag && !token.selfClosing) skipDepth++;
131
+ else if (token.kind === "close" && token.name === skipTag) skipDepth--;
132
+ continue;
133
+ }
134
+ if (token.kind === "text") {
135
+ const frame = top();
136
+ const text = frame.preformatted ? token.text : collapseWhitespace(token.text);
137
+ if (!frame.preformatted && text.trim() === "" && (frame === root || frame.children.every(isBlockNode))) continue;
138
+ appendText(text);
139
+ continue;
140
+ }
141
+ if (token.kind === "close") {
142
+ const depth = stack.findIndex((frame) => frame.tag === token.name);
143
+ if (depth <= 0) continue;
144
+ while (stack.length > depth) closeFrame();
145
+ continue;
146
+ }
147
+ const { name, attrs, selfClosing } = token;
148
+ if (DROPPED_TAGS.has(name)) {
149
+ if (!selfClosing) {
150
+ skipDepth = 1;
151
+ skipTag = name;
152
+ }
153
+ continue;
154
+ }
155
+ if (name === "br") {
156
+ appendText(" ");
157
+ continue;
158
+ }
159
+ const markType = MARK_TAGS[name];
160
+ if (markType) {
161
+ const frame = top();
162
+ if (name === "code" && frame.tag === "pre" && frame.node?.type === "codeBlock") {
163
+ const language = codeBlockLanguage(attrs);
164
+ if (language) frame.node.attrs = { language };
165
+ if (selfClosing) continue;
166
+ stack.push({
167
+ tag: name,
168
+ node: null,
169
+ children: [],
170
+ marks: [...frame.marks],
171
+ preformatted: true
172
+ });
173
+ continue;
174
+ }
175
+ const mark = markType === "link" ? linkMark(attrs) : { type: markType };
176
+ if (selfClosing) continue;
177
+ stack.push({
178
+ tag: name,
179
+ node: null,
180
+ children: [],
181
+ marks: [...frame.marks, mark],
182
+ preformatted: frame.preformatted
183
+ });
184
+ continue;
185
+ }
186
+ if (TRANSPARENT_TAGS.has(name)) {
187
+ if (selfClosing) continue;
188
+ const frame = top();
189
+ stack.push({
190
+ tag: name,
191
+ node: null,
192
+ children: [],
193
+ marks: [...frame.marks],
194
+ preformatted: frame.preformatted
195
+ });
196
+ continue;
197
+ }
198
+ if (BLOCK_TAGS.has(name)) {
199
+ const frame = top();
200
+ let node;
201
+ let preformatted = frame.preformatted;
202
+ if (name === "p") node = { type: "paragraph" };
203
+ else if (name in HEADING_TAGS) node = {
204
+ type: "heading",
205
+ attrs: { level: HEADING_TAGS[name] }
206
+ };
207
+ else if (name === "ul") node = { type: "bulletList" };
208
+ else if (name === "ol") {
209
+ const start = Number.parseInt(attrs.start ?? "", 10);
210
+ node = Number.isFinite(start) && start !== 1 ? {
211
+ type: "orderedList",
212
+ attrs: { start }
213
+ } : { type: "orderedList" };
214
+ } else if (name === "li") node = { type: "listItem" };
215
+ else {
216
+ node = { type: "codeBlock" };
217
+ preformatted = true;
218
+ }
219
+ if (selfClosing) {
220
+ frame.children.push(node);
221
+ continue;
222
+ }
223
+ stack.push({
224
+ tag: name,
225
+ node,
226
+ children: [],
227
+ marks: [...frame.marks],
228
+ preformatted
229
+ });
230
+ continue;
231
+ }
232
+ const action = tracker.record(name);
233
+ if (selfClosing) continue;
234
+ if (action === "skip") {
235
+ skipDepth = 1;
236
+ skipTag = name;
237
+ continue;
238
+ }
239
+ const frame = top();
240
+ stack.push({
241
+ tag: name,
242
+ node: null,
243
+ children: [],
244
+ marks: [...frame.marks],
245
+ preformatted: frame.preformatted
246
+ });
247
+ }
248
+ while (stack.length > 1) closeFrame();
249
+ return finalize(root.children);
250
+ }
251
+ /**
252
+ * Post-pass matching the model's shape rules:
253
+ * `<pre>` wraps a `<code>` in the renderer, so the parser lifts that `code`
254
+ * mark back onto the `codeBlock`'s `language`, and list items always hold
255
+ * blocks rather than bare text.
256
+ */
257
+ function finalize(nodes) {
258
+ return nodes.map((node) => {
259
+ if (node.type === "text") return node;
260
+ const content = node.content;
261
+ if (!content) return node;
262
+ if (node.type === "codeBlock") {
263
+ const text = flattenText(content);
264
+ const next = {
265
+ type: "codeBlock",
266
+ ...node.attrs ? { attrs: node.attrs } : {}
267
+ };
268
+ if (text !== "") next.content = [{
269
+ type: "text",
270
+ text
271
+ }];
272
+ return next;
273
+ }
274
+ const finalized = finalize(content);
275
+ if (node.type === "listItem" && finalized.some((child) => child.type === "text")) {
276
+ const wrapped = [];
277
+ let run = [];
278
+ for (const child of finalized) if (child.type === "text") run.push(child);
279
+ else {
280
+ if (run.length) {
281
+ wrapped.push({
282
+ type: "paragraph",
283
+ content: run
284
+ });
285
+ run = [];
286
+ }
287
+ wrapped.push(child);
288
+ }
289
+ if (run.length) wrapped.push({
290
+ type: "paragraph",
291
+ content: run
292
+ });
293
+ return {
294
+ ...node,
295
+ content: wrapped
296
+ };
297
+ }
298
+ return {
299
+ ...node,
300
+ content: finalized
301
+ };
302
+ });
303
+ }
304
+ function flattenText(nodes) {
305
+ let out = "";
306
+ for (const node of nodes) if (node.type === "text") out += node.text;
307
+ else out += flattenText(node.content ?? []);
308
+ return out;
309
+ }
310
+ //#endregion
311
+ //#region src/html-parser/tokenizer.ts
312
+ /** Elements that never have a closing tag. */
313
+ var VOID_ELEMENTS = /* @__PURE__ */ new Set([
314
+ "area",
315
+ "base",
316
+ "br",
317
+ "col",
318
+ "embed",
319
+ "hr",
320
+ "img",
321
+ "input",
322
+ "link",
323
+ "meta",
324
+ "param",
325
+ "source",
326
+ "track",
327
+ "wbr"
328
+ ]);
329
+ /** Elements whose content is raw text, not markup. */
330
+ var RAW_TEXT_ELEMENTS = /* @__PURE__ */ new Set(["script", "style"]);
331
+ var NAMED_ENTITIES = {
332
+ amp: "&",
333
+ lt: "<",
334
+ gt: ">",
335
+ quot: "\"",
336
+ apos: "'",
337
+ nbsp: "\xA0"
338
+ };
339
+ /**
340
+ * Decodes the named and numeric character references the renderer can emit,
341
+ * plus the few that appear in hand-written HTML. Unknown references are left
342
+ * verbatim rather than dropped, so no text is silently lost.
343
+ */
344
+ function decodeEntities(text) {
345
+ if (!text.includes("&")) return text;
346
+ return text.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (match, body) => {
347
+ if (body[0] === "#") {
348
+ const codePoint = body[1] === "x" || body[1] === "X" ? Number.parseInt(body.slice(2), 16) : Number.parseInt(body.slice(1), 10);
349
+ if (!Number.isFinite(codePoint) || codePoint < 0 || codePoint > 1114111) return match;
350
+ try {
351
+ return String.fromCodePoint(codePoint);
352
+ } catch {
353
+ return match;
354
+ }
355
+ }
356
+ return NAMED_ENTITIES[body.toLowerCase()] ?? match;
357
+ });
358
+ }
359
+ function parseAttributes(source) {
360
+ const attrs = {};
361
+ const pattern = /([^\s=/>]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
362
+ let match;
363
+ while ((match = pattern.exec(source)) !== null) {
364
+ const name = match[1].toLowerCase();
365
+ const value = match[2] ?? match[3] ?? match[4] ?? "";
366
+ if (!(name in attrs)) attrs[name] = decodeEntities(value);
367
+ }
368
+ return attrs;
369
+ }
370
+ /**
371
+ * Tokenizes HTML into a flat token stream.
372
+ *
373
+ * A deliberate **subset** tokenizer, not an HTML5-conformant one: it handles
374
+ * the tags the Localess model can represent, treats everything else as a
375
+ * generic element for the caller's unsupported policy, and performs no error
376
+ * recovery beyond never throwing. Chosen over branching on `DOMParser` so the
377
+ * result is identical in browsers, Node, and edge runtimes, and so the package
378
+ * keeps its zero-dependency guarantee.
379
+ *
380
+ * A stray `<` that does not begin a valid tag is emitted as text.
381
+ */
382
+ function tokenizeHtml(html) {
383
+ const tokens = [];
384
+ let index = 0;
385
+ let pendingText = "";
386
+ const flushText = () => {
387
+ if (pendingText === "") return;
388
+ tokens.push({
389
+ kind: "text",
390
+ text: decodeEntities(pendingText)
391
+ });
392
+ pendingText = "";
393
+ };
394
+ while (index < html.length) {
395
+ const next = html.indexOf("<", index);
396
+ if (next === -1) {
397
+ pendingText += html.slice(index);
398
+ break;
399
+ }
400
+ pendingText += html.slice(index, next);
401
+ const rest = html.slice(next);
402
+ if (rest.startsWith("<!--")) {
403
+ const end = html.indexOf("-->", next + 4);
404
+ index = end === -1 ? html.length : end + 3;
405
+ continue;
406
+ }
407
+ if (rest.startsWith("<!") || rest.startsWith("<?")) {
408
+ const end = html.indexOf(">", next + 1);
409
+ index = end === -1 ? html.length : end + 1;
410
+ continue;
411
+ }
412
+ const closeMatch = /^<\/\s*([a-zA-Z][^\s>]*)\s*>/.exec(rest);
413
+ if (closeMatch) {
414
+ flushText();
415
+ tokens.push({
416
+ kind: "close",
417
+ name: closeMatch[1].toLowerCase()
418
+ });
419
+ index = next + closeMatch[0].length;
420
+ continue;
421
+ }
422
+ const openMatch = /^<([a-zA-Z][^\s/>]*)((?:[^>"']|"[^"]*"|'[^']*')*)>/.exec(rest);
423
+ if (openMatch) {
424
+ flushText();
425
+ const name = openMatch[1].toLowerCase();
426
+ const raw = openMatch[2] ?? "";
427
+ const selfClosing = /\/\s*$/.test(raw) || VOID_ELEMENTS.has(name);
428
+ tokens.push({
429
+ kind: "open",
430
+ name,
431
+ attrs: parseAttributes(raw),
432
+ selfClosing
433
+ });
434
+ index = next + openMatch[0].length;
435
+ if (RAW_TEXT_ELEMENTS.has(name) && !selfClosing) {
436
+ const closeTag = `</${name}`;
437
+ const end = html.toLowerCase().indexOf(closeTag, index);
438
+ index = end === -1 ? html.length : end;
439
+ }
440
+ continue;
441
+ }
442
+ pendingText += "<";
443
+ index = next + 1;
444
+ }
445
+ flushText();
446
+ return tokens;
447
+ }
448
+ //#endregion
449
+ //#region src/html-parser/index.ts
450
+ /**
451
+ * Parses an HTML string into a Localess rich text document.
452
+ *
453
+ * The inverse of `renderRichTextToHtml`, and deliberately a **subset** parser:
454
+ * it understands the tags the Localess model can represent and routes
455
+ * everything else through {@link RichTextParseOptions.unsupported}. It is not
456
+ * HTML5-conformant and does not attempt full error recovery — the trade is
457
+ * identical behaviour across browsers, Node, and edge runtimes with no
458
+ * dependency.
459
+ *
460
+ * Supported: `p`, `h1`–`h6`, `ul`, `ol` (with `start`), `li`, `pre`/`code`
461
+ * blocks (with `language-*`), and the marks `strong`/`b`, `em`/`i`,
462
+ * `s`/`strike`/`del`, `u`, `code`, `a`. Structural wrappers such as `div` and
463
+ * `span` are transparent; `script` and `style` are always dropped.
464
+ *
465
+ * Link hrefs pass through the same allowlist the renderer applies, so
466
+ * `javascript:` and `data:` become `""`.
467
+ *
468
+ * Never throws for malformed input — except under `unsupported: 'throw'`.
469
+ *
470
+ * @example
471
+ * ```ts
472
+ * const { doc, unsupported } = parseHtmlToRichText('<p>Hello <strong>world</strong></p>');
473
+ * ```
474
+ */
475
+ function parseHtmlToRichText(html, options = {}) {
476
+ const tracker = new require_parse_common.UnsupportedTracker(options);
477
+ if (typeof html !== "string" || html === "") return {
478
+ doc: require_parse_common.emptyDocument(),
479
+ unsupported: []
480
+ };
481
+ return {
482
+ doc: {
483
+ type: "doc",
484
+ content: tokensToNodes(tokenizeHtml(html), tracker)
485
+ },
486
+ unsupported: tracker.report()
487
+ };
488
+ }
489
+ //#endregion
490
+ exports.RichTextParseError = require_parse_common.RichTextParseError;
491
+ exports.parseHtmlToRichText = parseHtmlToRichText;