@avocadostudio-ai/richtext 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/doc.d.ts CHANGED
@@ -57,6 +57,17 @@ export declare const NODE: {
57
57
  * survives a trip through the property panel in its original position.
58
58
  */
59
59
  readonly unknown: "avocadoUnknownBlock";
60
+ /**
61
+ * The same thing in inline position: a `<br class="hidden md:inline">`, an
62
+ * `<img>` or any other void element sitting inside a run of text.
63
+ *
64
+ * It is a separate name from `avocadoUnknownBlock` because ProseMirror makes
65
+ * a node inline or block and never both, and a paragraph's content is
66
+ * `inline*`. Emitting the block name here put a block node inside a
67
+ * paragraph, which is not a document the editor's schema can hold: it was
68
+ * dropped on load, taking the element with it.
69
+ */
70
+ readonly unknownInline: "avocadoUnknownInline";
60
71
  };
61
72
  /** Mark names this package produces and understands. */
62
73
  export declare const MARK: {
@@ -66,6 +77,21 @@ export declare const MARK: {
66
77
  readonly code: "code";
67
78
  readonly underline: "underline";
68
79
  readonly link: "link";
80
+ /**
81
+ * An inline element the pivot has no mark for, carried rather than dropped.
82
+ *
83
+ * The inline twin of `avocadoUnknownBlock`, and it exists for HTML: a
84
+ * template stores `<span class="hidden xl:inline">creating websites with</span>`
85
+ * as content, and neither half of that can be thrown away. The text inside
86
+ * is ordinary editable text; the wrapper rides along in `attrs` as its tag
87
+ * and its attributes in source order, and is re-emitted around whatever the
88
+ * text became.
89
+ *
90
+ * Unknown marks already survive a round trip untouched — the Storyblok
91
+ * converter relies on it for `styled` and `highlight`. This one is named so
92
+ * the HTML converter can recognise its own and rebuild the element.
93
+ */
94
+ readonly unknownHtml: "avocadoUnknownHtml";
69
95
  };
70
96
  /**
71
97
  * A richtext value stored as a document rather than a markdown string.
package/dist/doc.js CHANGED
@@ -42,7 +42,18 @@ export const NODE = {
42
42
  * unchanged. The editor registers a matching read-only node, so it also
43
43
  * survives a trip through the property panel in its original position.
44
44
  */
45
- unknown: "avocadoUnknownBlock"
45
+ unknown: "avocadoUnknownBlock",
46
+ /**
47
+ * The same thing in inline position: a `<br class="hidden md:inline">`, an
48
+ * `<img>` or any other void element sitting inside a run of text.
49
+ *
50
+ * It is a separate name from `avocadoUnknownBlock` because ProseMirror makes
51
+ * a node inline or block and never both, and a paragraph's content is
52
+ * `inline*`. Emitting the block name here put a block node inside a
53
+ * paragraph, which is not a document the editor's schema can hold: it was
54
+ * dropped on load, taking the element with it.
55
+ */
56
+ unknownInline: "avocadoUnknownInline"
46
57
  };
47
58
  /** Mark names this package produces and understands. */
48
59
  export const MARK = {
@@ -51,7 +62,22 @@ export const MARK = {
51
62
  strike: "strike",
52
63
  code: "code",
53
64
  underline: "underline",
54
- link: "link"
65
+ link: "link",
66
+ /**
67
+ * An inline element the pivot has no mark for, carried rather than dropped.
68
+ *
69
+ * The inline twin of `avocadoUnknownBlock`, and it exists for HTML: a
70
+ * template stores `<span class="hidden xl:inline">creating websites with</span>`
71
+ * as content, and neither half of that can be thrown away. The text inside
72
+ * is ordinary editable text; the wrapper rides along in `attrs` as its tag
73
+ * and its attributes in source order, and is re-emitted around whatever the
74
+ * text became.
75
+ *
76
+ * Unknown marks already survive a round trip untouched — the Storyblok
77
+ * converter relies on it for `styled` and `highlight`. This one is named so
78
+ * the HTML converter can recognise its own and rebuild the element.
79
+ */
80
+ unknownHtml: "avocadoUnknownHtml"
55
81
  };
56
82
  /**
57
83
  * A richtext value stored as a document rather than a markdown string.
package/dist/html.d.ts ADDED
@@ -0,0 +1,72 @@
1
+ /**
2
+ * HTML <-> the pivot document.
3
+ *
4
+ * The other converters in this package each serve one CMS. This one serves a
5
+ * shape no CMS produces and most templates do: a prop that is a string of HTML,
6
+ * rendered with Astro's `set:html`, React's `dangerouslySetInnerHTML` or Vue's
7
+ * `v-html`. A site that renders its own components stores those strings as
8
+ * content, because from the template's point of view that is what they are.
9
+ *
10
+ * Before this existed the only declarable kind close enough was `richtext`, and
11
+ * it is the wrong one: `richtext` means *a document*, and what arrives is a
12
+ * string. The panel therefore rendered the markup literally —
13
+ *
14
+ * Free template for <span class="hidden xl:inline">creating websites with</span> …
15
+ *
16
+ * — which a person cannot edit without breaking, and which is written back to
17
+ * the site's own source file if they edit it anyway.
18
+ *
19
+ * ## What survives
20
+ *
21
+ * Tags with a pivot equivalent become nodes and marks. Everything else is
22
+ * **preserved, not dropped**, which is the decision this file turns on. A
23
+ * converter that discarded an unrecognised tag would corrupt the source on the
24
+ * first save of a neighbouring field — the failure mode is silent, and the
25
+ * content is gone by the time anyone reads the page. So:
26
+ *
27
+ * - An unknown element wrapping text (`<span class="hidden xl:inline">…`)
28
+ * becomes a mark named `MARK.unknownHtml`, carrying its tag and its
29
+ * attributes in source order. The text inside stays editable; the wrapper
30
+ * rides along and is re-emitted around whatever the text became.
31
+ * - An unknown element with no children (`<div class="absolute inset-0 …">`,
32
+ * an `<img>`) becomes `avocadoUnknownBlock` holding its source markup. The
33
+ * panel shows it read-only, and `toHtml` re-emits it byte for byte.
34
+ *
35
+ * ## Round trip
36
+ *
37
+ * `toHtml(fromHtml(s))` is the contract, and it is *semantic* equality, not
38
+ * byte equality. It **is** idempotent from the first pass —
39
+ * `toHtml(fromHtml(toHtml(fromHtml(s))))` equals `toHtml(fromHtml(s))` — which
40
+ * is the property that actually matters, because that is what decides whether
41
+ * opening a page in the editor and saving it produces a diff for ever after.
42
+ *
43
+ * Measured against the pilot page's nineteen markup-bearing values, sixteen
44
+ * come back byte-identical and three are normalised once:
45
+ *
46
+ * - attribute quoting becomes double quotes, and a boolean attribute becomes
47
+ * `attr=""`;
48
+ * - stray whitespace inside a tag is dropped — `</span >` and `class="x" >`
49
+ * come back closed up;
50
+ * - loose text sitting after a block element is wrapped in the paragraph it
51
+ * already was — `<h3>Title</h3> Body` becomes `<h3>Title</h3><p> Body</p>`.
52
+ *
53
+ * All three are the parse writing down what the markup already meant. None of
54
+ * them loses a character of content, and none of them recurs.
55
+ *
56
+ * One deliberate asymmetry: an input with no block-level tag at all — which is
57
+ * every headline field we have seen — parses to a single paragraph and is
58
+ * emitted back **without** a `<p>` wrapper. Emitting one would mean that
59
+ * opening a page and saving it changed every title on it.
60
+ */
61
+ import { type RichTextDoc } from "./doc.ts";
62
+ export declare function decodeEntities(s: string): string;
63
+ /**
64
+ * Text, escaped for a text node.
65
+ *
66
+ * ` ` goes back out as `&nbsp;` rather than as a literal byte: it is
67
+ * invisible in a diff otherwise, and a non-breaking space in a headline is
68
+ * usually load-bearing — it is what keeps "Astro + Tailwind" off two lines.
69
+ */
70
+ export declare function escapeHtmlText(s: string): string;
71
+ export declare function fromHtml(html: unknown): RichTextDoc;
72
+ export declare function toHtml(doc: unknown): string;
package/dist/html.js ADDED
@@ -0,0 +1,471 @@
1
+ /**
2
+ * HTML <-> the pivot document.
3
+ *
4
+ * The other converters in this package each serve one CMS. This one serves a
5
+ * shape no CMS produces and most templates do: a prop that is a string of HTML,
6
+ * rendered with Astro's `set:html`, React's `dangerouslySetInnerHTML` or Vue's
7
+ * `v-html`. A site that renders its own components stores those strings as
8
+ * content, because from the template's point of view that is what they are.
9
+ *
10
+ * Before this existed the only declarable kind close enough was `richtext`, and
11
+ * it is the wrong one: `richtext` means *a document*, and what arrives is a
12
+ * string. The panel therefore rendered the markup literally —
13
+ *
14
+ * Free template for <span class="hidden xl:inline">creating websites with</span> …
15
+ *
16
+ * — which a person cannot edit without breaking, and which is written back to
17
+ * the site's own source file if they edit it anyway.
18
+ *
19
+ * ## What survives
20
+ *
21
+ * Tags with a pivot equivalent become nodes and marks. Everything else is
22
+ * **preserved, not dropped**, which is the decision this file turns on. A
23
+ * converter that discarded an unrecognised tag would corrupt the source on the
24
+ * first save of a neighbouring field — the failure mode is silent, and the
25
+ * content is gone by the time anyone reads the page. So:
26
+ *
27
+ * - An unknown element wrapping text (`<span class="hidden xl:inline">…`)
28
+ * becomes a mark named `MARK.unknownHtml`, carrying its tag and its
29
+ * attributes in source order. The text inside stays editable; the wrapper
30
+ * rides along and is re-emitted around whatever the text became.
31
+ * - An unknown element with no children (`<div class="absolute inset-0 …">`,
32
+ * an `<img>`) becomes `avocadoUnknownBlock` holding its source markup. The
33
+ * panel shows it read-only, and `toHtml` re-emits it byte for byte.
34
+ *
35
+ * ## Round trip
36
+ *
37
+ * `toHtml(fromHtml(s))` is the contract, and it is *semantic* equality, not
38
+ * byte equality. It **is** idempotent from the first pass —
39
+ * `toHtml(fromHtml(toHtml(fromHtml(s))))` equals `toHtml(fromHtml(s))` — which
40
+ * is the property that actually matters, because that is what decides whether
41
+ * opening a page in the editor and saving it produces a diff for ever after.
42
+ *
43
+ * Measured against the pilot page's nineteen markup-bearing values, sixteen
44
+ * come back byte-identical and three are normalised once:
45
+ *
46
+ * - attribute quoting becomes double quotes, and a boolean attribute becomes
47
+ * `attr=""`;
48
+ * - stray whitespace inside a tag is dropped — `</span >` and `class="x" >`
49
+ * come back closed up;
50
+ * - loose text sitting after a block element is wrapped in the paragraph it
51
+ * already was — `<h3>Title</h3> Body` becomes `<h3>Title</h3><p> Body</p>`.
52
+ *
53
+ * All three are the parse writing down what the markup already meant. None of
54
+ * them loses a character of content, and none of them recurs.
55
+ *
56
+ * One deliberate asymmetry: an input with no block-level tag at all — which is
57
+ * every headline field we have seen — parses to a single paragraph and is
58
+ * emitted back **without** a `<p>` wrapper. Emitting one would mean that
59
+ * opening a page and saving it changed every title on it.
60
+ */
61
+ import { NODE, MARK, emptyDoc } from "./doc.js";
62
+ /** Block-level tags with a pivot node, and the node they become. */
63
+ const BLOCK_TAGS = {
64
+ p: NODE.paragraph,
65
+ h1: NODE.heading, h2: NODE.heading, h3: NODE.heading,
66
+ h4: NODE.heading, h5: NODE.heading, h6: NODE.heading,
67
+ ul: NODE.bulletList,
68
+ ol: NODE.orderedList,
69
+ li: NODE.listItem,
70
+ blockquote: NODE.blockquote,
71
+ pre: NODE.codeBlock,
72
+ hr: NODE.horizontalRule
73
+ };
74
+ /** Inline tags with a pivot mark, and the mark they become. */
75
+ const MARK_TAGS = {
76
+ strong: MARK.bold, b: MARK.bold,
77
+ em: MARK.italic, i: MARK.italic,
78
+ s: MARK.strike, del: MARK.strike, strike: MARK.strike,
79
+ code: MARK.code,
80
+ u: MARK.underline,
81
+ a: MARK.link
82
+ };
83
+ /** The tag each mark is written back as. The first spelling above wins. */
84
+ const MARK_TAG_OUT = {
85
+ [MARK.bold]: "strong",
86
+ [MARK.italic]: "em",
87
+ [MARK.strike]: "s",
88
+ [MARK.code]: "code",
89
+ [MARK.underline]: "u",
90
+ [MARK.link]: "a"
91
+ };
92
+ /** Elements that never have children, whatever the source wrote. */
93
+ const VOID_TAGS = new Set(["br", "hr", "img", "input", "meta", "link", "source", "wbr", "area", "col", "embed", "track"]);
94
+ const ENTITIES = {
95
+ amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " "
96
+ };
97
+ export function decodeEntities(s) {
98
+ return s.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (whole, body) => {
99
+ if (body[0] === "#") {
100
+ const code = body[1] === "x" || body[1] === "X"
101
+ ? Number.parseInt(body.slice(2), 16)
102
+ : Number.parseInt(body.slice(1), 10);
103
+ return Number.isFinite(code) && code > 0 ? String.fromCodePoint(code) : whole;
104
+ }
105
+ return ENTITIES[body.toLowerCase()] ?? whole;
106
+ });
107
+ }
108
+ /**
109
+ * Text, escaped for a text node.
110
+ *
111
+ * ` ` goes back out as `&nbsp;` rather than as a literal byte: it is
112
+ * invisible in a diff otherwise, and a non-breaking space in a headline is
113
+ * usually load-bearing — it is what keeps "Astro + Tailwind" off two lines.
114
+ */
115
+ export function escapeHtmlText(s) {
116
+ return s
117
+ .replace(/&/g, "&amp;")
118
+ .replace(/</g, "&lt;")
119
+ .replace(/>/g, "&gt;")
120
+ .replace(/ /g, "&nbsp;");
121
+ }
122
+ function escapeAttr(s) {
123
+ return s.replace(/&/g, "&amp;").replace(/"/g, "&quot;").replace(/ /g, "&nbsp;");
124
+ }
125
+ const TAG_RE = /<(\/?)([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[^\s/>"'=]+(?:\s*=\s*(?:"[^"]*"|'[^']*'|[^\s"'=<>`]+))?)*)\s*(\/?)>/g;
126
+ const ATTR_RE = /([^\s/>"'=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
127
+ function tokenize(html) {
128
+ const tokens = [];
129
+ let last = 0;
130
+ TAG_RE.lastIndex = 0;
131
+ for (let m = TAG_RE.exec(html); m !== null; m = TAG_RE.exec(html)) {
132
+ if (m.index > last)
133
+ tokens.push({ kind: "text", text: html.slice(last, m.index) });
134
+ last = TAG_RE.lastIndex;
135
+ const tag = m[2].toLowerCase();
136
+ if (m[1] === "/") {
137
+ tokens.push({ kind: "close", tag });
138
+ continue;
139
+ }
140
+ const attrs = [];
141
+ ATTR_RE.lastIndex = 0;
142
+ for (let a = ATTR_RE.exec(m[3] ?? ""); a !== null; a = ATTR_RE.exec(m[3] ?? "")) {
143
+ attrs.push([a[1], decodeEntities(a[2] ?? a[3] ?? a[4] ?? "")]);
144
+ }
145
+ tokens.push({ kind: "open", tag, attrs, selfClosing: m[4] === "/" || VOID_TAGS.has(tag) });
146
+ }
147
+ if (last < html.length)
148
+ tokens.push({ kind: "text", text: html.slice(last) });
149
+ return tokens;
150
+ }
151
+ /**
152
+ * Attributes carried on a node or mark, rendered back.
153
+ *
154
+ * A tag the pivot *does* model can still have attributes it does not —
155
+ * `<h3 class="text-2xl font-bold …">` is a heading and the class is the whole
156
+ * of its appearance. Modelling the tag and dropping the class is the worst of
157
+ * the three outcomes: it looks like a faithful conversion and silently
158
+ * restyles the page.
159
+ */
160
+ function renderAttrs(value) {
161
+ if (!Array.isArray(value))
162
+ return "";
163
+ return value
164
+ .map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`)
165
+ .join("");
166
+ }
167
+ /** An element's source markup, for the unknown-block slot. */
168
+ function serializeUnknown(tag, attrs, selfClosing) {
169
+ const rendered = attrs.map(([k, v]) => ` ${k}="${escapeAttr(v)}"`).join("");
170
+ return selfClosing && VOID_TAGS.has(tag) ? `<${tag}${rendered} />` : `<${tag}${rendered}></${tag}>`;
171
+ }
172
+ export function fromHtml(html) {
173
+ if (typeof html !== "string" || html.trim().length === 0)
174
+ return emptyDoc();
175
+ const doc = emptyDoc();
176
+ /** Block nodes being filled, outermost first. `undefined` = top level. */
177
+ const blocks = [];
178
+ /** Inline marks currently open, innermost last. */
179
+ const marks = [];
180
+ const open = [];
181
+ const currentContainer = () => {
182
+ for (let i = blocks.length - 1; i >= 0; i--) {
183
+ const node = blocks[i];
184
+ if (!node.content)
185
+ node.content = [];
186
+ return node.content;
187
+ }
188
+ return doc.content;
189
+ };
190
+ /** Nodes that hold inline content directly; everything else needs a paragraph. */
191
+ const INLINE_CONTAINERS = new Set([NODE.paragraph, NODE.heading, NODE.codeBlock]);
192
+ /*
193
+ * The paragraph loose inline content is currently accumulating in, or null
194
+ * when nothing is open.
195
+ *
196
+ * It is the difference between "still in the same run of text" and "the
197
+ * previous sibling happened to be a paragraph", which matters when an
198
+ * element arrives that could go either way. Cleared whenever a block opens
199
+ * or closes, because either ends the run.
200
+ */
201
+ let looseParagraph = null;
202
+ /** The paragraph loose inline content belongs in, created on demand. */
203
+ const inlineTarget = () => {
204
+ const container = currentContainer();
205
+ const innermost = blocks[blocks.length - 1];
206
+ if (innermost && INLINE_CONTAINERS.has(innermost.type))
207
+ return container;
208
+ const tail = container[container.length - 1];
209
+ if (looseParagraph && tail === looseParagraph) {
210
+ if (!tail.content)
211
+ tail.content = [];
212
+ return tail.content;
213
+ }
214
+ const paragraph = { type: NODE.paragraph, content: [] };
215
+ container.push(paragraph);
216
+ looseParagraph = paragraph;
217
+ return paragraph.content;
218
+ };
219
+ /**
220
+ * An element the grammar has no node for, placed by where it was found.
221
+ *
222
+ * In inline position — inside a paragraph or a heading, or mid-run of loose
223
+ * text — it is an inline node, so it keeps its place among the words. At
224
+ * block level it is a block, a sibling of the paragraphs around it.
225
+ *
226
+ * Getting this wrong is not cosmetic. Appending a block-level `<div>` to the
227
+ * paragraph before it emits `<p>before<div></div></p>`, which is markup no
228
+ * browser keeps: the parser closes the `<p>` at the `<div>` and the page
229
+ * gains an empty paragraph on the first save of an unrelated field.
230
+ */
231
+ const pushUnknownElement = (html) => {
232
+ const innermost = blocks[blocks.length - 1];
233
+ const inlinePosition = (innermost && INLINE_CONTAINERS.has(innermost.type)) || looseParagraph !== null;
234
+ if (inlinePosition) {
235
+ inlineTarget().push({ type: NODE.unknownInline, attrs: { data: { html } } });
236
+ return;
237
+ }
238
+ currentContainer().push({ type: NODE.unknown, attrs: { data: { html } } });
239
+ };
240
+ let textCount = 0;
241
+ const pushText = (text) => {
242
+ if (text.length === 0)
243
+ return;
244
+ const node = { type: NODE.text, text };
245
+ if (marks.length > 0)
246
+ node.marks = marks.map((m) => ({ ...m }));
247
+ inlineTarget().push(node);
248
+ textCount++;
249
+ };
250
+ for (const token of tokenize(html)) {
251
+ if (token.kind === "text") {
252
+ pushText(decodeEntities(token.text));
253
+ continue;
254
+ }
255
+ if (token.kind === "close") {
256
+ for (let i = open.length - 1; i >= 0; i--) {
257
+ if (open[i].tag !== token.tag)
258
+ continue;
259
+ const frame = open.splice(i, 1)[0];
260
+ if (frame.mark) {
261
+ const at = marks.lastIndexOf(frame.mark);
262
+ if (at >= 0)
263
+ marks.splice(at, 1);
264
+ if (frame.unknownHtml !== undefined && textCount === frame.textCountAtOpen) {
265
+ pushUnknownElement(frame.unknownHtml);
266
+ }
267
+ }
268
+ else {
269
+ const at = blocks.lastIndexOf(frame.node);
270
+ if (at >= 0)
271
+ blocks.splice(at, 1);
272
+ looseParagraph = null;
273
+ }
274
+ break;
275
+ }
276
+ continue;
277
+ }
278
+ const { tag, attrs, selfClosing } = token;
279
+ if (tag === "br") {
280
+ /*
281
+ * Only a bare `<br>` is a hard break. `<br class="hidden md:inline" />`
282
+ * is a responsive layout decision, and a hard break has nowhere to keep
283
+ * the class — so it rides through whole instead. Dropping it would have
284
+ * changed where the line breaks on mobile, on the first save of an
285
+ * unrelated field, with nothing in the diff to explain it.
286
+ */
287
+ if (attrs.length === 0) {
288
+ inlineTarget().push({ type: NODE.hardBreak });
289
+ }
290
+ else {
291
+ pushUnknownElement(serializeUnknown(tag, attrs, true));
292
+ }
293
+ continue;
294
+ }
295
+ const blockType = BLOCK_TAGS[tag];
296
+ if (blockType !== undefined) {
297
+ const node = { type: blockType };
298
+ if (blockType === NODE.heading)
299
+ node.attrs = { level: Number(tag[1]) };
300
+ if (attrs.length > 0)
301
+ node.attrs = { ...(node.attrs ?? {}), htmlAttributes: attrs };
302
+ if (blockType !== NODE.horizontalRule)
303
+ node.content = [];
304
+ currentContainer().push(node);
305
+ looseParagraph = null;
306
+ if (blockType !== NODE.horizontalRule) {
307
+ blocks.push(node);
308
+ open.push({ tag, node });
309
+ }
310
+ continue;
311
+ }
312
+ const markType = MARK_TAGS[tag];
313
+ if (markType !== undefined && !selfClosing) {
314
+ const mark = { type: markType };
315
+ if (markType === MARK.link) {
316
+ const attrsOut = {};
317
+ for (const [k, v] of attrs)
318
+ attrsOut[k] = v;
319
+ mark.attrs = attrsOut;
320
+ }
321
+ else if (attrs.length > 0) {
322
+ mark.attrs = { htmlAttributes: attrs };
323
+ }
324
+ marks.push(mark);
325
+ open.push({ tag, node: { type: NODE.text }, mark });
326
+ continue;
327
+ }
328
+ /*
329
+ * Unknown. With children it is a mark, so the text inside stays editable;
330
+ * with none there is no text to hang a mark on, so it rides as a block
331
+ * carrying its own markup.
332
+ */
333
+ if (selfClosing) {
334
+ pushUnknownElement(serializeUnknown(tag, attrs, true));
335
+ continue;
336
+ }
337
+ const mark = {
338
+ type: MARK.unknownHtml,
339
+ attrs: { tag, attributes: attrs.map(([k, v]) => [k, v]) }
340
+ };
341
+ marks.push(mark);
342
+ open.push({
343
+ tag,
344
+ node: { type: NODE.text },
345
+ mark,
346
+ unknownHtml: serializeUnknown(tag, attrs, false),
347
+ textCountAtOpen: textCount
348
+ });
349
+ }
350
+ return doc;
351
+ }
352
+ // ---------------------------------------------------------------------------
353
+ // document -> html
354
+ // ---------------------------------------------------------------------------
355
+ function openTagFor(mark) {
356
+ if (mark.type === MARK.unknownHtml) {
357
+ const tag = String(mark.attrs?.tag ?? "span");
358
+ const pairs = Array.isArray(mark.attrs?.attributes) ? mark.attrs.attributes : [];
359
+ return `<${tag}${pairs.map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`).join("")}>`;
360
+ }
361
+ const tag = MARK_TAG_OUT[mark.type];
362
+ if (!tag)
363
+ return "";
364
+ if (mark.type === MARK.link) {
365
+ const attrs = (mark.attrs ?? {});
366
+ const rendered = Object.entries(attrs)
367
+ .filter(([, v]) => v !== null && v !== undefined)
368
+ .map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`)
369
+ .join("");
370
+ return `<a${rendered}>`;
371
+ }
372
+ return `<${tag}${renderAttrs(mark.attrs?.htmlAttributes)}>`;
373
+ }
374
+ function closeTagFor(mark) {
375
+ const tag = mark.type === MARK.unknownHtml
376
+ ? String(mark.attrs?.tag ?? "span")
377
+ : MARK_TAG_OUT[mark.type];
378
+ return tag ? `</${tag}>` : "";
379
+ }
380
+ function markKey(mark) {
381
+ return mark.type === MARK.unknownHtml
382
+ ? `${mark.type}:${openTagFor(mark)}`
383
+ : `${mark.type}:${JSON.stringify(mark.attrs ?? null)}`;
384
+ }
385
+ /** Inline content, reopening a mark only where it actually changes. */
386
+ function inlineToHtml(nodes) {
387
+ let out = "";
388
+ let openMarks = [];
389
+ const closeDownTo = (depth) => {
390
+ for (let i = openMarks.length - 1; i >= depth; i--)
391
+ out += closeTagFor(openMarks[i]);
392
+ openMarks = openMarks.slice(0, depth);
393
+ };
394
+ for (const node of nodes) {
395
+ if (node.type === NODE.hardBreak) {
396
+ closeDownTo(0);
397
+ out += "<br />";
398
+ continue;
399
+ }
400
+ if (node.type === NODE.unknownInline || node.type === NODE.unknown) {
401
+ closeDownTo(0);
402
+ const data = node.attrs?.data;
403
+ if (typeof data?.html === "string")
404
+ out += data.html;
405
+ continue;
406
+ }
407
+ if (node.type !== NODE.text || typeof node.text !== "string")
408
+ continue;
409
+ const wanted = node.marks ?? [];
410
+ let shared = 0;
411
+ while (shared < wanted.length && shared < openMarks.length && markKey(wanted[shared]) === markKey(openMarks[shared])) {
412
+ shared++;
413
+ }
414
+ closeDownTo(shared);
415
+ for (let i = shared; i < wanted.length; i++) {
416
+ out += openTagFor(wanted[i]);
417
+ openMarks.push(wanted[i]);
418
+ }
419
+ out += escapeHtmlText(node.text);
420
+ }
421
+ closeDownTo(0);
422
+ return out;
423
+ }
424
+ function blockToHtml(node) {
425
+ const inner = node.content ? node.content : [];
426
+ const a = renderAttrs(node.attrs?.htmlAttributes);
427
+ switch (node.type) {
428
+ case NODE.paragraph:
429
+ return `<p${a}>${inlineToHtml(inner)}</p>`;
430
+ case NODE.heading: {
431
+ const level = Number(node.attrs?.level ?? 1);
432
+ const tag = `h${level >= 1 && level <= 6 ? level : 1}`;
433
+ return `<${tag}${a}>${inlineToHtml(inner)}</${tag}>`;
434
+ }
435
+ case NODE.bulletList:
436
+ return `<ul${a}>${inner.map(blockToHtml).join("")}</ul>`;
437
+ case NODE.orderedList:
438
+ return `<ol${a}>${inner.map(blockToHtml).join("")}</ol>`;
439
+ case NODE.listItem:
440
+ return `<li${a}>${inner.map((c) => (c.type === NODE.paragraph && !c.attrs?.htmlAttributes ? inlineToHtml(c.content ?? []) : blockToHtml(c))).join("")}</li>`;
441
+ case NODE.blockquote:
442
+ return `<blockquote${a}>${inner.map(blockToHtml).join("")}</blockquote>`;
443
+ case NODE.codeBlock:
444
+ return `<pre${a}>${inlineToHtml(inner)}</pre>`;
445
+ case NODE.horizontalRule:
446
+ return `<hr${a} />`;
447
+ case NODE.unknownInline:
448
+ case NODE.unknown: {
449
+ const data = node.attrs?.data;
450
+ return typeof data?.html === "string" ? data.html : "";
451
+ }
452
+ default:
453
+ return inlineToHtml(inner);
454
+ }
455
+ }
456
+ export function toHtml(doc) {
457
+ if (typeof doc !== "object" || doc === null)
458
+ return "";
459
+ const content = doc.content;
460
+ if (!Array.isArray(content) || content.length === 0)
461
+ return "";
462
+ /*
463
+ * A field that arrived as inline markup goes back as inline markup. Wrapping
464
+ * it in the `<p>` the parse implied would mean every headline on the page
465
+ * came back changed the first time anyone opened it.
466
+ */
467
+ if (content.length === 1 && content[0].type === NODE.paragraph && !content[0].attrs?.htmlAttributes) {
468
+ return inlineToHtml(content[0].content ?? []);
469
+ }
470
+ return content.map(blockToHtml).join("");
471
+ }
package/dist/index.d.ts CHANGED
@@ -18,6 +18,7 @@
18
18
  export { parseInline, parseRichText, parseRichTextBlocks, normalizeRichTextBody, resolveRichTextHeadingLevel, clampMarkdownHeadings, unescapeMarkdownText, type InlineToken, type RichTextBlock, type RichTextList, type RichTextListItem } from "./parse.ts";
19
19
  export { NODE, MARK, isRichTextDoc, emptyDoc, fromMarkdown, toMarkdown, escapeMarkdownText, type RichTextDoc, type RichTextNode, type RichTextMark } from "./doc.ts";
20
20
  export { deepEqual, omitDeep, mergeByIdentity, mergeRichTextDoc } from "./merge.ts";
21
+ export { fromHtml, toHtml, decodeEntities, escapeHtmlText } from "./html.ts";
21
22
  export { fromPortableText, toPortableText, type PortableTextBlock, type PortableTextSpan, type PortableTextMarkDef, type ToPortableTextOptions } from "./portable-text.ts";
22
23
  export { fromContentful, toContentful, type ContentfulDocument, type ContentfulNode, type ContentfulMark } from "./contentful.ts";
23
24
  export { fromStrapiBlocks, toStrapiBlocks, type StrapiNode, type StrapiText } from "./strapi.ts";
package/dist/index.js CHANGED
@@ -18,6 +18,7 @@
18
18
  export { parseInline, parseRichText, parseRichTextBlocks, normalizeRichTextBody, resolveRichTextHeadingLevel, clampMarkdownHeadings, unescapeMarkdownText } from "./parse.js";
19
19
  export { NODE, MARK, isRichTextDoc, emptyDoc, fromMarkdown, toMarkdown, escapeMarkdownText } from "./doc.js";
20
20
  export { deepEqual, omitDeep, mergeByIdentity, mergeRichTextDoc } from "./merge.js";
21
+ export { fromHtml, toHtml, decodeEntities, escapeHtmlText } from "./html.js";
21
22
  export { fromPortableText, toPortableText } from "./portable-text.js";
22
23
  export { fromContentful, toContentful } from "./contentful.js";
23
24
  export { fromStrapiBlocks, toStrapiBlocks } from "./strapi.js";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@avocadostudio-ai/richtext",
3
- "version": "0.7.0",
3
+ "version": "0.8.0",
4
4
  "type": "module",
5
5
  "main": "dist/index.js",
6
6
  "types": "dist/index.d.ts",
@@ -12,11 +12,6 @@
12
12
  },
13
13
  "./package.json": "./package.json"
14
14
  },
15
- "repository": {
16
- "type": "git",
17
- "url": "https://github.com/avocadostudio-ai/avocado.git",
18
- "directory": "packages/richtext"
19
- },
20
15
  "publishConfig": {
21
16
  "registry": "https://registry.npmjs.org",
22
17
  "access": "public"
@@ -41,7 +36,7 @@
41
36
  "license": "Apache-2.0",
42
37
  "homepage": "https://docs.avocadostudio.dev",
43
38
  "bugs": {
44
- "url": "https://github.com/avocadostudio-ai/avocado/issues"
39
+ "url": "https://docs.avocadostudio.dev"
45
40
  },
46
41
  "scripts": {
47
42
  "build": "tsc -p tsconfig.build.json",