@avocadostudio-ai/richtext 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/doc.d.ts +26 -0
- package/dist/doc.js +28 -2
- package/dist/html.d.ts +72 -0
- package/dist/html.js +471 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/package.json +2 -7
package/dist/doc.d.ts
CHANGED
|
@@ -57,6 +57,17 @@ export declare const NODE: {
|
|
|
57
57
|
* survives a trip through the property panel in its original position.
|
|
58
58
|
*/
|
|
59
59
|
readonly unknown: "avocadoUnknownBlock";
|
|
60
|
+
/**
|
|
61
|
+
* The same thing in inline position: a `<br class="hidden md:inline">`, an
|
|
62
|
+
* `<img>` or any other void element sitting inside a run of text.
|
|
63
|
+
*
|
|
64
|
+
* It is a separate name from `avocadoUnknownBlock` because ProseMirror makes
|
|
65
|
+
* a node inline or block and never both, and a paragraph's content is
|
|
66
|
+
* `inline*`. Emitting the block name here put a block node inside a
|
|
67
|
+
* paragraph, which is not a document the editor's schema can hold: it was
|
|
68
|
+
* dropped on load, taking the element with it.
|
|
69
|
+
*/
|
|
70
|
+
readonly unknownInline: "avocadoUnknownInline";
|
|
60
71
|
};
|
|
61
72
|
/** Mark names this package produces and understands. */
|
|
62
73
|
export declare const MARK: {
|
|
@@ -66,6 +77,21 @@ export declare const MARK: {
|
|
|
66
77
|
readonly code: "code";
|
|
67
78
|
readonly underline: "underline";
|
|
68
79
|
readonly link: "link";
|
|
80
|
+
/**
|
|
81
|
+
* An inline element the pivot has no mark for, carried rather than dropped.
|
|
82
|
+
*
|
|
83
|
+
* The inline twin of `avocadoUnknownBlock`, and it exists for HTML: a
|
|
84
|
+
* template stores `<span class="hidden xl:inline">creating websites with</span>`
|
|
85
|
+
* as content, and neither half of that can be thrown away. The text inside
|
|
86
|
+
* is ordinary editable text; the wrapper rides along in `attrs` as its tag
|
|
87
|
+
* and its attributes in source order, and is re-emitted around whatever the
|
|
88
|
+
* text became.
|
|
89
|
+
*
|
|
90
|
+
* Unknown marks already survive a round trip untouched — the Storyblok
|
|
91
|
+
* converter relies on it for `styled` and `highlight`. This one is named so
|
|
92
|
+
* the HTML converter can recognise its own and rebuild the element.
|
|
93
|
+
*/
|
|
94
|
+
readonly unknownHtml: "avocadoUnknownHtml";
|
|
69
95
|
};
|
|
70
96
|
/**
|
|
71
97
|
* A richtext value stored as a document rather than a markdown string.
|
package/dist/doc.js
CHANGED
|
@@ -42,7 +42,18 @@ export const NODE = {
|
|
|
42
42
|
* unchanged. The editor registers a matching read-only node, so it also
|
|
43
43
|
* survives a trip through the property panel in its original position.
|
|
44
44
|
*/
|
|
45
|
-
unknown: "avocadoUnknownBlock"
|
|
45
|
+
unknown: "avocadoUnknownBlock",
|
|
46
|
+
/**
|
|
47
|
+
* The same thing in inline position: a `<br class="hidden md:inline">`, an
|
|
48
|
+
* `<img>` or any other void element sitting inside a run of text.
|
|
49
|
+
*
|
|
50
|
+
* It is a separate name from `avocadoUnknownBlock` because ProseMirror makes
|
|
51
|
+
* a node inline or block and never both, and a paragraph's content is
|
|
52
|
+
* `inline*`. Emitting the block name here put a block node inside a
|
|
53
|
+
* paragraph, which is not a document the editor's schema can hold: it was
|
|
54
|
+
* dropped on load, taking the element with it.
|
|
55
|
+
*/
|
|
56
|
+
unknownInline: "avocadoUnknownInline"
|
|
46
57
|
};
|
|
47
58
|
/** Mark names this package produces and understands. */
|
|
48
59
|
export const MARK = {
|
|
@@ -51,7 +62,22 @@ export const MARK = {
|
|
|
51
62
|
strike: "strike",
|
|
52
63
|
code: "code",
|
|
53
64
|
underline: "underline",
|
|
54
|
-
link: "link"
|
|
65
|
+
link: "link",
|
|
66
|
+
/**
|
|
67
|
+
* An inline element the pivot has no mark for, carried rather than dropped.
|
|
68
|
+
*
|
|
69
|
+
* The inline twin of `avocadoUnknownBlock`, and it exists for HTML: a
|
|
70
|
+
* template stores `<span class="hidden xl:inline">creating websites with</span>`
|
|
71
|
+
* as content, and neither half of that can be thrown away. The text inside
|
|
72
|
+
* is ordinary editable text; the wrapper rides along in `attrs` as its tag
|
|
73
|
+
* and its attributes in source order, and is re-emitted around whatever the
|
|
74
|
+
* text became.
|
|
75
|
+
*
|
|
76
|
+
* Unknown marks already survive a round trip untouched — the Storyblok
|
|
77
|
+
* converter relies on it for `styled` and `highlight`. This one is named so
|
|
78
|
+
* the HTML converter can recognise its own and rebuild the element.
|
|
79
|
+
*/
|
|
80
|
+
unknownHtml: "avocadoUnknownHtml"
|
|
55
81
|
};
|
|
56
82
|
/**
|
|
57
83
|
* A richtext value stored as a document rather than a markdown string.
|
package/dist/html.d.ts
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* HTML <-> the pivot document.
|
|
3
|
+
*
|
|
4
|
+
* The other converters in this package each serve one CMS. This one serves a
|
|
5
|
+
* shape no CMS produces and most templates do: a prop that is a string of HTML,
|
|
6
|
+
* rendered with Astro's `set:html`, React's `dangerouslySetInnerHTML` or Vue's
|
|
7
|
+
* `v-html`. A site that renders its own components stores those strings as
|
|
8
|
+
* content, because from the template's point of view that is what they are.
|
|
9
|
+
*
|
|
10
|
+
* Before this existed the only declarable kind close enough was `richtext`, and
|
|
11
|
+
* it is the wrong one: `richtext` means *a document*, and what arrives is a
|
|
12
|
+
* string. The panel therefore rendered the markup literally —
|
|
13
|
+
*
|
|
14
|
+
* Free template for <span class="hidden xl:inline">creating websites with</span> …
|
|
15
|
+
*
|
|
16
|
+
* — which a person cannot edit without breaking, and which is written back to
|
|
17
|
+
* the site's own source file if they edit it anyway.
|
|
18
|
+
*
|
|
19
|
+
* ## What survives
|
|
20
|
+
*
|
|
21
|
+
* Tags with a pivot equivalent become nodes and marks. Everything else is
|
|
22
|
+
* **preserved, not dropped**, which is the decision this file turns on. A
|
|
23
|
+
* converter that discarded an unrecognised tag would corrupt the source on the
|
|
24
|
+
* first save of a neighbouring field — the failure mode is silent, and the
|
|
25
|
+
* content is gone by the time anyone reads the page. So:
|
|
26
|
+
*
|
|
27
|
+
* - An unknown element wrapping text (`<span class="hidden xl:inline">…`)
|
|
28
|
+
* becomes a mark named `MARK.unknownHtml`, carrying its tag and its
|
|
29
|
+
* attributes in source order. The text inside stays editable; the wrapper
|
|
30
|
+
* rides along and is re-emitted around whatever the text became.
|
|
31
|
+
* - An unknown element with no children (`<div class="absolute inset-0 …">`,
|
|
32
|
+
* an `<img>`) becomes `avocadoUnknownBlock` holding its source markup. The
|
|
33
|
+
* panel shows it read-only, and `toHtml` re-emits it byte for byte.
|
|
34
|
+
*
|
|
35
|
+
* ## Round trip
|
|
36
|
+
*
|
|
37
|
+
* `toHtml(fromHtml(s))` is the contract, and it is *semantic* equality, not
|
|
38
|
+
* byte equality. It **is** idempotent from the first pass —
|
|
39
|
+
* `toHtml(fromHtml(toHtml(fromHtml(s))))` equals `toHtml(fromHtml(s))` — which
|
|
40
|
+
* is the property that actually matters, because that is what decides whether
|
|
41
|
+
* opening a page in the editor and saving it produces a diff for ever after.
|
|
42
|
+
*
|
|
43
|
+
* Measured against the pilot page's nineteen markup-bearing values, sixteen
|
|
44
|
+
* come back byte-identical and three are normalised once:
|
|
45
|
+
*
|
|
46
|
+
* - attribute quoting becomes double quotes, and a boolean attribute becomes
|
|
47
|
+
* `attr=""`;
|
|
48
|
+
* - stray whitespace inside a tag is dropped — `</span >` and `class="x" >`
|
|
49
|
+
* come back closed up;
|
|
50
|
+
* - loose text sitting after a block element is wrapped in the paragraph it
|
|
51
|
+
* already was — `<h3>Title</h3> Body` becomes `<h3>Title</h3><p> Body</p>`.
|
|
52
|
+
*
|
|
53
|
+
* All three are the parse writing down what the markup already meant. None of
|
|
54
|
+
* them loses a character of content, and none of them recurs.
|
|
55
|
+
*
|
|
56
|
+
* One deliberate asymmetry: an input with no block-level tag at all — which is
|
|
57
|
+
* every headline field we have seen — parses to a single paragraph and is
|
|
58
|
+
* emitted back **without** a `<p>` wrapper. Emitting one would mean that
|
|
59
|
+
* opening a page and saving it changed every title on it.
|
|
60
|
+
*/
|
|
61
|
+
import { type RichTextDoc } from "./doc.ts";
|
|
62
|
+
export declare function decodeEntities(s: string): string;
|
|
63
|
+
/**
|
|
64
|
+
* Text, escaped for a text node.
|
|
65
|
+
*
|
|
66
|
+
* ` ` goes back out as ` ` rather than as a literal byte: it is
|
|
67
|
+
* invisible in a diff otherwise, and a non-breaking space in a headline is
|
|
68
|
+
* usually load-bearing — it is what keeps "Astro + Tailwind" off two lines.
|
|
69
|
+
*/
|
|
70
|
+
export declare function escapeHtmlText(s: string): string;
|
|
71
|
+
export declare function fromHtml(html: unknown): RichTextDoc;
|
|
72
|
+
export declare function toHtml(doc: unknown): string;
|
package/dist/html.js
ADDED
|
@@ -0,0 +1,471 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* HTML <-> the pivot document.
|
|
3
|
+
*
|
|
4
|
+
* The other converters in this package each serve one CMS. This one serves a
|
|
5
|
+
* shape no CMS produces and most templates do: a prop that is a string of HTML,
|
|
6
|
+
* rendered with Astro's `set:html`, React's `dangerouslySetInnerHTML` or Vue's
|
|
7
|
+
* `v-html`. A site that renders its own components stores those strings as
|
|
8
|
+
* content, because from the template's point of view that is what they are.
|
|
9
|
+
*
|
|
10
|
+
* Before this existed the only declarable kind close enough was `richtext`, and
|
|
11
|
+
* it is the wrong one: `richtext` means *a document*, and what arrives is a
|
|
12
|
+
* string. The panel therefore rendered the markup literally —
|
|
13
|
+
*
|
|
14
|
+
* Free template for <span class="hidden xl:inline">creating websites with</span> …
|
|
15
|
+
*
|
|
16
|
+
* — which a person cannot edit without breaking, and which is written back to
|
|
17
|
+
* the site's own source file if they edit it anyway.
|
|
18
|
+
*
|
|
19
|
+
* ## What survives
|
|
20
|
+
*
|
|
21
|
+
* Tags with a pivot equivalent become nodes and marks. Everything else is
|
|
22
|
+
* **preserved, not dropped**, which is the decision this file turns on. A
|
|
23
|
+
* converter that discarded an unrecognised tag would corrupt the source on the
|
|
24
|
+
* first save of a neighbouring field — the failure mode is silent, and the
|
|
25
|
+
* content is gone by the time anyone reads the page. So:
|
|
26
|
+
*
|
|
27
|
+
* - An unknown element wrapping text (`<span class="hidden xl:inline">…`)
|
|
28
|
+
* becomes a mark named `MARK.unknownHtml`, carrying its tag and its
|
|
29
|
+
* attributes in source order. The text inside stays editable; the wrapper
|
|
30
|
+
* rides along and is re-emitted around whatever the text became.
|
|
31
|
+
* - An unknown element with no children (`<div class="absolute inset-0 …">`,
|
|
32
|
+
* an `<img>`) becomes `avocadoUnknownBlock` holding its source markup. The
|
|
33
|
+
* panel shows it read-only, and `toHtml` re-emits it byte for byte.
|
|
34
|
+
*
|
|
35
|
+
* ## Round trip
|
|
36
|
+
*
|
|
37
|
+
* `toHtml(fromHtml(s))` is the contract, and it is *semantic* equality, not
|
|
38
|
+
* byte equality. It **is** idempotent from the first pass —
|
|
39
|
+
* `toHtml(fromHtml(toHtml(fromHtml(s))))` equals `toHtml(fromHtml(s))` — which
|
|
40
|
+
* is the property that actually matters, because that is what decides whether
|
|
41
|
+
* opening a page in the editor and saving it produces a diff for ever after.
|
|
42
|
+
*
|
|
43
|
+
* Measured against the pilot page's nineteen markup-bearing values, sixteen
|
|
44
|
+
* come back byte-identical and three are normalised once:
|
|
45
|
+
*
|
|
46
|
+
* - attribute quoting becomes double quotes, and a boolean attribute becomes
|
|
47
|
+
* `attr=""`;
|
|
48
|
+
* - stray whitespace inside a tag is dropped — `</span >` and `class="x" >`
|
|
49
|
+
* come back closed up;
|
|
50
|
+
* - loose text sitting after a block element is wrapped in the paragraph it
|
|
51
|
+
* already was — `<h3>Title</h3> Body` becomes `<h3>Title</h3><p> Body</p>`.
|
|
52
|
+
*
|
|
53
|
+
* All three are the parse writing down what the markup already meant. None of
|
|
54
|
+
* them loses a character of content, and none of them recurs.
|
|
55
|
+
*
|
|
56
|
+
* One deliberate asymmetry: an input with no block-level tag at all — which is
|
|
57
|
+
* every headline field we have seen — parses to a single paragraph and is
|
|
58
|
+
* emitted back **without** a `<p>` wrapper. Emitting one would mean that
|
|
59
|
+
* opening a page and saving it changed every title on it.
|
|
60
|
+
*/
|
|
61
|
+
import { NODE, MARK, emptyDoc } from "./doc.js";
|
|
62
|
+
/** Block-level tags with a pivot node, and the node they become. */
|
|
63
|
+
const BLOCK_TAGS = {
|
|
64
|
+
p: NODE.paragraph,
|
|
65
|
+
h1: NODE.heading, h2: NODE.heading, h3: NODE.heading,
|
|
66
|
+
h4: NODE.heading, h5: NODE.heading, h6: NODE.heading,
|
|
67
|
+
ul: NODE.bulletList,
|
|
68
|
+
ol: NODE.orderedList,
|
|
69
|
+
li: NODE.listItem,
|
|
70
|
+
blockquote: NODE.blockquote,
|
|
71
|
+
pre: NODE.codeBlock,
|
|
72
|
+
hr: NODE.horizontalRule
|
|
73
|
+
};
|
|
74
|
+
/** Inline tags with a pivot mark, and the mark they become. */
|
|
75
|
+
const MARK_TAGS = {
|
|
76
|
+
strong: MARK.bold, b: MARK.bold,
|
|
77
|
+
em: MARK.italic, i: MARK.italic,
|
|
78
|
+
s: MARK.strike, del: MARK.strike, strike: MARK.strike,
|
|
79
|
+
code: MARK.code,
|
|
80
|
+
u: MARK.underline,
|
|
81
|
+
a: MARK.link
|
|
82
|
+
};
|
|
83
|
+
/** The tag each mark is written back as. The first spelling above wins. */
|
|
84
|
+
const MARK_TAG_OUT = {
|
|
85
|
+
[MARK.bold]: "strong",
|
|
86
|
+
[MARK.italic]: "em",
|
|
87
|
+
[MARK.strike]: "s",
|
|
88
|
+
[MARK.code]: "code",
|
|
89
|
+
[MARK.underline]: "u",
|
|
90
|
+
[MARK.link]: "a"
|
|
91
|
+
};
|
|
92
|
+
/** Elements that never have children, whatever the source wrote. */
|
|
93
|
+
const VOID_TAGS = new Set(["br", "hr", "img", "input", "meta", "link", "source", "wbr", "area", "col", "embed", "track"]);
|
|
94
|
+
const ENTITIES = {
|
|
95
|
+
amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " "
|
|
96
|
+
};
|
|
97
|
+
export function decodeEntities(s) {
|
|
98
|
+
return s.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (whole, body) => {
|
|
99
|
+
if (body[0] === "#") {
|
|
100
|
+
const code = body[1] === "x" || body[1] === "X"
|
|
101
|
+
? Number.parseInt(body.slice(2), 16)
|
|
102
|
+
: Number.parseInt(body.slice(1), 10);
|
|
103
|
+
return Number.isFinite(code) && code > 0 ? String.fromCodePoint(code) : whole;
|
|
104
|
+
}
|
|
105
|
+
return ENTITIES[body.toLowerCase()] ?? whole;
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* Text, escaped for a text node.
|
|
110
|
+
*
|
|
111
|
+
* ` ` goes back out as ` ` rather than as a literal byte: it is
|
|
112
|
+
* invisible in a diff otherwise, and a non-breaking space in a headline is
|
|
113
|
+
* usually load-bearing — it is what keeps "Astro + Tailwind" off two lines.
|
|
114
|
+
*/
|
|
115
|
+
export function escapeHtmlText(s) {
|
|
116
|
+
return s
|
|
117
|
+
.replace(/&/g, "&")
|
|
118
|
+
.replace(/</g, "<")
|
|
119
|
+
.replace(/>/g, ">")
|
|
120
|
+
.replace(/ /g, " ");
|
|
121
|
+
}
|
|
122
|
+
function escapeAttr(s) {
|
|
123
|
+
return s.replace(/&/g, "&").replace(/"/g, """).replace(/ /g, " ");
|
|
124
|
+
}
|
|
125
|
+
const TAG_RE = /<(\/?)([a-zA-Z][a-zA-Z0-9-]*)((?:\s+[^\s/>"'=]+(?:\s*=\s*(?:"[^"]*"|'[^']*'|[^\s"'=<>`]+))?)*)\s*(\/?)>/g;
|
|
126
|
+
const ATTR_RE = /([^\s/>"'=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
|
|
127
|
+
function tokenize(html) {
|
|
128
|
+
const tokens = [];
|
|
129
|
+
let last = 0;
|
|
130
|
+
TAG_RE.lastIndex = 0;
|
|
131
|
+
for (let m = TAG_RE.exec(html); m !== null; m = TAG_RE.exec(html)) {
|
|
132
|
+
if (m.index > last)
|
|
133
|
+
tokens.push({ kind: "text", text: html.slice(last, m.index) });
|
|
134
|
+
last = TAG_RE.lastIndex;
|
|
135
|
+
const tag = m[2].toLowerCase();
|
|
136
|
+
if (m[1] === "/") {
|
|
137
|
+
tokens.push({ kind: "close", tag });
|
|
138
|
+
continue;
|
|
139
|
+
}
|
|
140
|
+
const attrs = [];
|
|
141
|
+
ATTR_RE.lastIndex = 0;
|
|
142
|
+
for (let a = ATTR_RE.exec(m[3] ?? ""); a !== null; a = ATTR_RE.exec(m[3] ?? "")) {
|
|
143
|
+
attrs.push([a[1], decodeEntities(a[2] ?? a[3] ?? a[4] ?? "")]);
|
|
144
|
+
}
|
|
145
|
+
tokens.push({ kind: "open", tag, attrs, selfClosing: m[4] === "/" || VOID_TAGS.has(tag) });
|
|
146
|
+
}
|
|
147
|
+
if (last < html.length)
|
|
148
|
+
tokens.push({ kind: "text", text: html.slice(last) });
|
|
149
|
+
return tokens;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Attributes carried on a node or mark, rendered back.
|
|
153
|
+
*
|
|
154
|
+
* A tag the pivot *does* model can still have attributes it does not —
|
|
155
|
+
* `<h3 class="text-2xl font-bold …">` is a heading and the class is the whole
|
|
156
|
+
* of its appearance. Modelling the tag and dropping the class is the worst of
|
|
157
|
+
* the three outcomes: it looks like a faithful conversion and silently
|
|
158
|
+
* restyles the page.
|
|
159
|
+
*/
|
|
160
|
+
function renderAttrs(value) {
|
|
161
|
+
if (!Array.isArray(value))
|
|
162
|
+
return "";
|
|
163
|
+
return value
|
|
164
|
+
.map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`)
|
|
165
|
+
.join("");
|
|
166
|
+
}
|
|
167
|
+
/** An element's source markup, for the unknown-block slot. */
|
|
168
|
+
function serializeUnknown(tag, attrs, selfClosing) {
|
|
169
|
+
const rendered = attrs.map(([k, v]) => ` ${k}="${escapeAttr(v)}"`).join("");
|
|
170
|
+
return selfClosing && VOID_TAGS.has(tag) ? `<${tag}${rendered} />` : `<${tag}${rendered}></${tag}>`;
|
|
171
|
+
}
|
|
172
|
+
export function fromHtml(html) {
|
|
173
|
+
if (typeof html !== "string" || html.trim().length === 0)
|
|
174
|
+
return emptyDoc();
|
|
175
|
+
const doc = emptyDoc();
|
|
176
|
+
/** Block nodes being filled, outermost first. `undefined` = top level. */
|
|
177
|
+
const blocks = [];
|
|
178
|
+
/** Inline marks currently open, innermost last. */
|
|
179
|
+
const marks = [];
|
|
180
|
+
const open = [];
|
|
181
|
+
const currentContainer = () => {
|
|
182
|
+
for (let i = blocks.length - 1; i >= 0; i--) {
|
|
183
|
+
const node = blocks[i];
|
|
184
|
+
if (!node.content)
|
|
185
|
+
node.content = [];
|
|
186
|
+
return node.content;
|
|
187
|
+
}
|
|
188
|
+
return doc.content;
|
|
189
|
+
};
|
|
190
|
+
/** Nodes that hold inline content directly; everything else needs a paragraph. */
|
|
191
|
+
const INLINE_CONTAINERS = new Set([NODE.paragraph, NODE.heading, NODE.codeBlock]);
|
|
192
|
+
/*
|
|
193
|
+
* The paragraph loose inline content is currently accumulating in, or null
|
|
194
|
+
* when nothing is open.
|
|
195
|
+
*
|
|
196
|
+
* It is the difference between "still in the same run of text" and "the
|
|
197
|
+
* previous sibling happened to be a paragraph", which matters when an
|
|
198
|
+
* element arrives that could go either way. Cleared whenever a block opens
|
|
199
|
+
* or closes, because either ends the run.
|
|
200
|
+
*/
|
|
201
|
+
let looseParagraph = null;
|
|
202
|
+
/** The paragraph loose inline content belongs in, created on demand. */
|
|
203
|
+
const inlineTarget = () => {
|
|
204
|
+
const container = currentContainer();
|
|
205
|
+
const innermost = blocks[blocks.length - 1];
|
|
206
|
+
if (innermost && INLINE_CONTAINERS.has(innermost.type))
|
|
207
|
+
return container;
|
|
208
|
+
const tail = container[container.length - 1];
|
|
209
|
+
if (looseParagraph && tail === looseParagraph) {
|
|
210
|
+
if (!tail.content)
|
|
211
|
+
tail.content = [];
|
|
212
|
+
return tail.content;
|
|
213
|
+
}
|
|
214
|
+
const paragraph = { type: NODE.paragraph, content: [] };
|
|
215
|
+
container.push(paragraph);
|
|
216
|
+
looseParagraph = paragraph;
|
|
217
|
+
return paragraph.content;
|
|
218
|
+
};
|
|
219
|
+
/**
|
|
220
|
+
* An element the grammar has no node for, placed by where it was found.
|
|
221
|
+
*
|
|
222
|
+
* In inline position — inside a paragraph or a heading, or mid-run of loose
|
|
223
|
+
* text — it is an inline node, so it keeps its place among the words. At
|
|
224
|
+
* block level it is a block, a sibling of the paragraphs around it.
|
|
225
|
+
*
|
|
226
|
+
* Getting this wrong is not cosmetic. Appending a block-level `<div>` to the
|
|
227
|
+
* paragraph before it emits `<p>before<div></div></p>`, which is markup no
|
|
228
|
+
* browser keeps: the parser closes the `<p>` at the `<div>` and the page
|
|
229
|
+
* gains an empty paragraph on the first save of an unrelated field.
|
|
230
|
+
*/
|
|
231
|
+
const pushUnknownElement = (html) => {
|
|
232
|
+
const innermost = blocks[blocks.length - 1];
|
|
233
|
+
const inlinePosition = (innermost && INLINE_CONTAINERS.has(innermost.type)) || looseParagraph !== null;
|
|
234
|
+
if (inlinePosition) {
|
|
235
|
+
inlineTarget().push({ type: NODE.unknownInline, attrs: { data: { html } } });
|
|
236
|
+
return;
|
|
237
|
+
}
|
|
238
|
+
currentContainer().push({ type: NODE.unknown, attrs: { data: { html } } });
|
|
239
|
+
};
|
|
240
|
+
let textCount = 0;
|
|
241
|
+
const pushText = (text) => {
|
|
242
|
+
if (text.length === 0)
|
|
243
|
+
return;
|
|
244
|
+
const node = { type: NODE.text, text };
|
|
245
|
+
if (marks.length > 0)
|
|
246
|
+
node.marks = marks.map((m) => ({ ...m }));
|
|
247
|
+
inlineTarget().push(node);
|
|
248
|
+
textCount++;
|
|
249
|
+
};
|
|
250
|
+
for (const token of tokenize(html)) {
|
|
251
|
+
if (token.kind === "text") {
|
|
252
|
+
pushText(decodeEntities(token.text));
|
|
253
|
+
continue;
|
|
254
|
+
}
|
|
255
|
+
if (token.kind === "close") {
|
|
256
|
+
for (let i = open.length - 1; i >= 0; i--) {
|
|
257
|
+
if (open[i].tag !== token.tag)
|
|
258
|
+
continue;
|
|
259
|
+
const frame = open.splice(i, 1)[0];
|
|
260
|
+
if (frame.mark) {
|
|
261
|
+
const at = marks.lastIndexOf(frame.mark);
|
|
262
|
+
if (at >= 0)
|
|
263
|
+
marks.splice(at, 1);
|
|
264
|
+
if (frame.unknownHtml !== undefined && textCount === frame.textCountAtOpen) {
|
|
265
|
+
pushUnknownElement(frame.unknownHtml);
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
else {
|
|
269
|
+
const at = blocks.lastIndexOf(frame.node);
|
|
270
|
+
if (at >= 0)
|
|
271
|
+
blocks.splice(at, 1);
|
|
272
|
+
looseParagraph = null;
|
|
273
|
+
}
|
|
274
|
+
break;
|
|
275
|
+
}
|
|
276
|
+
continue;
|
|
277
|
+
}
|
|
278
|
+
const { tag, attrs, selfClosing } = token;
|
|
279
|
+
if (tag === "br") {
|
|
280
|
+
/*
|
|
281
|
+
* Only a bare `<br>` is a hard break. `<br class="hidden md:inline" />`
|
|
282
|
+
* is a responsive layout decision, and a hard break has nowhere to keep
|
|
283
|
+
* the class — so it rides through whole instead. Dropping it would have
|
|
284
|
+
* changed where the line breaks on mobile, on the first save of an
|
|
285
|
+
* unrelated field, with nothing in the diff to explain it.
|
|
286
|
+
*/
|
|
287
|
+
if (attrs.length === 0) {
|
|
288
|
+
inlineTarget().push({ type: NODE.hardBreak });
|
|
289
|
+
}
|
|
290
|
+
else {
|
|
291
|
+
pushUnknownElement(serializeUnknown(tag, attrs, true));
|
|
292
|
+
}
|
|
293
|
+
continue;
|
|
294
|
+
}
|
|
295
|
+
const blockType = BLOCK_TAGS[tag];
|
|
296
|
+
if (blockType !== undefined) {
|
|
297
|
+
const node = { type: blockType };
|
|
298
|
+
if (blockType === NODE.heading)
|
|
299
|
+
node.attrs = { level: Number(tag[1]) };
|
|
300
|
+
if (attrs.length > 0)
|
|
301
|
+
node.attrs = { ...(node.attrs ?? {}), htmlAttributes: attrs };
|
|
302
|
+
if (blockType !== NODE.horizontalRule)
|
|
303
|
+
node.content = [];
|
|
304
|
+
currentContainer().push(node);
|
|
305
|
+
looseParagraph = null;
|
|
306
|
+
if (blockType !== NODE.horizontalRule) {
|
|
307
|
+
blocks.push(node);
|
|
308
|
+
open.push({ tag, node });
|
|
309
|
+
}
|
|
310
|
+
continue;
|
|
311
|
+
}
|
|
312
|
+
const markType = MARK_TAGS[tag];
|
|
313
|
+
if (markType !== undefined && !selfClosing) {
|
|
314
|
+
const mark = { type: markType };
|
|
315
|
+
if (markType === MARK.link) {
|
|
316
|
+
const attrsOut = {};
|
|
317
|
+
for (const [k, v] of attrs)
|
|
318
|
+
attrsOut[k] = v;
|
|
319
|
+
mark.attrs = attrsOut;
|
|
320
|
+
}
|
|
321
|
+
else if (attrs.length > 0) {
|
|
322
|
+
mark.attrs = { htmlAttributes: attrs };
|
|
323
|
+
}
|
|
324
|
+
marks.push(mark);
|
|
325
|
+
open.push({ tag, node: { type: NODE.text }, mark });
|
|
326
|
+
continue;
|
|
327
|
+
}
|
|
328
|
+
/*
|
|
329
|
+
* Unknown. With children it is a mark, so the text inside stays editable;
|
|
330
|
+
* with none there is no text to hang a mark on, so it rides as a block
|
|
331
|
+
* carrying its own markup.
|
|
332
|
+
*/
|
|
333
|
+
if (selfClosing) {
|
|
334
|
+
pushUnknownElement(serializeUnknown(tag, attrs, true));
|
|
335
|
+
continue;
|
|
336
|
+
}
|
|
337
|
+
const mark = {
|
|
338
|
+
type: MARK.unknownHtml,
|
|
339
|
+
attrs: { tag, attributes: attrs.map(([k, v]) => [k, v]) }
|
|
340
|
+
};
|
|
341
|
+
marks.push(mark);
|
|
342
|
+
open.push({
|
|
343
|
+
tag,
|
|
344
|
+
node: { type: NODE.text },
|
|
345
|
+
mark,
|
|
346
|
+
unknownHtml: serializeUnknown(tag, attrs, false),
|
|
347
|
+
textCountAtOpen: textCount
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
return doc;
|
|
351
|
+
}
|
|
352
|
+
// ---------------------------------------------------------------------------
|
|
353
|
+
// document -> html
|
|
354
|
+
// ---------------------------------------------------------------------------
|
|
355
|
+
function openTagFor(mark) {
|
|
356
|
+
if (mark.type === MARK.unknownHtml) {
|
|
357
|
+
const tag = String(mark.attrs?.tag ?? "span");
|
|
358
|
+
const pairs = Array.isArray(mark.attrs?.attributes) ? mark.attrs.attributes : [];
|
|
359
|
+
return `<${tag}${pairs.map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`).join("")}>`;
|
|
360
|
+
}
|
|
361
|
+
const tag = MARK_TAG_OUT[mark.type];
|
|
362
|
+
if (!tag)
|
|
363
|
+
return "";
|
|
364
|
+
if (mark.type === MARK.link) {
|
|
365
|
+
const attrs = (mark.attrs ?? {});
|
|
366
|
+
const rendered = Object.entries(attrs)
|
|
367
|
+
.filter(([, v]) => v !== null && v !== undefined)
|
|
368
|
+
.map(([k, v]) => ` ${k}="${escapeAttr(String(v))}"`)
|
|
369
|
+
.join("");
|
|
370
|
+
return `<a${rendered}>`;
|
|
371
|
+
}
|
|
372
|
+
return `<${tag}${renderAttrs(mark.attrs?.htmlAttributes)}>`;
|
|
373
|
+
}
|
|
374
|
+
function closeTagFor(mark) {
|
|
375
|
+
const tag = mark.type === MARK.unknownHtml
|
|
376
|
+
? String(mark.attrs?.tag ?? "span")
|
|
377
|
+
: MARK_TAG_OUT[mark.type];
|
|
378
|
+
return tag ? `</${tag}>` : "";
|
|
379
|
+
}
|
|
380
|
+
function markKey(mark) {
|
|
381
|
+
return mark.type === MARK.unknownHtml
|
|
382
|
+
? `${mark.type}:${openTagFor(mark)}`
|
|
383
|
+
: `${mark.type}:${JSON.stringify(mark.attrs ?? null)}`;
|
|
384
|
+
}
|
|
385
|
+
/** Inline content, reopening a mark only where it actually changes. */
|
|
386
|
+
function inlineToHtml(nodes) {
|
|
387
|
+
let out = "";
|
|
388
|
+
let openMarks = [];
|
|
389
|
+
const closeDownTo = (depth) => {
|
|
390
|
+
for (let i = openMarks.length - 1; i >= depth; i--)
|
|
391
|
+
out += closeTagFor(openMarks[i]);
|
|
392
|
+
openMarks = openMarks.slice(0, depth);
|
|
393
|
+
};
|
|
394
|
+
for (const node of nodes) {
|
|
395
|
+
if (node.type === NODE.hardBreak) {
|
|
396
|
+
closeDownTo(0);
|
|
397
|
+
out += "<br />";
|
|
398
|
+
continue;
|
|
399
|
+
}
|
|
400
|
+
if (node.type === NODE.unknownInline || node.type === NODE.unknown) {
|
|
401
|
+
closeDownTo(0);
|
|
402
|
+
const data = node.attrs?.data;
|
|
403
|
+
if (typeof data?.html === "string")
|
|
404
|
+
out += data.html;
|
|
405
|
+
continue;
|
|
406
|
+
}
|
|
407
|
+
if (node.type !== NODE.text || typeof node.text !== "string")
|
|
408
|
+
continue;
|
|
409
|
+
const wanted = node.marks ?? [];
|
|
410
|
+
let shared = 0;
|
|
411
|
+
while (shared < wanted.length && shared < openMarks.length && markKey(wanted[shared]) === markKey(openMarks[shared])) {
|
|
412
|
+
shared++;
|
|
413
|
+
}
|
|
414
|
+
closeDownTo(shared);
|
|
415
|
+
for (let i = shared; i < wanted.length; i++) {
|
|
416
|
+
out += openTagFor(wanted[i]);
|
|
417
|
+
openMarks.push(wanted[i]);
|
|
418
|
+
}
|
|
419
|
+
out += escapeHtmlText(node.text);
|
|
420
|
+
}
|
|
421
|
+
closeDownTo(0);
|
|
422
|
+
return out;
|
|
423
|
+
}
|
|
424
|
+
function blockToHtml(node) {
|
|
425
|
+
const inner = node.content ? node.content : [];
|
|
426
|
+
const a = renderAttrs(node.attrs?.htmlAttributes);
|
|
427
|
+
switch (node.type) {
|
|
428
|
+
case NODE.paragraph:
|
|
429
|
+
return `<p${a}>${inlineToHtml(inner)}</p>`;
|
|
430
|
+
case NODE.heading: {
|
|
431
|
+
const level = Number(node.attrs?.level ?? 1);
|
|
432
|
+
const tag = `h${level >= 1 && level <= 6 ? level : 1}`;
|
|
433
|
+
return `<${tag}${a}>${inlineToHtml(inner)}</${tag}>`;
|
|
434
|
+
}
|
|
435
|
+
case NODE.bulletList:
|
|
436
|
+
return `<ul${a}>${inner.map(blockToHtml).join("")}</ul>`;
|
|
437
|
+
case NODE.orderedList:
|
|
438
|
+
return `<ol${a}>${inner.map(blockToHtml).join("")}</ol>`;
|
|
439
|
+
case NODE.listItem:
|
|
440
|
+
return `<li${a}>${inner.map((c) => (c.type === NODE.paragraph && !c.attrs?.htmlAttributes ? inlineToHtml(c.content ?? []) : blockToHtml(c))).join("")}</li>`;
|
|
441
|
+
case NODE.blockquote:
|
|
442
|
+
return `<blockquote${a}>${inner.map(blockToHtml).join("")}</blockquote>`;
|
|
443
|
+
case NODE.codeBlock:
|
|
444
|
+
return `<pre${a}>${inlineToHtml(inner)}</pre>`;
|
|
445
|
+
case NODE.horizontalRule:
|
|
446
|
+
return `<hr${a} />`;
|
|
447
|
+
case NODE.unknownInline:
|
|
448
|
+
case NODE.unknown: {
|
|
449
|
+
const data = node.attrs?.data;
|
|
450
|
+
return typeof data?.html === "string" ? data.html : "";
|
|
451
|
+
}
|
|
452
|
+
default:
|
|
453
|
+
return inlineToHtml(inner);
|
|
454
|
+
}
|
|
455
|
+
}
|
|
456
|
+
export function toHtml(doc) {
|
|
457
|
+
if (typeof doc !== "object" || doc === null)
|
|
458
|
+
return "";
|
|
459
|
+
const content = doc.content;
|
|
460
|
+
if (!Array.isArray(content) || content.length === 0)
|
|
461
|
+
return "";
|
|
462
|
+
/*
|
|
463
|
+
* A field that arrived as inline markup goes back as inline markup. Wrapping
|
|
464
|
+
* it in the `<p>` the parse implied would mean every headline on the page
|
|
465
|
+
* came back changed the first time anyone opened it.
|
|
466
|
+
*/
|
|
467
|
+
if (content.length === 1 && content[0].type === NODE.paragraph && !content[0].attrs?.htmlAttributes) {
|
|
468
|
+
return inlineToHtml(content[0].content ?? []);
|
|
469
|
+
}
|
|
470
|
+
return content.map(blockToHtml).join("");
|
|
471
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -18,6 +18,7 @@
|
|
|
18
18
|
export { parseInline, parseRichText, parseRichTextBlocks, normalizeRichTextBody, resolveRichTextHeadingLevel, clampMarkdownHeadings, unescapeMarkdownText, type InlineToken, type RichTextBlock, type RichTextList, type RichTextListItem } from "./parse.ts";
|
|
19
19
|
export { NODE, MARK, isRichTextDoc, emptyDoc, fromMarkdown, toMarkdown, escapeMarkdownText, type RichTextDoc, type RichTextNode, type RichTextMark } from "./doc.ts";
|
|
20
20
|
export { deepEqual, omitDeep, mergeByIdentity, mergeRichTextDoc } from "./merge.ts";
|
|
21
|
+
export { fromHtml, toHtml, decodeEntities, escapeHtmlText } from "./html.ts";
|
|
21
22
|
export { fromPortableText, toPortableText, type PortableTextBlock, type PortableTextSpan, type PortableTextMarkDef, type ToPortableTextOptions } from "./portable-text.ts";
|
|
22
23
|
export { fromContentful, toContentful, type ContentfulDocument, type ContentfulNode, type ContentfulMark } from "./contentful.ts";
|
|
23
24
|
export { fromStrapiBlocks, toStrapiBlocks, type StrapiNode, type StrapiText } from "./strapi.ts";
|
package/dist/index.js
CHANGED
|
@@ -18,6 +18,7 @@
|
|
|
18
18
|
export { parseInline, parseRichText, parseRichTextBlocks, normalizeRichTextBody, resolveRichTextHeadingLevel, clampMarkdownHeadings, unescapeMarkdownText } from "./parse.js";
|
|
19
19
|
export { NODE, MARK, isRichTextDoc, emptyDoc, fromMarkdown, toMarkdown, escapeMarkdownText } from "./doc.js";
|
|
20
20
|
export { deepEqual, omitDeep, mergeByIdentity, mergeRichTextDoc } from "./merge.js";
|
|
21
|
+
export { fromHtml, toHtml, decodeEntities, escapeHtmlText } from "./html.js";
|
|
21
22
|
export { fromPortableText, toPortableText } from "./portable-text.js";
|
|
22
23
|
export { fromContentful, toContentful } from "./contentful.js";
|
|
23
24
|
export { fromStrapiBlocks, toStrapiBlocks } from "./strapi.js";
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@avocadostudio-ai/richtext",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
@@ -12,11 +12,6 @@
|
|
|
12
12
|
},
|
|
13
13
|
"./package.json": "./package.json"
|
|
14
14
|
},
|
|
15
|
-
"repository": {
|
|
16
|
-
"type": "git",
|
|
17
|
-
"url": "https://github.com/avocadostudio-ai/avocado.git",
|
|
18
|
-
"directory": "packages/richtext"
|
|
19
|
-
},
|
|
20
15
|
"publishConfig": {
|
|
21
16
|
"registry": "https://registry.npmjs.org",
|
|
22
17
|
"access": "public"
|
|
@@ -41,7 +36,7 @@
|
|
|
41
36
|
"license": "Apache-2.0",
|
|
42
37
|
"homepage": "https://docs.avocadostudio.dev",
|
|
43
38
|
"bugs": {
|
|
44
|
-
"url": "https://
|
|
39
|
+
"url": "https://docs.avocadostudio.dev"
|
|
45
40
|
},
|
|
46
41
|
"scripts": {
|
|
47
42
|
"build": "tsc -p tsconfig.build.json",
|