smartrte-core 1.0.0-beta.9 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +1 -0
- package/dist/foundation/atom/formats.d.ts +0 -11
- package/dist/foundation/atom/formats.js +38 -8
- package/dist/foundation/atom/imageExtras.d.ts +48 -0
- package/dist/foundation/atom/imageExtras.js +88 -0
- package/dist/foundation/atom/index.d.ts +1 -0
- package/dist/foundation/atom/index.js +1 -0
- package/dist/foundation/atom/schema.js +21 -2
- package/dist/foundation/block/schema.js +10 -1
- package/dist/foundation/clipboard/pipeline.js +1 -1
- package/dist/foundation/clipboard/serialization.js +5 -2
- package/dist/foundation/clipboard/types.d.ts +2 -0
- package/dist/foundation/formats/docx/export.js +12 -2
- package/dist/foundation/formats/fidelity.d.ts +1 -1
- package/dist/foundation/formats/fidelity.js +10 -1
- package/dist/foundation/index.d.ts +1 -0
- package/dist/foundation/index.js +1 -0
- package/dist/foundation/list/formats.d.ts +12 -1
- package/dist/foundation/list/formats.js +116 -13
- package/dist/foundation/modelDom.d.ts +12 -0
- package/dist/foundation/modelDom.js +45 -3
- package/dist/foundation/query/index.d.ts +1 -0
- package/dist/foundation/query/index.js +1 -0
- package/dist/foundation/query/sectionContext.d.ts +41 -0
- package/dist/foundation/query/sectionContext.js +94 -0
- package/dist/foundation/surface/input.js +9 -3
- package/dist/foundation/surface/renderer.d.ts +21 -1
- package/dist/foundation/surface/renderer.js +99 -7
- package/dist/foundation/surface/types.d.ts +11 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,26 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.2.0
|
|
4
|
+
|
|
5
|
+
All changes are additive: documents without captions or metadata export byte-identically to 1.1.x (HTML, clipboard HTML/plain text, Markdown, DOCX), checked against golden output from the 1.1.x source.
|
|
6
|
+
|
|
7
|
+
- **Image captions.** `block_image` gains an optional plain-text `caption` attribute (normalised: control characters stripped, whitespace collapsed, at most 500 characters). A captioned image exports as `<figure data-smart-figure="true">(<a>)<img …><figcaption>…</figcaption></figure>`, with every attribute still on the `<img>`. It also exports to Markdown (an italic line under the image), DOCX (a Caption-style paragraph) and plain text (`alt - caption`). On import, any `<figure>` with one image and one `<figcaption>` becomes a captioned image; marks inside the caption are flattened. A figcaption with no image becomes a paragraph instead of `[Unsupported: figcaption]`. The live renderer shows the caption as a non-editable projection directly under the image. See `docs/bugs/pasted-web-images-and-hr-render-as-unsupported.md`.
|
|
8
|
+
- **Heads-up, changes how existing content parses:** HTML that earlier versions saved as a loose `<img>` followed by `<figcaption>` (a pasted `<figure>` flattened on save) is now repaired on load. The figcaption becomes the image's caption, and the next save rewrites it as a `<figure>`. See `docs/bugs/figure-wrapper-lost-on-save.md`. Stored JSON documents still hold the old `unknown` figcaption node; the repair runs on HTML parsing only.
|
|
9
|
+
- **Host image metadata.** `block_image` gains an optional `metadata` map of host-defined `data-*` attributes (at most 10; values at most 500 characters; the `data-smart-` namespace is reserved). It's exported as escaped attributes on the `<img>`. `parseCanonicalListHtml(html, { imageMetadataAttributes })` (a new optional options parameter, also threaded through paste/drop via `CanonicalInputPipelineOptions.imageMetadataAttributes`) keeps only the names you allowlist; the default keeps none, as before.
|
|
10
|
+
- **Document queries:** `getSectionContext(document, selection, { maxChars })` returns the nearest top-level heading at the caret and that section's plain text. `listDocumentImages(document)` lists every image in document order.
|
|
11
|
+
- `CanonicalSubtreeRenderer.render` takes an optional `{ syncDomSelection }`, which renders without moving the browser selection (and therefore focus) into the editor.
|
|
12
|
+
- Fix: a block image's alignment was lost on every HTML save → reload (`data-smart-align` was written but never read back). See `docs/bugs/block-image-align-lost-on-html-reload.md`.
|
|
13
|
+
- Fix (latent): model/DOM offset mapping counted renderer projections (table caption/colgroup, checklist controls) as block children. See `docs/bugs/block-boundary-offsets-counted-renderer-projections.md`.
|
|
14
|
+
- New `image-captions` row in `builtInFormatFidelity`.
|
|
15
|
+
|
|
16
|
+
## 1.1.0
|
|
17
|
+
|
|
18
|
+
- Add `backgroundColor`, `textColor`, and `borderLeft` attrs to `blockquote` (the last a single composed CSS shorthand, since a blockquote only ever shows one visible border side, unlike `table_cell`'s 4-sided borders) — rendered as real CSS in the same change, alongside the existing `align`/`indentLevel`/`lineHeight` block attrs. See `docs/bugs/blockquote-styling-context-menu.md`.
|
|
19
|
+
|
|
20
|
+
## 1.0.0
|
|
21
|
+
|
|
22
|
+
General availability of the canonical/foundation architecture introduced in `1.0.0-beta.1` — promoted from the `beta` npm dist-tag to `latest`. No functional changes beyond the `1.0.0-beta.2`–`beta.9` entries below; see those for the full list of changes across the beta cycle.
|
|
23
|
+
|
|
3
24
|
## 1.0.0-beta.9
|
|
4
25
|
|
|
5
26
|
- Fix converting a paragraph sitting directly after an existing list into a list of the same kind always creating a second, independently-numbered list instead of continuing the existing one — e.g. typing after exiting a numbered list (Enter twice) and clicking "Numbered list" again now correctly appends as the next item rather than restarting at "1." next to it. See `docs/bugs/list-creation-ignores-adjacent-identical-list.md`.
|
package/README.md
CHANGED
|
@@ -30,6 +30,7 @@ import {
|
|
|
30
30
|
- **Browser input surface** (`surface/`) — the `beforeinput`/paste/cut/drop/composition handling layer a real contentEditable-backed UI wires up to; this is what `smartrte-react` builds its editing surface on.
|
|
31
31
|
- **Format adapters** (`formats/`) — HTML, Markdown, DOCX (via `mammoth`/`@xmldom`), and PDF (`pdfjs-dist`) import/export, each declaring an explicit fidelity contract (`builtInFormatFidelity`) for exactly what's lossless vs. lossy per format/feature pair.
|
|
32
32
|
- **Diff, versioning, comments, suggestions, plugin system** (`diff/`, `versioning/`, `comments/`, `suggestions/`, `plugin/`) — the primitives `smartrte-react`'s version history, comment threads, and track-changes UI are built on. A `PluginRegistry` (`createPluginRegistry`, `builtInPlugins`) is how new commands, keyboard shortcuts, toolbar contributions, and renderer behavior are added — see [`docs/PLUGIN_ARCHITECTURE.md`](https://github.com/ayush1852017/smart-rte/blob/master/docs/PLUGIN_ARCHITECTURE.md) in the repository.
|
|
33
|
+
- **Images: captions, host metadata, document queries** (`atom/imageExtras`, `query/`) — `block_image` takes an optional plain-text `caption` (exported as `<figure>`/`<figcaption>`) and an optional `metadata` map of host-defined `data-*` attributes. `parseCanonicalListHtml(html, { imageMetadataAttributes: ["data-source"] })` keeps only the attributes you allowlist (none by default). `getSectionContext(document, selection)` returns the nearest top-level heading at the caret plus that section's plain text, and `listDocumentImages(document)` lists every image in document order.
|
|
33
34
|
- **Collab contract** (`collab/`) — the operation-transform machinery (`mapOperation`) a real-time transport would rebase concurrent edits through. A contract for hosts building multi-writer collaboration against; no transport implementation ships here.
|
|
34
35
|
|
|
35
36
|
## Subpath exports
|
|
@@ -4,17 +4,6 @@ export declare const atomToHtml: (node: SmartElementNode, options?: {
|
|
|
4
4
|
renderFormulaHtml?: boolean;
|
|
5
5
|
}) => string;
|
|
6
6
|
export declare const atomFromHtmlElement: (rawElement: Element) => SmartElementNode | null;
|
|
7
|
-
/**
|
|
8
|
-
* page_break: Markdown has no pagination concept at all (declared
|
|
9
|
-
* `unsupported` in formats/fidelity.ts) - emitting nothing would silently
|
|
10
|
-
* delete the marker with no trace, the same class of bug this project has
|
|
11
|
-
* hit before for images/formulas (see formats/fidelity.ts's own
|
|
12
|
-
* images-media/formulas notes on that history). An HTML comment is inert
|
|
13
|
-
* in every real Markdown renderer (so it never appears as visible garbage
|
|
14
|
-
* text) but keeps the marker's *position* recorded in the exported file -
|
|
15
|
-
* genuinely honest `unsupported`, not a round-trippable format: nothing
|
|
16
|
-
* parses this comment back into a page_break node on import.
|
|
17
|
-
*/
|
|
18
7
|
export declare const atomToMarkdown: (node: SmartElementNode) => string;
|
|
19
8
|
export interface AtomDocxRun {
|
|
20
9
|
readonly kind: "image" | "text" | "pageBreak";
|
|
@@ -8,6 +8,7 @@ import "katex/contrib/mhchem";
|
|
|
8
8
|
import { createNodeId } from "../identity.js";
|
|
9
9
|
import { sanitizeLinkHref, sanitizeLinkTarget } from "../security/urlPolicy.js";
|
|
10
10
|
import { sanitizeAtomSource } from "./security.js";
|
|
11
|
+
import { isValidImageMetadataName, normalizeImageCaption } from "./imageExtras.js";
|
|
11
12
|
const escape = (value) => String(value ?? "").replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """);
|
|
12
13
|
const attr = (name, value) => value === undefined ? "" : ` ${name}="${escape(value)}"`;
|
|
13
14
|
const dimensions = (node) => `${attr("width", node.attrs?.width)}${attr("height", node.attrs?.height)}`;
|
|
@@ -23,6 +24,15 @@ const renderFormulaToHtml = (source) => {
|
|
|
23
24
|
return escape(source);
|
|
24
25
|
}
|
|
25
26
|
};
|
|
27
|
+
/** Host metadata (atom/imageExtras.ts) as plain escaped `data-*` attributes, in stored order. Only `block_image` carries it; the stored map is already shape-validated by the schema. */
|
|
28
|
+
const imageMetadataAttributes = (node) => {
|
|
29
|
+
const metadata = node.type === "block_image" ? node.attrs?.metadata : undefined;
|
|
30
|
+
if (!metadata || typeof metadata !== "object" || Array.isArray(metadata))
|
|
31
|
+
return "";
|
|
32
|
+
return Object.entries(metadata)
|
|
33
|
+
.filter(([name, value]) => isValidImageMetadataName(name) && typeof value === "string")
|
|
34
|
+
.map(([name, value]) => attr(name, value)).join("");
|
|
35
|
+
};
|
|
26
36
|
export const atomToHtml = (node, options) => {
|
|
27
37
|
if (node.type === "image" || node.type === "block_image") {
|
|
28
38
|
const src = sanitizeAtomSource(String(node.attrs?.src || ""), { kind: "image", allowBlobPreview: node.attrs?.status === "pending" }) || "";
|
|
@@ -32,17 +42,24 @@ export const atomToHtml = (node, options) => {
|
|
|
32
42
|
// running the live surface renderer's own style logic against it
|
|
33
43
|
// wouldn't see visual rounding either way, same as align today; that's
|
|
34
44
|
// a deliberate, pre-existing scope boundary this doesn't change.
|
|
35
|
-
const img = `<img data-smart-id="${escape(node.id)}" data-smart-type="${node.type}" src="${escape(src)}" alt="${escape(node.attrs?.alt)}"${dimensions(node)}${attr("data-smart-status", node.attrs?.status || "ready")}${node.attrs?.decorative === true ? ' data-smart-decorative="true"' : ""}${attr("data-smart-align", node.attrs?.align)}${attr("data-smart-radius", node.attrs?.borderRadius)}${attr("data-smart-license-description", node.attrs?.licenseDescription)}${attr("data-smart-license-source-url", node.attrs?.licenseSourceUrl)}${attr("data-smart-license-type", node.attrs?.licenseType)}${attr("data-smart-license-version", node.attrs?.licenseVersion)}${attr("data-smart-license-attribution", node.attrs?.licenseAttribution)}${attr("data-smart-href", node.attrs?.href)}${attr("data-smart-target", node.attrs?.target)}>`;
|
|
45
|
+
const img = `<img data-smart-id="${escape(node.id)}" data-smart-type="${node.type}" src="${escape(src)}" alt="${escape(node.attrs?.alt)}"${dimensions(node)}${attr("data-smart-status", node.attrs?.status || "ready")}${node.attrs?.decorative === true ? ' data-smart-decorative="true"' : ""}${attr("data-smart-align", node.attrs?.align)}${attr("data-smart-radius", node.attrs?.borderRadius)}${attr("data-smart-license-description", node.attrs?.licenseDescription)}${attr("data-smart-license-source-url", node.attrs?.licenseSourceUrl)}${attr("data-smart-license-type", node.attrs?.licenseType)}${attr("data-smart-license-version", node.attrs?.licenseVersion)}${attr("data-smart-license-attribution", node.attrs?.licenseAttribution)}${attr("data-smart-href", node.attrs?.href)}${attr("data-smart-target", node.attrs?.target)}${imageMetadataAttributes(node)}>`;
|
|
36
46
|
// A real <a> wrapper, unlike the live surface renderer's plain data
|
|
37
47
|
// attributes on the bare <img> - see surface/renderer.ts's own
|
|
38
48
|
// reasoning (stable-element-per-id constraint that doesn't apply to
|
|
39
49
|
// this string-based export) and modelDom.ts's identical choice for the
|
|
40
50
|
// one-shot clipboard/print DOM builder.
|
|
41
51
|
const linkHref = sanitizeLinkHref(typeof node.attrs?.href === "string" ? node.attrs.href : undefined);
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
52
|
+
const target = linkHref ? sanitizeLinkTarget(typeof node.attrs?.target === "string" ? node.attrs.target : undefined) : undefined;
|
|
53
|
+
const linked = linkHref ? `<a href="${escape(linkHref)}"${target ? ` target="${escape(target)}"` : ""}>${img}</a>` : img;
|
|
54
|
+
// A captioned block image becomes a real <figure>/<figcaption> - the
|
|
55
|
+
// accessible caption for any static HTML consumer. Every attribute stays
|
|
56
|
+
// on the <img> itself, so a pre-1.2 parser (which unwraps <figure> as a
|
|
57
|
+
// transparent container) still reads the image correctly. The link, if
|
|
58
|
+
// any, wraps only the <img>. data-smart-figure is a debugging marker;
|
|
59
|
+
// list/formats.ts's parser never depends on it, since third-party
|
|
60
|
+
// figures don't carry it. Uncaptioned output is unchanged.
|
|
61
|
+
const caption = node.type === "block_image" && typeof node.attrs?.caption === "string" ? node.attrs.caption : "";
|
|
62
|
+
return caption ? `<figure data-smart-figure="true">${linked}<figcaption>${escape(caption)}</figcaption></figure>` : linked;
|
|
46
63
|
}
|
|
47
64
|
if (node.type === "formula" || node.type === "block_formula") {
|
|
48
65
|
const tag = node.type === "formula" ? "span" : "div";
|
|
@@ -82,7 +99,13 @@ export const atomFromHtmlElement = (rawElement) => {
|
|
|
82
99
|
// (which still carries every attribute, including data-smart-href,
|
|
83
100
|
// directly) before parsing, matching what a single element's own
|
|
84
101
|
// attributes already fully describe.
|
|
85
|
-
|
|
102
|
+
// A captioned block image exports as <figure>(<a>)<img><figcaption>; the
|
|
103
|
+
// image element still carries every attribute, so unwrap to it and read
|
|
104
|
+
// the caption from the figcaption.
|
|
105
|
+
const figureImage = rawElement.tagName === "FIGURE" ? rawElement.querySelector(":scope > img, :scope > a > img") : null;
|
|
106
|
+
const figureCaption = figureImage ? normalizeImageCaption(rawElement.querySelector(":scope > figcaption")?.textContent) : undefined;
|
|
107
|
+
const unwrapped = figureImage || rawElement;
|
|
108
|
+
const element = unwrapped.tagName === "A" && unwrapped.children.length === 1 ? unwrapped.children[0] : unwrapped;
|
|
86
109
|
const declared = element.getAttribute("data-smart-type");
|
|
87
110
|
const type = declared || (element.tagName === "IMG" ? "image" : element.tagName.toLowerCase());
|
|
88
111
|
const id = element.getAttribute("data-smart-id") || createNodeId();
|
|
@@ -105,6 +128,7 @@ export const atomFromHtmlElement = (rawElement) => {
|
|
|
105
128
|
...(element.getAttribute("data-smart-license-type") ? { licenseType: element.getAttribute("data-smart-license-type") } : {}),
|
|
106
129
|
...(element.getAttribute("data-smart-license-version") ? { licenseVersion: element.getAttribute("data-smart-license-version") } : {}),
|
|
107
130
|
...(element.getAttribute("data-smart-license-attribution") ? { licenseAttribution: element.getAttribute("data-smart-license-attribution") } : {}),
|
|
131
|
+
...(type === "block_image" && figureCaption ? { caption: figureCaption } : {}),
|
|
108
132
|
} };
|
|
109
133
|
}
|
|
110
134
|
if (type === "formula" || type === "block_formula") {
|
|
@@ -133,9 +157,15 @@ export const atomFromHtmlElement = (rawElement) => {
|
|
|
133
157
|
* genuinely honest `unsupported`, not a round-trippable format: nothing
|
|
134
158
|
* parses this comment back into a page_break node on import.
|
|
135
159
|
*/
|
|
160
|
+
const escapeMarkdownText = (value) => value.replace(/[\\`*_[\]<>]/g, (character) => `\\${character}`);
|
|
136
161
|
export const atomToMarkdown = (node) => {
|
|
137
|
-
if (node.type === "image" || node.type === "block_image")
|
|
138
|
-
|
|
162
|
+
if (node.type === "image" || node.type === "block_image") {
|
|
163
|
+
const image = `})`;
|
|
164
|
+
// Markdown has no caption syntax; an italic line straight after the
|
|
165
|
+
// image is the common convention. Not parsed back into a caption.
|
|
166
|
+
const caption = node.type === "block_image" && typeof node.attrs?.caption === "string" ? node.attrs.caption : "";
|
|
167
|
+
return caption ? `${image}\n*${escapeMarkdownText(caption)}*` : image;
|
|
168
|
+
}
|
|
139
169
|
if (node.type === "formula" || node.type === "block_formula")
|
|
140
170
|
return node.type === "formula" ? `$${String(node.attrs?.source || "")}$` : `$$\n${String(node.attrs?.source || "")}\n$$`;
|
|
141
171
|
// Media is unsupported in Markdown. Preserve a readable link instead of dropping content.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Block-image captions and host-defined metadata (1.2.0). Shared by the
|
|
3
|
+
* schema (shape validation), the HTML parser (normalisation on import), the
|
|
4
|
+
* exporters and the React insert/edit paths, so every entry point agrees on
|
|
5
|
+
* what a valid caption or metadata entry is.
|
|
6
|
+
*
|
|
7
|
+
* Both are `block_image`-only: a caption under an image that sits inside a
|
|
8
|
+
* line of text has no layout meaning, and keeping the inline `image` node
|
|
9
|
+
* unchanged keeps the risk of this feature contained.
|
|
10
|
+
*/
|
|
11
|
+
export declare const IMAGE_CAPTION_MAX_LENGTH = 500;
|
|
12
|
+
export declare const IMAGE_METADATA_MAX_ENTRIES = 10;
|
|
13
|
+
export declare const IMAGE_METADATA_MAX_NAME_LENGTH = 40;
|
|
14
|
+
export declare const IMAGE_METADATA_MAX_VALUE_LENGTH = 500;
|
|
15
|
+
/**
|
|
16
|
+
* Plain text, control characters stripped, whitespace collapsed to single
|
|
17
|
+
* spaces, trimmed, capped at IMAGE_CAPTION_MAX_LENGTH. Returns `undefined`
|
|
18
|
+
* for anything that normalises to empty, so callers delete the attribute
|
|
19
|
+
* instead of storing `""` (an absent caption must export byte-identically
|
|
20
|
+
* to a pre-1.2 document).
|
|
21
|
+
*/
|
|
22
|
+
export declare const normalizeImageCaption: (value: unknown) => string | undefined;
|
|
23
|
+
/** Schema check: only an already-normalised, non-empty caption is valid. */
|
|
24
|
+
export declare const isValidImageCaption: (value: unknown) => boolean;
|
|
25
|
+
/** `data-*` in lowercase kebab-case, at most 40 characters, never in the reserved `data-smart-` namespace. */
|
|
26
|
+
export declare const isValidImageMetadataName: (name: unknown) => name is string;
|
|
27
|
+
/**
|
|
28
|
+
* Schema check for `block_image.attrs.metadata`. Shape only - the schema is
|
|
29
|
+
* static, so it cannot know a host's allowlist. Allowlist enforcement lives
|
|
30
|
+
* at the entry points (parser option, insertImage).
|
|
31
|
+
*/
|
|
32
|
+
export declare const isValidImageMetadata: (value: unknown) => boolean;
|
|
33
|
+
export interface ImageMetadataAllowlist {
|
|
34
|
+
readonly names: readonly string[];
|
|
35
|
+
readonly rejected: readonly string[];
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Validates a host-supplied `imageMetadataAttributes` list: invalid names
|
|
39
|
+
* are dropped (reported in `rejected` so the caller can warn), duplicates
|
|
40
|
+
* collapse, and at most IMAGE_METADATA_MAX_ENTRIES names are kept.
|
|
41
|
+
*/
|
|
42
|
+
export declare const normalizeImageMetadataAllowlist: (names: readonly unknown[] | undefined) => ImageMetadataAllowlist;
|
|
43
|
+
/**
|
|
44
|
+
* Keeps only allowlisted names with string values, truncating each value to
|
|
45
|
+
* IMAGE_METADATA_MAX_VALUE_LENGTH. Returns `undefined` when nothing
|
|
46
|
+
* survives, so the attribute is omitted rather than stored as `{}`.
|
|
47
|
+
*/
|
|
48
|
+
export declare const filterImageMetadata: (read: (name: string) => unknown, allowlist: readonly string[]) => Record<string, string> | undefined;
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Block-image captions and host-defined metadata (1.2.0). Shared by the
|
|
3
|
+
* schema (shape validation), the HTML parser (normalisation on import), the
|
|
4
|
+
* exporters and the React insert/edit paths, so every entry point agrees on
|
|
5
|
+
* what a valid caption or metadata entry is.
|
|
6
|
+
*
|
|
7
|
+
* Both are `block_image`-only: a caption under an image that sits inside a
|
|
8
|
+
* line of text has no layout meaning, and keeping the inline `image` node
|
|
9
|
+
* unchanged keeps the risk of this feature contained.
|
|
10
|
+
*/
|
|
11
|
+
export const IMAGE_CAPTION_MAX_LENGTH = 500;
|
|
12
|
+
export const IMAGE_METADATA_MAX_ENTRIES = 10;
|
|
13
|
+
export const IMAGE_METADATA_MAX_NAME_LENGTH = 40;
|
|
14
|
+
export const IMAGE_METADATA_MAX_VALUE_LENGTH = 500;
|
|
15
|
+
// C0/C1 control characters other than the whitespace ones collapsed below.
|
|
16
|
+
// eslint-disable-next-line no-control-regex
|
|
17
|
+
const CONTROL_CHARACTERS = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F-\u009F]/g;
|
|
18
|
+
/**
|
|
19
|
+
* Plain text, control characters stripped, whitespace collapsed to single
|
|
20
|
+
* spaces, trimmed, capped at IMAGE_CAPTION_MAX_LENGTH. Returns `undefined`
|
|
21
|
+
* for anything that normalises to empty, so callers delete the attribute
|
|
22
|
+
* instead of storing `""` (an absent caption must export byte-identically
|
|
23
|
+
* to a pre-1.2 document).
|
|
24
|
+
*/
|
|
25
|
+
export const normalizeImageCaption = (value) => {
|
|
26
|
+
if (typeof value !== "string")
|
|
27
|
+
return undefined;
|
|
28
|
+
const normalized = value.replace(CONTROL_CHARACTERS, "").replace(/\s+/g, " ").trim().slice(0, IMAGE_CAPTION_MAX_LENGTH).trim();
|
|
29
|
+
return normalized || undefined;
|
|
30
|
+
};
|
|
31
|
+
/** Schema check: only an already-normalised, non-empty caption is valid. */
|
|
32
|
+
export const isValidImageCaption = (value) => typeof value === "string" && value.length > 0 && normalizeImageCaption(value) === value;
|
|
33
|
+
const METADATA_NAME = /^data-[a-z0-9]+(?:-[a-z0-9]+)*$/;
|
|
34
|
+
/** `data-*` in lowercase kebab-case, at most 40 characters, never in the reserved `data-smart-` namespace. */
|
|
35
|
+
export const isValidImageMetadataName = (name) => typeof name === "string" && name.length <= IMAGE_METADATA_MAX_NAME_LENGTH
|
|
36
|
+
&& METADATA_NAME.test(name) && !name.startsWith("data-smart-");
|
|
37
|
+
/**
|
|
38
|
+
* Schema check for `block_image.attrs.metadata`. Shape only - the schema is
|
|
39
|
+
* static, so it cannot know a host's allowlist. Allowlist enforcement lives
|
|
40
|
+
* at the entry points (parser option, insertImage).
|
|
41
|
+
*/
|
|
42
|
+
export const isValidImageMetadata = (value) => {
|
|
43
|
+
if (!value || typeof value !== "object" || Array.isArray(value))
|
|
44
|
+
return false;
|
|
45
|
+
const entries = Object.entries(value);
|
|
46
|
+
return entries.length > 0 && entries.length <= IMAGE_METADATA_MAX_ENTRIES
|
|
47
|
+
&& entries.every(([name, entry]) => isValidImageMetadataName(name)
|
|
48
|
+
&& typeof entry === "string" && entry.length <= IMAGE_METADATA_MAX_VALUE_LENGTH);
|
|
49
|
+
};
|
|
50
|
+
/**
|
|
51
|
+
* Validates a host-supplied `imageMetadataAttributes` list: invalid names
|
|
52
|
+
* are dropped (reported in `rejected` so the caller can warn), duplicates
|
|
53
|
+
* collapse, and at most IMAGE_METADATA_MAX_ENTRIES names are kept.
|
|
54
|
+
*/
|
|
55
|
+
export const normalizeImageMetadataAllowlist = (names) => {
|
|
56
|
+
const accepted = [];
|
|
57
|
+
const rejected = [];
|
|
58
|
+
(names || []).forEach((name) => {
|
|
59
|
+
if (!isValidImageMetadataName(name)) {
|
|
60
|
+
rejected.push(String(name));
|
|
61
|
+
return;
|
|
62
|
+
}
|
|
63
|
+
if (accepted.includes(name))
|
|
64
|
+
return;
|
|
65
|
+
if (accepted.length >= IMAGE_METADATA_MAX_ENTRIES) {
|
|
66
|
+
rejected.push(name);
|
|
67
|
+
return;
|
|
68
|
+
}
|
|
69
|
+
accepted.push(name);
|
|
70
|
+
});
|
|
71
|
+
return { names: accepted, rejected };
|
|
72
|
+
};
|
|
73
|
+
/**
|
|
74
|
+
* Keeps only allowlisted names with string values, truncating each value to
|
|
75
|
+
* IMAGE_METADATA_MAX_VALUE_LENGTH. Returns `undefined` when nothing
|
|
76
|
+
* survives, so the attribute is omitted rather than stored as `{}`.
|
|
77
|
+
*/
|
|
78
|
+
export const filterImageMetadata = (read, allowlist) => {
|
|
79
|
+
const output = {};
|
|
80
|
+
allowlist.forEach((name) => {
|
|
81
|
+
if (!isValidImageMetadataName(name))
|
|
82
|
+
return;
|
|
83
|
+
const value = read(name);
|
|
84
|
+
if (typeof value === "string")
|
|
85
|
+
output[name] = value.slice(0, IMAGE_METADATA_MAX_VALUE_LENGTH);
|
|
86
|
+
});
|
|
87
|
+
return Object.keys(output).length ? output : undefined;
|
|
88
|
+
};
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { normalizeLinkInput } from "../security/urlPolicy.js";
|
|
2
|
+
import { isValidImageCaption, isValidImageMetadata } from "./imageExtras.js";
|
|
2
3
|
const optionalString = { validate: (value) => typeof value === "string" };
|
|
3
4
|
const requiredString = { required: true, validate: (value) => typeof value === "string" };
|
|
4
5
|
const dimension = { validate: (value) => Number.isFinite(value) && Number(value) > 0 && Number(value) <= 100000 };
|
|
@@ -36,6 +37,24 @@ const imageAttrs = {
|
|
|
36
37
|
licenseDescription: optionalString, licenseSourceUrl: optionalString,
|
|
37
38
|
licenseType: optionalString, licenseVersion: optionalString, licenseAttribution: optionalString,
|
|
38
39
|
};
|
|
40
|
+
/**
|
|
41
|
+
* block_image only (atom/imageExtras.ts explains why the inline `image`
|
|
42
|
+
* stays unchanged). `caption` is plain text; `metadata` is a host-defined
|
|
43
|
+
* map of allowlisted `data-*` attributes, shape-checked here and
|
|
44
|
+
* allowlist-checked at every entry point. Both are absent - never "" or {} -
|
|
45
|
+
* when unset, so a pre-1.2 document exports byte-identically.
|
|
46
|
+
*/
|
|
47
|
+
const blockImageAttrs = {
|
|
48
|
+
...imageAttrs,
|
|
49
|
+
caption: { validate: isValidImageCaption },
|
|
50
|
+
metadata: { validate: isValidImageMetadata },
|
|
51
|
+
};
|
|
52
|
+
/** The schema otherwise ignores unknown attributes; these two are explicitly rejected on the inline image so repair() strips them instead of carrying block-only data on an inline node. */
|
|
53
|
+
const inlineImageAttrs = {
|
|
54
|
+
...imageAttrs,
|
|
55
|
+
caption: { validate: () => false },
|
|
56
|
+
metadata: { validate: () => false },
|
|
57
|
+
};
|
|
39
58
|
const formulaAttrs = {
|
|
40
59
|
source: requiredString,
|
|
41
60
|
notation: { required: true, default: "latex", validate: (value) => value === "latex" || value === "mathml" },
|
|
@@ -44,8 +63,8 @@ const formulaAttrs = {
|
|
|
44
63
|
const mediaAttrs = { src: requiredString, poster: optionalString, width: dimension, height: dimension, status, uploadId: optionalString, error: optionalString };
|
|
45
64
|
/** Inline and block variants are distinct because schema groups are static. */
|
|
46
65
|
export const atomNodeSpecs = [
|
|
47
|
-
{ type: "image", group: "inline", atomic: true, selectable: true, marks: "", attributes:
|
|
48
|
-
{ type: "block_image", group: "block", atomic: true, selectable: true, marks: "", attributes:
|
|
66
|
+
{ type: "image", group: "inline", atomic: true, selectable: true, marks: "", attributes: inlineImageAttrs },
|
|
67
|
+
{ type: "block_image", group: "block", atomic: true, selectable: true, marks: "", attributes: blockImageAttrs },
|
|
49
68
|
{ type: "formula", group: "inline", atomic: true, selectable: true, marks: "", attributes: formulaAttrs },
|
|
50
69
|
{ type: "block_formula", group: "block", atomic: true, selectable: true, marks: "", attributes: formulaAttrs },
|
|
51
70
|
{ type: "video", group: "block", atomic: true, selectable: true, marks: "", attributes: mediaAttrs },
|
|
@@ -18,7 +18,16 @@ const blockAttrs = { align: alignmentAttr, indentLevel: indentLevelAttr, lineHei
|
|
|
18
18
|
export const blockNodeSpecs = [
|
|
19
19
|
{ type: "paragraph", group: "block", content: "inline*", attributes: blockAttrs },
|
|
20
20
|
{ type: "heading", group: "block", content: "inline*", attributes: { ...blockAttrs, level: { required: true, default: 1, validate: (v) => Number.isInteger(v) && Number(v) >= 1 && Number(v) <= 6 } } },
|
|
21
|
-
{
|
|
21
|
+
{
|
|
22
|
+
type: "blockquote", group: "block", content: "block+", defining: true,
|
|
23
|
+
// backgroundColor/textColor are plain CSS colour values, matching
|
|
24
|
+
// table_cell's own background/textColor attrs. borderLeft stores one
|
|
25
|
+
// composed CSS shorthand ("4px solid #0284c7") rather than separate
|
|
26
|
+
// width/style/colour attrs - blockquote only ever shows a single
|
|
27
|
+
// visible border side, so there's no per-side independence to
|
|
28
|
+
// preserve the way table_cell's 4-sided borders need.
|
|
29
|
+
attributes: { ...blockAttrs, backgroundColor: stringAttr, textColor: stringAttr, borderLeft: stringAttr },
|
|
30
|
+
},
|
|
22
31
|
{ type: "code_block", group: "block", content: "text*", marks: "", attributes: { ...blockAttrs, language: stringAttr }, defining: true },
|
|
23
32
|
];
|
|
24
33
|
export const blockToolDeclarations = [
|
|
@@ -52,7 +52,7 @@ export const parseClipboardPayload = (payload, options) => {
|
|
|
52
52
|
const normalized = normalizer.normalize(sanitized);
|
|
53
53
|
const parsed = detection.source === "native" && payload.native
|
|
54
54
|
? parseNativeClipboardDocument(payload.native)
|
|
55
|
-
: parseCanonicalListHtml(normalized.html);
|
|
55
|
+
: parseCanonicalListHtml(normalized.html, { imageMetadataAttributes: options.imageMetadataAttributes });
|
|
56
56
|
const repaired = repair({ ...parsed, id: parsed.id || createNodeId() }, options.schema ?? foundationSchema);
|
|
57
57
|
return { source: detection.source, document: repaired.doc, repairs: [...normalized.repairs, ...repaired.repairs] };
|
|
58
58
|
};
|
|
@@ -6,8 +6,11 @@ const nodeText = (node) => {
|
|
|
6
6
|
return node.text;
|
|
7
7
|
if (node.type === "hard_break")
|
|
8
8
|
return "\n";
|
|
9
|
-
if (node.type === "image" || node.type === "block_image")
|
|
10
|
-
|
|
9
|
+
if (node.type === "image" || node.type === "block_image") {
|
|
10
|
+
const alt = String(node.attrs?.alt || "");
|
|
11
|
+
const caption = node.type === "block_image" && typeof node.attrs?.caption === "string" ? node.attrs.caption : "";
|
|
12
|
+
return caption ? (alt ? `${alt} - ${caption}` : caption) : alt;
|
|
13
|
+
}
|
|
11
14
|
if (node.type === "formula" || node.type === "block_formula")
|
|
12
15
|
return String(node.attrs?.source || "");
|
|
13
16
|
const separator = node.type === "table_cell" ? "\t" : node.type === "paragraph" || node.type === "heading"
|
|
@@ -45,6 +45,8 @@ export interface ClipboardRepresentations {
|
|
|
45
45
|
}
|
|
46
46
|
export interface ClipboardPipelineOptions {
|
|
47
47
|
readonly ownerDocument: Document;
|
|
48
|
+
/** Passed to parseCanonicalListHtml for HTML payloads. Native SmartRTE fragments keep their stored metadata as-is. */
|
|
49
|
+
readonly imageMetadataAttributes?: readonly string[];
|
|
48
50
|
readonly normalizers?: readonly SourceNormalizer[];
|
|
49
51
|
/** Test/audit switch proving detection is never required for correctness. */
|
|
50
52
|
readonly normalizerMode?: "detected" | "generic";
|
|
@@ -243,8 +243,18 @@ const blockXml = (block, context, listLevel = 0) => {
|
|
|
243
243
|
}
|
|
244
244
|
if (block.type === "table")
|
|
245
245
|
return tableXml(block, context);
|
|
246
|
-
if (block.type === "block_image")
|
|
247
|
-
|
|
246
|
+
if (block.type === "block_image") {
|
|
247
|
+
const image = paragraphXml([{ ...block, type: "image" }], block, context);
|
|
248
|
+
const caption = typeof block.attrs?.caption === "string" ? block.attrs.caption : "";
|
|
249
|
+
if (!caption)
|
|
250
|
+
return image;
|
|
251
|
+
// Word's built-in "Caption" paragraph style, aligned like the image.
|
|
252
|
+
// This package writes no styles.xml (headings reference Heading1-6 the
|
|
253
|
+
// same undefined way), so the run also carries the caption look
|
|
254
|
+
// directly - italic, 9pt - and reads as a caption even where the style
|
|
255
|
+
// id resolves to Normal.
|
|
256
|
+
return `${image}<w:p><w:pPr>${paragraphProperties(block, '<w:pStyle w:val="Caption"/>')}</w:pPr><w:r><w:rPr><w:i/><w:sz w:val="18"/></w:rPr><w:t xml:space="preserve">${xmlEscape(caption)}</w:t></w:r></w:p>`;
|
|
257
|
+
}
|
|
248
258
|
if (block.type === "block_formula")
|
|
249
259
|
return paragraphXml([{ ...block, type: "formula" }], block, context);
|
|
250
260
|
// Previously fell through to the generic fallback below (an empty
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import type { FormatFidelityLevel, FormatId } from "./codec.js";
|
|
2
2
|
export type FidelityFormat = FormatId;
|
|
3
3
|
export type FidelityLevel = FormatFidelityLevel;
|
|
4
|
-
export type FidelityFeature = "inline-marks" | "colors-fonts-sizes" | "headings-alignment" | "blockquote-code" | "lists" | "checklists" | "tables" | "links" | "images-media" | "formulas" | "special-characters" | "page-break" | "line-height";
|
|
4
|
+
export type FidelityFeature = "inline-marks" | "colors-fonts-sizes" | "headings-alignment" | "blockquote-code" | "lists" | "checklists" | "tables" | "links" | "images-media" | "formulas" | "special-characters" | "page-break" | "line-height" | "image-captions";
|
|
5
5
|
export interface FormatFidelityCapability {
|
|
6
6
|
level: FidelityLevel;
|
|
7
7
|
note: string;
|
|
@@ -95,7 +95,7 @@ export const builtInFormatFidelity = [
|
|
|
95
95
|
{
|
|
96
96
|
feature: "images-media",
|
|
97
97
|
formats: {
|
|
98
|
-
html: capability("semantic", "Images, audio, and video round-trip
|
|
98
|
+
html: capability("semantic", "Images, audio, and video round-trip. Host-defined block-image metadata (`data-*` attributes) is exported as stored, but on import is kept only for names in the `imageMetadataAttributes` allowlist passed to the parser; anything else is dropped."),
|
|
99
99
|
markdown: capability("lossy", "Inline and block images round-trip as  (SS2.3: fixed a real silent-data-loss bug - markdownInlineText/markdownBlock's fallback for any atom node was an empty string, deleting images with no trace; images are now routed through atomToMarkdown/parsed back via a real image AST case). Audio and video degrade to a readable [video: url](url) link on export - the link and URL survive, but re-import produces a generic link, not a video/audio atom."),
|
|
100
100
|
docx: capability("semantic", "Data-URL PNG/JPEG/GIF images embed as native Word media relationships; remote or unsupported sources fall back to a portable text marker recovered on import (SS2.1, re-verified). No canonical DOCX projection exists yet for video/audio (block_image only)."),
|
|
101
101
|
pdf: capability("lossy", "Export is visual; semantic media import is unsupported."),
|
|
@@ -137,5 +137,14 @@ export const builtInFormatFidelity = [
|
|
|
137
137
|
pdf: capability("full", "This package's actual 'Save as PDF' is a real browser print of the same HTML export, so it inherits the html row's real, correctly-spaced layout exactly - verified against the real print document's computed styles, not assumed."),
|
|
138
138
|
},
|
|
139
139
|
},
|
|
140
|
+
{
|
|
141
|
+
feature: "image-captions",
|
|
142
|
+
formats: {
|
|
143
|
+
html: capability("full", "A captioned block image exports as <figure data-smart-figure=\"true\"><img ...><figcaption>text</figcaption></figure> with every image attribute still on the <img>; any <figure> holding one image plus one <figcaption> (third-party or our own) imports back as one captioned block_image. Uncaptioned images export unchanged. Captions are plain text: marks and links inside an imported figcaption are flattened."),
|
|
144
|
+
markdown: capability("lossy", "Exported as an italic line directly under the image (`` then `*caption*`). Markdown has no caption syntax, so re-import yields an image followed by an italic paragraph, not a caption."),
|
|
145
|
+
docx: capability("lossy", "Exported as a paragraph in Word's built-in Caption style (plus direct italic 9pt formatting, since this exporter writes no styles.xml) directly after the image paragraph. DOCX import does not reconstruct the caption: mammoth places images inline inside paragraphs, so the caption returns as an ordinary paragraph."),
|
|
146
|
+
pdf: capability("full", "'Save as PDF' prints the HTML export, so the <figcaption> prints under the image exactly as in the html row."),
|
|
147
|
+
},
|
|
148
|
+
},
|
|
140
149
|
];
|
|
141
150
|
export const getFormatFidelity = (feature, format) => builtInFormatFidelity.find((entry) => entry.feature === feature).formats[format];
|
package/dist/foundation/index.js
CHANGED
|
@@ -4,7 +4,18 @@ export declare const serializeCanonicalListHtml: (document: SmartDocument, optio
|
|
|
4
4
|
fragment?: boolean;
|
|
5
5
|
renderFormulaHtml?: boolean;
|
|
6
6
|
}) => string;
|
|
7
|
-
export
|
|
7
|
+
export interface ParseCanonicalListHtmlOptions {
|
|
8
|
+
/**
|
|
9
|
+
* Host-defined `data-*` attributes to keep from each block image as
|
|
10
|
+
* `attrs.metadata` (atom/imageExtras.ts). Invalid names are ignored.
|
|
11
|
+
* Omitted or empty (the default): nothing is collected, exactly as
|
|
12
|
+
* before this option existed. A host that persists HTML and reloads it
|
|
13
|
+
* through this function directly must pass the same list it gives the
|
|
14
|
+
* editor, or the metadata is dropped on that path.
|
|
15
|
+
*/
|
|
16
|
+
readonly imageMetadataAttributes?: readonly string[];
|
|
17
|
+
}
|
|
18
|
+
export declare const parseCanonicalListHtml: (html: string, options?: ParseCanonicalListHtmlOptions) => SmartDocument;
|
|
8
19
|
/** Alignment and indent are unsupported in Markdown; block content survives semantically. */
|
|
9
20
|
export declare const serializeCanonicalListMarkdown: (document: SmartDocument) => string;
|
|
10
21
|
export declare const parseCanonicalListMarkdown: (markdown: string) => SmartDocument;
|
|
@@ -7,6 +7,7 @@ import { canonicalMarkAttrs, canonicalMarkOrder } from "../marks/canonical.js";
|
|
|
7
7
|
import { occupancyGridFor } from "../table/grid.js";
|
|
8
8
|
import { atomToHtml, atomToMarkdown } from "../atom/formats.js";
|
|
9
9
|
import { sanitizeAtomSource } from "../atom/security.js";
|
|
10
|
+
import { filterImageMetadata, normalizeImageCaption, normalizeImageMetadataAllowlist } from "../atom/imageExtras.js";
|
|
10
11
|
import { foundationListStyleForPresetDepth, isFoundationSmartListPreset } from "./presets.js";
|
|
11
12
|
const escapeHtml = (value) => String(value).replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">").replace(/"/g, """);
|
|
12
13
|
const attr = (node, name) => node.attrs?.find((candidate) => candidate.name === name)?.value;
|
|
@@ -39,6 +40,14 @@ const imageStyleAndLicenseAttrs = (node) => {
|
|
|
39
40
|
};
|
|
40
41
|
/** See serializeCanonicalListHtml's own doc comment for why this is a module-scoped flag rather than a threaded parameter. */
|
|
41
42
|
let renderFormulaHtmlMode = false;
|
|
43
|
+
/** parseCanonicalListHtml's `imageMetadataAttributes` option - module-scoped for the same reason as renderFormulaHtmlMode, restored in `finally`. Empty (the default) collects nothing, matching pre-1.2 behaviour. */
|
|
44
|
+
let imageMetadataAllowlist = [];
|
|
45
|
+
const parsedImageMetadata = (node) => {
|
|
46
|
+
if (!imageMetadataAllowlist.length)
|
|
47
|
+
return {};
|
|
48
|
+
const metadata = filterImageMetadata((name) => attr(node, name), imageMetadataAllowlist);
|
|
49
|
+
return metadata ? { metadata } : {};
|
|
50
|
+
};
|
|
42
51
|
const serializeInline = (node) => {
|
|
43
52
|
if (!isTextNode(node)) {
|
|
44
53
|
if (node.type === "hard_break")
|
|
@@ -443,7 +452,14 @@ const parseBlock = (node) => {
|
|
|
443
452
|
// third-party block-level image paste, not just this app's own.
|
|
444
453
|
...(attr(node, "data-smart-href") ? { href: attr(node, "data-smart-href") } : {}),
|
|
445
454
|
...(attr(node, "data-smart-target") ? { target: attr(node, "data-smart-target") } : {}),
|
|
455
|
+
// atomToHtml has always written data-smart-align for a block image,
|
|
456
|
+
// but this branch never read it back, so an aligned image lost its
|
|
457
|
+
// alignment on every HTML save -> reload (docs/bugs/
|
|
458
|
+
// block-image-align-lost-on-html-reload.md). Same allowed values as
|
|
459
|
+
// the schema's own align validator.
|
|
460
|
+
...(["left", "center", "right"].includes(attr(node, "data-smart-align") || "") ? { align: attr(node, "data-smart-align") } : {}),
|
|
446
461
|
...imageStyleAndLicenseAttrs(node),
|
|
462
|
+
...parsedImageMetadata(node),
|
|
447
463
|
} };
|
|
448
464
|
}
|
|
449
465
|
if (declaredAtom === "block_formula")
|
|
@@ -477,6 +493,7 @@ const parseBlock = (node) => {
|
|
|
477
493
|
return { type: "block_image", id: generatedId(node, "image"), attrs: {
|
|
478
494
|
src, alt: attr(node, "alt") || "", status: "ready",
|
|
479
495
|
...(width !== null ? { width } : {}), ...(height !== null ? { height } : {}),
|
|
496
|
+
...parsedImageMetadata(node),
|
|
480
497
|
} };
|
|
481
498
|
}
|
|
482
499
|
}
|
|
@@ -712,9 +729,48 @@ const GENERIC_INLINE_TAGS = ["span", "strong", "b", "em", "i", "u", "s", "strike
|
|
|
712
729
|
* block image - so it's tried as a block first and only folded into the
|
|
713
730
|
* inline run if that doesn't apply.
|
|
714
731
|
*/
|
|
715
|
-
const
|
|
732
|
+
const isWhitespaceText = (node) => node.nodeName === "#text" && !(node.value || "").trim();
|
|
733
|
+
const isImageSource = (node) => node.tagName === "img"
|
|
734
|
+
|| node.tagName === "a" && elementChildren(node).length === 1 && elementChildren(node)[0].tagName === "img"
|
|
735
|
+
&& (node.childNodes || []).every((child) => child.tagName || isWhitespaceText(child));
|
|
736
|
+
/**
|
|
737
|
+
* A captioned figure - exactly one image source (`<img>`, `<a><img></a>`, or
|
|
738
|
+
* this app's own `img[data-smart-type="block_image"]`) plus at most one
|
|
739
|
+
* `<figcaption>`, in either order, ignoring whitespace - becomes ONE
|
|
740
|
+
* block_image with `caption`. The image itself goes through parseBlock, so
|
|
741
|
+
* every existing attribute/width/license rule applies unchanged; the
|
|
742
|
+
* caption is the figcaption's plain text (marks and links flattened).
|
|
743
|
+
* Returns null for any other figure shape (several images, a table, no
|
|
744
|
+
* image, stray text) so the caller falls back to the transparent-container
|
|
745
|
+
* path. Deliberately does NOT depend on atomToHtml's data-smart-figure
|
|
746
|
+
* marker - third-party figures never carry it.
|
|
747
|
+
*/
|
|
748
|
+
const parseFigure = (node) => {
|
|
749
|
+
const significant = (node.childNodes || []).filter((child) => !isWhitespaceText(child) && !isEditorUiNode(child));
|
|
750
|
+
const images = significant.filter(isImageSource);
|
|
751
|
+
const captions = significant.filter((child) => child.tagName === "figcaption");
|
|
752
|
+
if (images.length !== 1 || captions.length > 1 || images.length + captions.length !== significant.length)
|
|
753
|
+
return null;
|
|
754
|
+
const image = parseBlock(images[0]);
|
|
755
|
+
if (!image || image.type !== "block_image")
|
|
756
|
+
return null;
|
|
757
|
+
const caption = captions.length ? normalizeImageCaption(rawText(captions[0])) : undefined;
|
|
758
|
+
return caption ? { ...image, attrs: { ...image.attrs, caption } } : image;
|
|
759
|
+
};
|
|
760
|
+
/** A figcaption that has no image to attach to keeps its text as an ordinary paragraph instead of an "[Unsupported: figcaption]" placeholder. */
|
|
761
|
+
const figcaptionParagraph = (node) => {
|
|
762
|
+
const children = (node.childNodes || []).flatMap((child) => textWithMarks(child));
|
|
763
|
+
return children.some((child) => !isTextNode(child) || child.text.trim() !== "")
|
|
764
|
+
? { type: "paragraph", id: generatedId(node, "p"), children }
|
|
765
|
+
: null;
|
|
766
|
+
};
|
|
767
|
+
const parseMixedBlockContent = (nodes, blockTags, insideUnrecognizedFigure = false) => {
|
|
716
768
|
const result = [];
|
|
717
769
|
let inlineRun = [];
|
|
770
|
+
// Index (in `result`) of a block_image that a directly following
|
|
771
|
+
// <figcaption> sibling may still attach to - see the figcaption branch
|
|
772
|
+
// below. Cleared by anything other than whitespace in between.
|
|
773
|
+
let captionTarget = -1;
|
|
718
774
|
const flushInlineRun = () => {
|
|
719
775
|
if (!inlineRun.length)
|
|
720
776
|
return;
|
|
@@ -731,11 +787,16 @@ const parseMixedBlockContent = (nodes, blockTags) => {
|
|
|
731
787
|
if (hasContent)
|
|
732
788
|
result.push({ type: "paragraph", id: createNodeId(), children });
|
|
733
789
|
};
|
|
790
|
+
const pushParsed = (parsed) => {
|
|
791
|
+
if (!parsed)
|
|
792
|
+
return;
|
|
793
|
+
result.push(parsed);
|
|
794
|
+
if (parsed.type === "block_image" && parsed.attrs?.caption === undefined)
|
|
795
|
+
captionTarget = result.length - 1;
|
|
796
|
+
};
|
|
734
797
|
const pushBlock = (node) => {
|
|
735
798
|
flushInlineRun();
|
|
736
|
-
|
|
737
|
-
if (parsed)
|
|
738
|
-
result.push(parsed);
|
|
799
|
+
pushParsed(parseBlock(node));
|
|
739
800
|
};
|
|
740
801
|
nodes.forEach((node) => {
|
|
741
802
|
// Matches textWithMarks's own first check - a UI-only element (e.g. a
|
|
@@ -746,10 +807,45 @@ const parseMixedBlockContent = (nodes, blockTags) => {
|
|
|
746
807
|
if (isEditorUiNode(node))
|
|
747
808
|
return;
|
|
748
809
|
const tag = node.tagName;
|
|
810
|
+
const target = captionTarget;
|
|
811
|
+
if (!isWhitespaceText(node))
|
|
812
|
+
captionTarget = -1;
|
|
749
813
|
if (!tag) {
|
|
750
814
|
inlineRun.push(node);
|
|
751
815
|
return;
|
|
752
816
|
}
|
|
817
|
+
if (tag === "figcaption") {
|
|
818
|
+
// Legacy repair: before 1.2, a <figure> was unwrapped and its
|
|
819
|
+
// <figcaption> survived only as a verbatim "unknown" node, so every
|
|
820
|
+
// save wrote `<img ...><figcaption>...</figcaption>` as loose
|
|
821
|
+
// siblings (docs/bugs/figure-wrapper-lost-on-save.md). A figcaption
|
|
822
|
+
// directly after a still-uncaptioned block image (whitespace aside)
|
|
823
|
+
// is re-attached as that image's caption. Not inside a figure this
|
|
824
|
+
// parser already declined to treat as one captioned image (e.g. two
|
|
825
|
+
// images): there the caption describes the whole group, so it stays
|
|
826
|
+
// a paragraph rather than being pinned to the last image.
|
|
827
|
+
flushInlineRun();
|
|
828
|
+
const caption = normalizeImageCaption(rawText(node));
|
|
829
|
+
const image = target >= 0 && target === result.length - 1 ? result[target] : null;
|
|
830
|
+
if (!insideUnrecognizedFigure && image?.type === "block_image" && caption) {
|
|
831
|
+
result[target] = { ...image, attrs: { ...image.attrs, caption } };
|
|
832
|
+
}
|
|
833
|
+
else {
|
|
834
|
+
pushParsed(figcaptionParagraph(node));
|
|
835
|
+
}
|
|
836
|
+
return;
|
|
837
|
+
}
|
|
838
|
+
if (tag === "figure" && !attr(node, "data-smart-type")) {
|
|
839
|
+
const figure = parseFigure(node);
|
|
840
|
+
if (figure) {
|
|
841
|
+
flushInlineRun();
|
|
842
|
+
result.push(figure);
|
|
843
|
+
return;
|
|
844
|
+
}
|
|
845
|
+
flushInlineRun();
|
|
846
|
+
result.push(...parseMixedBlockContent(node.childNodes || [], blockTags, true));
|
|
847
|
+
return;
|
|
848
|
+
}
|
|
753
849
|
// A bare third-party <div> (no data-smart-type) is genuinely
|
|
754
850
|
// meaningless wrapping and should be unwrapped - but this app's own
|
|
755
851
|
// round-tripped div-tagged atoms (block_formula, page_break; see
|
|
@@ -763,7 +859,7 @@ const parseMixedBlockContent = (nodes, blockTags) => {
|
|
|
763
859
|
// children back.
|
|
764
860
|
if (TRANSPARENT_CONTAINER_TAGS.includes(tag) && !attr(node, "data-smart-type")) {
|
|
765
861
|
flushInlineRun();
|
|
766
|
-
result.push(...parseMixedBlockContent(node.childNodes || [], blockTags));
|
|
862
|
+
result.push(...parseMixedBlockContent(node.childNodes || [], blockTags, insideUnrecognizedFigure));
|
|
767
863
|
return;
|
|
768
864
|
}
|
|
769
865
|
if (blockTags.includes(tag))
|
|
@@ -773,7 +869,7 @@ const parseMixedBlockContent = (nodes, blockTags) => {
|
|
|
773
869
|
const parsed = parseBlock(node);
|
|
774
870
|
if (parsed && parsed.type !== "unknown") {
|
|
775
871
|
flushInlineRun();
|
|
776
|
-
|
|
872
|
+
pushParsed(parsed);
|
|
777
873
|
return;
|
|
778
874
|
}
|
|
779
875
|
}
|
|
@@ -785,13 +881,20 @@ const parseMixedBlockContent = (nodes, blockTags) => {
|
|
|
785
881
|
flushInlineRun();
|
|
786
882
|
return result;
|
|
787
883
|
};
|
|
788
|
-
export const parseCanonicalListHtml = (html) => {
|
|
789
|
-
const
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
884
|
+
export const parseCanonicalListHtml = (html, options = {}) => {
|
|
885
|
+
const previous = imageMetadataAllowlist;
|
|
886
|
+
imageMetadataAllowlist = normalizeImageMetadataAllowlist(options.imageMetadataAttributes).names;
|
|
887
|
+
try {
|
|
888
|
+
const fragment = parseFragment(html);
|
|
889
|
+
const wrapper = elementChildren(fragment).find((node) => attr(node, "data-smart-document") === "true");
|
|
890
|
+
const source = wrapper || fragment;
|
|
891
|
+
const rootBlockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "table", "hr", "img", "video", "audio"];
|
|
892
|
+
const children = parseMixedBlockContent(source.childNodes || [], rootBlockTags);
|
|
893
|
+
return { type: "doc", id: attr(source, "data-smart-id") || createNodeId(), children: children.length ? children : [{ type: "paragraph", id: createNodeId(), children: [] }] };
|
|
894
|
+
}
|
|
895
|
+
finally {
|
|
896
|
+
imageMetadataAllowlist = previous;
|
|
897
|
+
}
|
|
795
898
|
};
|
|
796
899
|
const plainText = (node) => (node.children || []).map((child) => isTextNode(child) ? child.text : child.type === "hard_break" ? "\n" : "").join("");
|
|
797
900
|
const markdownInlineText = (node) => (node.children || []).map((child) => {
|
|
@@ -33,6 +33,18 @@ export declare class FoundationModelDomMapping implements ModelDomMapping {
|
|
|
33
33
|
node: Node;
|
|
34
34
|
offset: number;
|
|
35
35
|
} | null;
|
|
36
|
+
/**
|
|
37
|
+
* Model offsets between block children count model children only; a
|
|
38
|
+
* renderer projection sitting between blocks (a block image's caption,
|
|
39
|
+
* a table's caption/colgroup, a checklist control) must not shift them.
|
|
40
|
+
* Offset k maps to just before the k-th model child. The end offset maps
|
|
41
|
+
* past the last model child and past any image caption trailing it, so a
|
|
42
|
+
* caret "after the image" lands below its caption, not between the two.
|
|
43
|
+
*/
|
|
44
|
+
private blockBoundaryDomOffset;
|
|
45
|
+
private isImageCaptionProjection;
|
|
46
|
+
/** A DOM point inside (or on) a block image's caption projection reads as the boundary right after that image. */
|
|
47
|
+
private captionBoundaryPos;
|
|
36
48
|
domToPos(node: Node, offset: number): SmartPos | null;
|
|
37
49
|
isEditorUiNode(node: Node): boolean;
|
|
38
50
|
}
|
|
@@ -248,7 +248,9 @@ export class FoundationModelDomMapping {
|
|
|
248
248
|
return null;
|
|
249
249
|
if (!this.isInlineOwner(model)) {
|
|
250
250
|
const children = this.modelDomChildren(element);
|
|
251
|
-
|
|
251
|
+
if (pos.offset > children.length)
|
|
252
|
+
return null;
|
|
253
|
+
return { node: element, offset: this.blockBoundaryDomOffset(element, children, pos.offset) };
|
|
252
254
|
}
|
|
253
255
|
let remaining = pos.offset;
|
|
254
256
|
const domChildren = this.modelDomChildren(element);
|
|
@@ -284,9 +286,46 @@ export class FoundationModelDomMapping {
|
|
|
284
286
|
offset: lastModelChild ? [...element.childNodes].indexOf(lastModelChild) + 1 : 0,
|
|
285
287
|
};
|
|
286
288
|
}
|
|
289
|
+
/**
|
|
290
|
+
* Model offsets between block children count model children only; a
|
|
291
|
+
* renderer projection sitting between blocks (a block image's caption,
|
|
292
|
+
* a table's caption/colgroup, a checklist control) must not shift them.
|
|
293
|
+
* Offset k maps to just before the k-th model child. The end offset maps
|
|
294
|
+
* past the last model child and past any image caption trailing it, so a
|
|
295
|
+
* caret "after the image" lands below its caption, not between the two.
|
|
296
|
+
*/
|
|
297
|
+
blockBoundaryDomOffset(element, children, offset) {
|
|
298
|
+
const all = [...element.childNodes];
|
|
299
|
+
if (offset < children.length)
|
|
300
|
+
return all.indexOf(children[offset]);
|
|
301
|
+
if (!children.length)
|
|
302
|
+
return 0;
|
|
303
|
+
let index = all.indexOf(children[children.length - 1]) + 1;
|
|
304
|
+
while (index < all.length && this.isImageCaptionProjection(all[index]))
|
|
305
|
+
index += 1;
|
|
306
|
+
return index;
|
|
307
|
+
}
|
|
308
|
+
isImageCaptionProjection(node) {
|
|
309
|
+
return node instanceof Element && node.getAttribute(SMART_PROJECTION_ATTRIBUTE) === "image-caption";
|
|
310
|
+
}
|
|
311
|
+
/** A DOM point inside (or on) a block image's caption projection reads as the boundary right after that image. */
|
|
312
|
+
captionBoundaryPos(node) {
|
|
313
|
+
const element = node.nodeType === node.ELEMENT_NODE ? node : node.parentElement;
|
|
314
|
+
const caption = element?.closest(`[${SMART_PROJECTION_ATTRIBUTE}="image-caption"]`);
|
|
315
|
+
if (!caption || !this.root?.contains(caption))
|
|
316
|
+
return null;
|
|
317
|
+
const imageId = caption.getAttribute("data-smart-caption-for");
|
|
318
|
+
const imagePath = imageId ? this.pathById.get(imageId) : undefined;
|
|
319
|
+
if (!imagePath?.length)
|
|
320
|
+
return null;
|
|
321
|
+
return { path: imagePath.slice(0, -1), offset: imagePath[imagePath.length - 1] + 1 };
|
|
322
|
+
}
|
|
287
323
|
domToPos(node, offset) {
|
|
288
324
|
if (!this.root || !this.document || this.isEditorUiNode(node))
|
|
289
325
|
return null;
|
|
326
|
+
const captionPos = this.captionBoundaryPos(node);
|
|
327
|
+
if (captionPos)
|
|
328
|
+
return captionPos;
|
|
290
329
|
const owner = this.domToNode(node);
|
|
291
330
|
if (!owner)
|
|
292
331
|
return null;
|
|
@@ -298,14 +337,17 @@ export class FoundationModelDomMapping {
|
|
|
298
337
|
if (!path || !element)
|
|
299
338
|
return null;
|
|
300
339
|
if (!this.isInlineOwner(ownerNode)) {
|
|
340
|
+
const children = this.modelDomChildren(element);
|
|
301
341
|
if (node !== element) {
|
|
302
342
|
const direct = node.nodeType === node.ELEMENT_NODE ? node : node.parentElement;
|
|
303
343
|
const child = direct?.closest(`[${SMART_NODE_ID_ATTRIBUTE}]`);
|
|
304
|
-
const children = [...element.children].filter((candidate) => !this.isEditorUiNode(candidate));
|
|
305
344
|
const index = child ? children.indexOf(child) : -1;
|
|
306
345
|
return index >= 0 ? { path: [...path], offset: index + (offset > 0 ? 1 : 0) } : null;
|
|
307
346
|
}
|
|
308
|
-
|
|
347
|
+
// A raw DOM offset counts projections too; count only model children
|
|
348
|
+
// before it (see blockBoundaryDomOffset for the inverse).
|
|
349
|
+
const all = [...element.childNodes];
|
|
350
|
+
return { path: [...path], offset: children.filter((child) => all.indexOf(child) < offset).length };
|
|
309
351
|
}
|
|
310
352
|
if (node === element) {
|
|
311
353
|
const direct = this.modelDomChildren(element);
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export * from "./sectionContext.js";
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export * from "./sectionContext.js";
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import type { SmartDocument, SmartSelection } from "../types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Read-only document queries for host apps (1.2.0) - "which section is the
|
|
4
|
+
* cursor in, and what does it say?" and "which images does this document
|
|
5
|
+
* already have?". Pure functions of (document, selection); the React
|
|
6
|
+
* runtime's getSectionContext/listImages are thin wrappers over these.
|
|
7
|
+
*/
|
|
8
|
+
export type HeadingLevel = 1 | 2 | 3 | 4 | 5 | 6;
|
|
9
|
+
export interface SectionContext {
|
|
10
|
+
/** Nearest top-level heading at or above the cursor, or null when the cursor precedes every heading. */
|
|
11
|
+
heading: {
|
|
12
|
+
id: string;
|
|
13
|
+
level: HeadingLevel;
|
|
14
|
+
text: string;
|
|
15
|
+
} | null;
|
|
16
|
+
/** Plain text from that heading (exclusive) to the next top-level heading of the same or higher level. */
|
|
17
|
+
text: string;
|
|
18
|
+
/** True when `text` was cut to `maxChars`. */
|
|
19
|
+
truncated: boolean;
|
|
20
|
+
}
|
|
21
|
+
export interface SectionContextOptions {
|
|
22
|
+
/** Default 8000. */
|
|
23
|
+
maxChars?: number;
|
|
24
|
+
}
|
|
25
|
+
export interface DocumentImageInfo {
|
|
26
|
+
nodeId: string;
|
|
27
|
+
src: string;
|
|
28
|
+
alt: string;
|
|
29
|
+
caption?: string;
|
|
30
|
+
metadata?: Record<string, string>;
|
|
31
|
+
}
|
|
32
|
+
export declare const DEFAULT_SECTION_CONTEXT_MAX_CHARS = 8000;
|
|
33
|
+
/**
|
|
34
|
+
* Only top-level blocks start or end a section; a heading nested inside a
|
|
35
|
+
* table, list or blockquote is ordinary text of the section it sits in.
|
|
36
|
+
* With no meaningful cursor (`type: "none"` or an empty path) the start of
|
|
37
|
+
* the document is used.
|
|
38
|
+
*/
|
|
39
|
+
export declare const getSectionContext: (document: SmartDocument, selection: SmartSelection | null | undefined, options?: SectionContextOptions) => SectionContext;
|
|
40
|
+
/** Every inline `image` and `block_image`, in document order. */
|
|
41
|
+
export declare const listDocumentImages: (document: SmartDocument) => DocumentImageInfo[];
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { isTextNode } from "../identity.js";
|
|
2
|
+
export const DEFAULT_SECTION_CONTEXT_MAX_CHARS = 8000;
|
|
3
|
+
const headingLevel = (node) => {
|
|
4
|
+
const level = Number(node.attrs?.level);
|
|
5
|
+
return (Number.isInteger(level) && level >= 1 && level <= 6 ? level : 1);
|
|
6
|
+
};
|
|
7
|
+
const inlineText = (node) => (node.children || []).map((child) => isTextNode(child) ? child.text : child.type === "hard_break" ? "\n" : "").join("");
|
|
8
|
+
/**
|
|
9
|
+
* Text extraction skips atoms, except that a captioned block image becomes
|
|
10
|
+
* `[Figure: caption]` so a host sees the figures a section already has.
|
|
11
|
+
* Tables become tab-separated cells, one row per line.
|
|
12
|
+
*/
|
|
13
|
+
const blockText = (node) => {
|
|
14
|
+
if (isTextNode(node))
|
|
15
|
+
return node.text;
|
|
16
|
+
if (node.type === "paragraph" || node.type === "heading" || node.type === "code_block")
|
|
17
|
+
return inlineText(node);
|
|
18
|
+
if (node.type === "block_image") {
|
|
19
|
+
return typeof node.attrs?.caption === "string" && node.attrs.caption ? `[Figure: ${node.attrs.caption}]` : "";
|
|
20
|
+
}
|
|
21
|
+
if (node.type === "table") {
|
|
22
|
+
return (node.children || []).map((row) => isTextNode(row) ? ""
|
|
23
|
+
: (row.children || []).map((cell) => isTextNode(cell) ? cell.text : childBlocksText(cell, " ")).join("\t")).join("\n");
|
|
24
|
+
}
|
|
25
|
+
if (node.type === "table_cell")
|
|
26
|
+
return childBlocksText(node, " ");
|
|
27
|
+
return childBlocksText(node, "\n");
|
|
28
|
+
};
|
|
29
|
+
const childBlocksText = (node, separator) => (node.children || []).map(blockText).filter((text) => text !== "").join(separator);
|
|
30
|
+
/**
|
|
31
|
+
* Only top-level blocks start or end a section; a heading nested inside a
|
|
32
|
+
* table, list or blockquote is ordinary text of the section it sits in.
|
|
33
|
+
* With no meaningful cursor (`type: "none"` or an empty path) the start of
|
|
34
|
+
* the document is used.
|
|
35
|
+
*/
|
|
36
|
+
export const getSectionContext = (document, selection, options = {}) => {
|
|
37
|
+
const blocks = document.children;
|
|
38
|
+
const requested = Number(options.maxChars);
|
|
39
|
+
const maxChars = Number.isFinite(requested) && requested >= 0 ? Math.floor(requested) : DEFAULT_SECTION_CONTEXT_MAX_CHARS;
|
|
40
|
+
const cursor = selection && selection.type !== "none" && selection.head.path.length
|
|
41
|
+
? Math.min(Math.max(0, selection.head.path[0]), Math.max(0, blocks.length - 1))
|
|
42
|
+
: 0;
|
|
43
|
+
let headingIndex = -1;
|
|
44
|
+
for (let index = cursor; index >= 0; index -= 1) {
|
|
45
|
+
const block = blocks[index];
|
|
46
|
+
if (block && !isTextNode(block) && block.type === "heading") {
|
|
47
|
+
headingIndex = index;
|
|
48
|
+
break;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
const headingNode = headingIndex >= 0 ? blocks[headingIndex] : null;
|
|
52
|
+
const level = headingNode ? headingLevel(headingNode) : 6;
|
|
53
|
+
const parts = [];
|
|
54
|
+
for (let index = headingIndex + 1; index < blocks.length; index += 1) {
|
|
55
|
+
const block = blocks[index];
|
|
56
|
+
// With no heading above the cursor the "section" is the preamble, which
|
|
57
|
+
// any heading ends.
|
|
58
|
+
if (!isTextNode(block) && block.type === "heading" && headingLevel(block) <= level)
|
|
59
|
+
break;
|
|
60
|
+
const text = blockText(block);
|
|
61
|
+
if (text !== "")
|
|
62
|
+
parts.push(text);
|
|
63
|
+
}
|
|
64
|
+
const full = parts.join("\n\n");
|
|
65
|
+
const truncated = full.length > maxChars;
|
|
66
|
+
return {
|
|
67
|
+
heading: headingNode ? { id: headingNode.id, level, text: inlineText(headingNode) } : null,
|
|
68
|
+
text: truncated ? full.slice(0, maxChars) : full,
|
|
69
|
+
truncated,
|
|
70
|
+
};
|
|
71
|
+
};
|
|
72
|
+
/** Every inline `image` and `block_image`, in document order. */
|
|
73
|
+
export const listDocumentImages = (document) => {
|
|
74
|
+
const images = [];
|
|
75
|
+
const visit = (node) => {
|
|
76
|
+
if (isTextNode(node))
|
|
77
|
+
return;
|
|
78
|
+
if (node.type === "image" || node.type === "block_image") {
|
|
79
|
+
const caption = typeof node.attrs?.caption === "string" && node.attrs.caption ? node.attrs.caption : undefined;
|
|
80
|
+
const metadata = node.attrs?.metadata && typeof node.attrs.metadata === "object" && !Array.isArray(node.attrs.metadata)
|
|
81
|
+
? { ...node.attrs.metadata } : undefined;
|
|
82
|
+
images.push({
|
|
83
|
+
nodeId: node.id,
|
|
84
|
+
src: String(node.attrs?.src || ""),
|
|
85
|
+
alt: String(node.attrs?.alt || ""),
|
|
86
|
+
...(caption ? { caption } : {}),
|
|
87
|
+
...(metadata ? { metadata } : {}),
|
|
88
|
+
});
|
|
89
|
+
}
|
|
90
|
+
node.children?.forEach(visit);
|
|
91
|
+
};
|
|
92
|
+
document.children.forEach(visit);
|
|
93
|
+
return images;
|
|
94
|
+
};
|
|
@@ -511,7 +511,13 @@ export class FoundationInputPipeline {
|
|
|
511
511
|
this.renderer.render(this.editor.document, this.editor.selection);
|
|
512
512
|
return;
|
|
513
513
|
}
|
|
514
|
-
|
|
514
|
+
// A block image's caption is a renderer projection after the <img>, not
|
|
515
|
+
// part of it; clicking the caption selects its image, the same as
|
|
516
|
+
// clicking the image itself.
|
|
517
|
+
const caption = event.target instanceof Element ? event.target.closest('[data-smart-projection="image-caption"]') : null;
|
|
518
|
+
const captionOwner = caption?.getAttribute("data-smart-caption-for");
|
|
519
|
+
const target = captionOwner ? this.renderer.mapping.nodeToDom(captionOwner)
|
|
520
|
+
: event.target instanceof Element ? event.target.closest("[data-smart-atomic]") : null;
|
|
515
521
|
const mapped = target ? this.renderer.mapping.domToNode(target) : null;
|
|
516
522
|
if (!mapped || isTextNode(mapped.node) || this.editor.schema.nodes[mapped.node.type]?.selectable !== true)
|
|
517
523
|
return;
|
|
@@ -629,7 +635,7 @@ export class FoundationInputPipeline {
|
|
|
629
635
|
}
|
|
630
636
|
const payload = this.payloadFromTransfer(event.clipboardData);
|
|
631
637
|
try {
|
|
632
|
-
const parsed = parseClipboardPayload(payload, { ownerDocument: this.ownerDocument, schema: this.editor.schema });
|
|
638
|
+
const parsed = parseClipboardPayload(payload, { ownerDocument: this.ownerDocument, schema: this.editor.schema, imageMetadataAttributes: this.options.imageMetadataAttributes });
|
|
633
639
|
this.options.onClipboardDiagnostic?.(reportParsedClipboard(payload, parsed));
|
|
634
640
|
const fragment = remintClipboardFragmentIds(parsed.document, createNodeId);
|
|
635
641
|
const result = insertClipboardFragment(this.editor.document, this.editor.selection, fragment, {
|
|
@@ -695,7 +701,7 @@ export class FoundationInputPipeline {
|
|
|
695
701
|
this.commitClipboard({ ...insertion, operations: [...deletion.operations, ...insertion.operations] }, "drop");
|
|
696
702
|
return;
|
|
697
703
|
}
|
|
698
|
-
const parsed = parseClipboardPayload(this.payloadFromTransfer(event.dataTransfer), { ownerDocument: this.ownerDocument, schema: this.editor.schema });
|
|
704
|
+
const parsed = parseClipboardPayload(this.payloadFromTransfer(event.dataTransfer), { ownerDocument: this.ownerDocument, schema: this.editor.schema, imageMetadataAttributes: this.options.imageMetadataAttributes });
|
|
699
705
|
const fragment = remintClipboardFragmentIds(parsed.document, createNodeId);
|
|
700
706
|
const result = insertClipboardFragment(this.editor.document, targetSelection, fragment, {
|
|
701
707
|
schema: this.editor.schema, positions: this.editor.positions, idFactory: createNodeId,
|
|
@@ -56,10 +56,30 @@ export declare class FoundationSubtreeRenderer implements CanonicalSubtreeRender
|
|
|
56
56
|
* must not leave a stale checkbox projection behind. Reconcile these
|
|
57
57
|
* projections from the current model after every render; attribute writes
|
|
58
58
|
* remain idempotent.
|
|
59
|
+
*
|
|
60
|
+
* The same single document walk also collects every captioned block
|
|
61
|
+
* image for syncImageCaptions, so captions add no second traversal.
|
|
59
62
|
*/
|
|
60
63
|
private syncListProjections;
|
|
64
|
+
/**
|
|
65
|
+
* A block image's caption is a renderer projection - a non-editable
|
|
66
|
+
* sibling element placed immediately after the <img>, since an <img> can't
|
|
67
|
+
* hold children the way <table> holds its caption projection. Projection
|
|
68
|
+
* elements are already excluded from modelChildren/modelDomChildren, so
|
|
69
|
+
* the caption never enters the model or the selection mapping.
|
|
70
|
+
*
|
|
71
|
+
* diffElement moves, replaces and removes model elements without knowing
|
|
72
|
+
* about this sibling, so instead of patching every one of those paths this
|
|
73
|
+
* re-asserts the invariant after each render pass: every captioned image
|
|
74
|
+
* is followed directly by exactly one caption projection with the current
|
|
75
|
+
* text, and no other caption projection exists. Reuses (moves) an
|
|
76
|
+
* existing projection element rather than recreating it.
|
|
77
|
+
*/
|
|
78
|
+
private syncImageCaptions;
|
|
61
79
|
private announceSelectedLevel;
|
|
62
|
-
render(document: SmartDocument, selection: SmartSelection
|
|
80
|
+
render(document: SmartDocument, selection: SmartSelection, options?: {
|
|
81
|
+
syncDomSelection?: boolean;
|
|
82
|
+
}): void;
|
|
63
83
|
/**
|
|
64
84
|
* Renderer-integrated content-visibility (Phase 11 Tier 3): the naive
|
|
65
85
|
* per-block experiment (stamping content-visibility:auto on every
|
|
@@ -140,6 +140,27 @@ export class FoundationSubtreeRenderer {
|
|
|
140
140
|
element.style.lineHeight = String(node.attrs.lineHeight);
|
|
141
141
|
else if (element.style.lineHeight)
|
|
142
142
|
element.style.removeProperty("line-height");
|
|
143
|
+
if (node.type === "blockquote") {
|
|
144
|
+
// Inline styles always beat theme.ts's own `.srte-editor
|
|
145
|
+
// [contenteditable] blockquote { border-left: ...; background: ...;
|
|
146
|
+
// color: ...; }` rule regardless of selector specificity, so
|
|
147
|
+
// clearing an attr here lets that default show through again
|
|
148
|
+
// automatically - same pattern as align/indentLevel/lineHeight
|
|
149
|
+
// above, just scoped to this one node type since these three are
|
|
150
|
+
// blockquote-only, not shared block attrs.
|
|
151
|
+
if (node.attrs?.backgroundColor)
|
|
152
|
+
element.style.background = String(node.attrs.backgroundColor);
|
|
153
|
+
else
|
|
154
|
+
element.style.removeProperty("background");
|
|
155
|
+
if (node.attrs?.textColor)
|
|
156
|
+
element.style.color = String(node.attrs.textColor);
|
|
157
|
+
else
|
|
158
|
+
element.style.removeProperty("color");
|
|
159
|
+
if (node.attrs?.borderLeft)
|
|
160
|
+
element.style.borderLeft = String(node.attrs.borderLeft);
|
|
161
|
+
else
|
|
162
|
+
element.style.removeProperty("border-left");
|
|
163
|
+
}
|
|
143
164
|
if (node.type === "code_block") {
|
|
144
165
|
const language = typeof node.attrs?.language === "string" && node.attrs.language.trim()
|
|
145
166
|
? node.attrs.language.trim()
|
|
@@ -819,12 +840,12 @@ export class FoundationSubtreeRenderer {
|
|
|
819
840
|
return selection.anchorNode === anchor.node && selection.anchorOffset === anchor.offset
|
|
820
841
|
&& selection.focusNode === head.node && selection.focusOffset === head.offset;
|
|
821
842
|
}
|
|
822
|
-
restoreSelection(model) {
|
|
843
|
+
restoreSelection(model, syncDom = true) {
|
|
823
844
|
const selection = this.root.ownerDocument.getSelection();
|
|
824
845
|
if (!selection)
|
|
825
846
|
return;
|
|
826
847
|
if (model.type === "none") {
|
|
827
|
-
if (selection.rangeCount)
|
|
848
|
+
if (syncDom && selection.rangeCount)
|
|
828
849
|
selection.removeAllRanges();
|
|
829
850
|
return;
|
|
830
851
|
}
|
|
@@ -832,7 +853,7 @@ export class FoundationSubtreeRenderer {
|
|
|
832
853
|
const head = this.mapping.posToDom(model.head);
|
|
833
854
|
if (!anchor || !head)
|
|
834
855
|
return;
|
|
835
|
-
if (!this.selectionMatches(selection, anchor, head)) {
|
|
856
|
+
if (syncDom && !this.selectionMatches(selection, anchor, head)) {
|
|
836
857
|
if (typeof selection.setBaseAndExtent === "function") {
|
|
837
858
|
selection.setBaseAndExtent(anchor.node, anchor.offset, head.node, head.offset);
|
|
838
859
|
}
|
|
@@ -937,8 +958,12 @@ export class FoundationSubtreeRenderer {
|
|
|
937
958
|
* must not leave a stale checkbox projection behind. Reconcile these
|
|
938
959
|
* projections from the current model after every render; attribute writes
|
|
939
960
|
* remain idempotent.
|
|
961
|
+
*
|
|
962
|
+
* The same single document walk also collects every captioned block
|
|
963
|
+
* image for syncImageCaptions, so captions add no second traversal.
|
|
940
964
|
*/
|
|
941
965
|
syncListProjections(document) {
|
|
966
|
+
const captions = new Map();
|
|
942
967
|
const visit = (node, listDepth = 0) => {
|
|
943
968
|
if (isTextNode(node))
|
|
944
969
|
return;
|
|
@@ -947,10 +972,76 @@ export class FoundationSubtreeRenderer {
|
|
|
947
972
|
if (element)
|
|
948
973
|
this.syncNodeAttributes(element, node, listDepth);
|
|
949
974
|
}
|
|
975
|
+
else if (node.type === "block_image" && typeof node.attrs?.caption === "string" && node.attrs.caption) {
|
|
976
|
+
captions.set(node.id, node);
|
|
977
|
+
}
|
|
950
978
|
const childListDepth = node.type === "list" ? listDepth + 1 : listDepth;
|
|
951
979
|
node.children?.forEach((child) => visit(child, childListDepth));
|
|
952
980
|
};
|
|
953
981
|
visit(document);
|
|
982
|
+
this.syncImageCaptions(captions);
|
|
983
|
+
}
|
|
984
|
+
/**
|
|
985
|
+
* A block image's caption is a renderer projection - a non-editable
|
|
986
|
+
* sibling element placed immediately after the <img>, since an <img> can't
|
|
987
|
+
* hold children the way <table> holds its caption projection. Projection
|
|
988
|
+
* elements are already excluded from modelChildren/modelDomChildren, so
|
|
989
|
+
* the caption never enters the model or the selection mapping.
|
|
990
|
+
*
|
|
991
|
+
* diffElement moves, replaces and removes model elements without knowing
|
|
992
|
+
* about this sibling, so instead of patching every one of those paths this
|
|
993
|
+
* re-asserts the invariant after each render pass: every captioned image
|
|
994
|
+
* is followed directly by exactly one caption projection with the current
|
|
995
|
+
* text, and no other caption projection exists. Reuses (moves) an
|
|
996
|
+
* existing projection element rather than recreating it.
|
|
997
|
+
*/
|
|
998
|
+
syncImageCaptions(captions) {
|
|
999
|
+
const existing = new Map();
|
|
1000
|
+
this.root.querySelectorAll(`[${SMART_PROJECTION_ATTRIBUTE}="image-caption"]`).forEach((element) => {
|
|
1001
|
+
const owner = element.getAttribute("data-smart-caption-for") || "";
|
|
1002
|
+
if (!captions.has(owner) || existing.has(owner)) {
|
|
1003
|
+
element.remove();
|
|
1004
|
+
this.recordWrite(owner);
|
|
1005
|
+
}
|
|
1006
|
+
else
|
|
1007
|
+
existing.set(owner, element);
|
|
1008
|
+
});
|
|
1009
|
+
captions.forEach((node, id) => {
|
|
1010
|
+
const image = this.mapping.nodeToDom(id);
|
|
1011
|
+
if (!image?.parentNode || image === this.root)
|
|
1012
|
+
return;
|
|
1013
|
+
let caption = existing.get(id);
|
|
1014
|
+
if (!caption) {
|
|
1015
|
+
caption = image.ownerDocument.createElement("div");
|
|
1016
|
+
caption.setAttribute(SMART_PROJECTION_ATTRIBUTE, "image-caption");
|
|
1017
|
+
caption.setAttribute("data-smart-caption-for", id);
|
|
1018
|
+
caption.setAttribute("aria-hidden", "true");
|
|
1019
|
+
caption.setAttribute("contenteditable", "false");
|
|
1020
|
+
this.recordWrite(id);
|
|
1021
|
+
}
|
|
1022
|
+
if (image.nextSibling !== caption) {
|
|
1023
|
+
image.after(caption);
|
|
1024
|
+
this.recordWrite(id);
|
|
1025
|
+
}
|
|
1026
|
+
const text = String(node.attrs?.caption);
|
|
1027
|
+
if (caption.textContent !== text) {
|
|
1028
|
+
caption.textContent = text;
|
|
1029
|
+
this.recordWrite(id);
|
|
1030
|
+
}
|
|
1031
|
+
const align = node.attrs?.align === "left" || node.attrs?.align === "center" || node.attrs?.align === "right" ? String(node.attrs.align) : null;
|
|
1032
|
+
if (align)
|
|
1033
|
+
this.setAttribute(caption, "data-smart-align", align, id);
|
|
1034
|
+
else
|
|
1035
|
+
this.removeAttribute(caption, "data-smart-align", id);
|
|
1036
|
+
// Wrap to the image's own width when it has one, so a narrow image
|
|
1037
|
+
// doesn't get a page-wide caption line.
|
|
1038
|
+
const width = Number(node.attrs?.width);
|
|
1039
|
+
const maxWidth = Number.isFinite(width) && width > 0 ? `${width}px` : "";
|
|
1040
|
+
if (caption.style.maxWidth !== maxWidth) {
|
|
1041
|
+
caption.style.maxWidth = maxWidth;
|
|
1042
|
+
this.recordWrite(id);
|
|
1043
|
+
}
|
|
1044
|
+
});
|
|
954
1045
|
}
|
|
955
1046
|
announceSelectedLevel(before, after, selection) {
|
|
956
1047
|
const previous = this.listItemDepths(before);
|
|
@@ -983,7 +1074,8 @@ export class FoundationSubtreeRenderer {
|
|
|
983
1074
|
}
|
|
984
1075
|
this.liveRegion.textContent = `List level ${active[1].depth + 1}`;
|
|
985
1076
|
}
|
|
986
|
-
render(document, selection) {
|
|
1077
|
+
render(document, selection, options = {}) {
|
|
1078
|
+
const syncDom = options.syncDomSelection !== false;
|
|
987
1079
|
if (!this.current) {
|
|
988
1080
|
this.mapping.beginUpdate(document);
|
|
989
1081
|
document.children.forEach((node, index) => this.root.appendChild(this.createNode(node, [index])));
|
|
@@ -992,12 +1084,12 @@ export class FoundationSubtreeRenderer {
|
|
|
992
1084
|
this.syncListProjections(document);
|
|
993
1085
|
this.syncTableAccessibility();
|
|
994
1086
|
this.modelById.set(document.id, document);
|
|
995
|
-
this.restoreSelection(selection);
|
|
1087
|
+
this.restoreSelection(selection, syncDom);
|
|
996
1088
|
this.syncCellSelectionProjection(selection);
|
|
997
1089
|
return;
|
|
998
1090
|
}
|
|
999
1091
|
if (this.current === document) {
|
|
1000
|
-
this.restoreSelection(selection);
|
|
1092
|
+
this.restoreSelection(selection, syncDom);
|
|
1001
1093
|
this.syncCellSelectionProjection(selection);
|
|
1002
1094
|
this.syncContentVisibility(document, document, selection);
|
|
1003
1095
|
return;
|
|
@@ -1012,7 +1104,7 @@ export class FoundationSubtreeRenderer {
|
|
|
1012
1104
|
this.syncListProjections(document);
|
|
1013
1105
|
if (structural)
|
|
1014
1106
|
this.syncTableAccessibility();
|
|
1015
|
-
this.restoreSelection(selection);
|
|
1107
|
+
this.restoreSelection(selection, syncDom);
|
|
1016
1108
|
this.syncCellSelectionProjection(selection);
|
|
1017
1109
|
this.syncContentVisibility(before, document, selection);
|
|
1018
1110
|
this.announceSelectedLevel(before, document, selection);
|
|
@@ -6,7 +6,15 @@ export interface CanonicalSubtreeRenderer {
|
|
|
6
6
|
readonly composingNodeId: string | null;
|
|
7
7
|
readonly domWriteCount: number;
|
|
8
8
|
readonly composingDomWriteCount: number;
|
|
9
|
-
|
|
9
|
+
/**
|
|
10
|
+
* `syncDomSelection: false` renders and scrolls the selection into view
|
|
11
|
+
* without moving the browser selection into the editor - setting it
|
|
12
|
+
* would pull focus out of a host's own UI (e.g. a side panel calling
|
|
13
|
+
* insertImage). Default true.
|
|
14
|
+
*/
|
|
15
|
+
render(document: SmartDocument, selection: SmartSelection, options?: {
|
|
16
|
+
syncDomSelection?: boolean;
|
|
17
|
+
}): void;
|
|
10
18
|
beginComposition(nodeId: string): void;
|
|
11
19
|
endComposition(): void;
|
|
12
20
|
resetWriteCounters(): void;
|
|
@@ -45,4 +53,6 @@ export interface CanonicalInputPipelineOptions {
|
|
|
45
53
|
onFiles?: (files: readonly File[], position: SmartSelection) => void;
|
|
46
54
|
/** Privacy-safe clipboard telemetry; reports hashes and structure, never text. */
|
|
47
55
|
onClipboardDiagnostic?: (report: ClipboardDiagnosticReport) => void;
|
|
56
|
+
/** Host-defined `data-*` attributes kept on pasted block images (see parseCanonicalListHtml's option of the same name). Default: none. */
|
|
57
|
+
imageMetadataAttributes?: readonly string[];
|
|
48
58
|
}
|