smartrte-core 1.0.0-beta.2 → 1.0.0-beta.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,13 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.0.0-beta.4
4
+
5
+ - Fix HTML import (paste and loading a document's initial value) rendering real content as unreadable `[Unsupported: ...]` placeholders when a bare `<span>`/`<b>`/other inline-formatting element sits directly at the document root, or is interleaved between real blocks (a `<blockquote>`, `<div>`-wrapped lines, images) with no wrapping `<p>` — a common shape from legacy, pre-migration editor exports. Also fixes the same gap in `<blockquote>`/table cells/list items for content interleaved between blocks (previously only content entirely before any block content was recovered correctly).
6
+
7
+ ## 1.0.0-beta.3
8
+
9
+ - Add `href`/`target` (a clickable link), `borderRadius` (corner rounding), and license metadata fields (`licenseDescription`, `licenseSourceUrl`, `licenseType`, `licenseVersion`, `licenseAttribution`) to the image atom schema. `href` is validated the same way as the existing `link` mark; exported HTML wraps a linked image in a real `<a>` for portability outside the editor. Fix a pre-existing renderer gap where clearing an image's `align` back to unset never removed the CSS float/display/margin it had previously applied.
10
+
3
11
  ## 1.0.0-beta.2
4
12
 
5
13
  - Add an opt-in `renderFormulaHtml` option to `serializeCanonicalListHtml` that bakes real KaTeX-rendered HTML into exported formula elements, instead of leaving an empty placeholder — for consumers (e.g. a read-only preview) that display this HTML directly without also running KaTeX against it themselves. Off by default; no change for existing consumers.
@@ -3,7 +3,7 @@ import type { SmartElementNode } from "../types.js";
3
3
  export declare const atomToHtml: (node: SmartElementNode, options?: {
4
4
  renderFormulaHtml?: boolean;
5
5
  }) => string;
6
- export declare const atomFromHtmlElement: (element: Element) => SmartElementNode | null;
6
+ export declare const atomFromHtmlElement: (rawElement: Element) => SmartElementNode | null;
7
7
  export declare const atomToMarkdown: (node: SmartElementNode) => string;
8
8
  export interface AtomDocxRun {
9
9
  readonly kind: "image" | "text";
@@ -6,6 +6,7 @@ import katex from "katex";
6
6
  // fall back to plain text in the exported/baked HTML.
7
7
  import "katex/contrib/mhchem";
8
8
  import { createNodeId } from "../identity.js";
9
+ import { sanitizeLinkHref, sanitizeLinkTarget } from "../security/urlPolicy.js";
9
10
  import { sanitizeAtomSource } from "./security.js";
10
11
  const escape = (value) => String(value ?? "").replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;").replace(/"/g, "&quot;");
11
12
  const attr = (name, value) => value === undefined ? "" : ` ${name}="${escape(value)}"`;
@@ -25,7 +26,23 @@ const renderFormulaToHtml = (source) => {
25
26
  export const atomToHtml = (node, options) => {
26
27
  if (node.type === "image" || node.type === "block_image") {
27
28
  const src = sanitizeAtomSource(String(node.attrs?.src || ""), { kind: "image", allowBlobPreview: node.attrs?.status === "pending" }) || "";
28
- return `<img data-smart-id="${escape(node.id)}" data-smart-type="${node.type}" src="${escape(src)}" alt="${escape(node.attrs?.alt)}"${dimensions(node)}${attr("data-smart-status", node.attrs?.status || "ready")}${node.attrs?.decorative === true ? ' data-smart-decorative="true"' : ""}${attr("data-smart-align", node.attrs?.align)}>`;
29
+ // Corner radius and license fields are data-attributes only (round-trip
30
+ // fidelity), matching this same function's existing data-smart-align -
31
+ // a raw HTML consumer displaying this string directly without also
32
+ // running the live surface renderer's own style logic against it
33
+ // wouldn't see visual rounding either way, same as align today; that's
34
+ // a deliberate, pre-existing scope boundary this doesn't change.
35
+ const img = `<img data-smart-id="${escape(node.id)}" data-smart-type="${node.type}" src="${escape(src)}" alt="${escape(node.attrs?.alt)}"${dimensions(node)}${attr("data-smart-status", node.attrs?.status || "ready")}${node.attrs?.decorative === true ? ' data-smart-decorative="true"' : ""}${attr("data-smart-align", node.attrs?.align)}${attr("data-smart-radius", node.attrs?.borderRadius)}${attr("data-smart-license-description", node.attrs?.licenseDescription)}${attr("data-smart-license-source-url", node.attrs?.licenseSourceUrl)}${attr("data-smart-license-type", node.attrs?.licenseType)}${attr("data-smart-license-version", node.attrs?.licenseVersion)}${attr("data-smart-license-attribution", node.attrs?.licenseAttribution)}${attr("data-smart-href", node.attrs?.href)}${attr("data-smart-target", node.attrs?.target)}>`;
36
+ // A real <a> wrapper, unlike the live surface renderer's plain data
37
+ // attributes on the bare <img> - see surface/renderer.ts's own
38
+ // reasoning (stable-element-per-id constraint that doesn't apply to
39
+ // this string-based export) and modelDom.ts's identical choice for the
40
+ // one-shot clipboard/print DOM builder.
41
+ const linkHref = sanitizeLinkHref(typeof node.attrs?.href === "string" ? node.attrs.href : undefined);
42
+ if (!linkHref)
43
+ return img;
44
+ const target = sanitizeLinkTarget(typeof node.attrs?.target === "string" ? node.attrs.target : undefined);
45
+ return `<a href="${escape(linkHref)}"${target ? ` target="${escape(target)}"` : ""}>${img}</a>`;
29
46
  }
30
47
  if (node.type === "formula" || node.type === "block_formula") {
31
48
  const tag = node.type === "formula" ? "span" : "div";
@@ -48,7 +65,14 @@ export const atomToHtml = (node, options) => {
48
65
  return `<hr data-smart-id="${escape(node.id)}" data-smart-type="divider">`;
49
66
  throw new Error(`Unsupported atom type "${node.type}".`);
50
67
  };
51
- export const atomFromHtmlElement = (element) => {
68
+ export const atomFromHtmlElement = (rawElement) => {
69
+ // atomToHtml's own inverse - a linked image now wraps in a real <a>
70
+ // (needed so the link survives outside a live editor at all, see
71
+ // atomToHtml's own doc comment); unwrap to the actual atom element
72
+ // (which still carries every attribute, including data-smart-href,
73
+ // directly) before parsing, matching what a single element's own
74
+ // attributes already fully describe.
75
+ const element = rawElement.tagName === "A" && rawElement.children.length === 1 ? rawElement.children[0] : rawElement;
52
76
  const declared = element.getAttribute("data-smart-type");
53
77
  const type = declared || (element.tagName === "IMG" ? "image" : element.tagName.toLowerCase());
54
78
  const id = element.getAttribute("data-smart-id") || createNodeId();
@@ -57,7 +81,21 @@ export const atomFromHtmlElement = (element) => {
57
81
  const src = sanitizeAtomSource(element.getAttribute("src"), { kind: "image" });
58
82
  if (!src)
59
83
  return null;
60
- return { type, id, attrs: { src, alt: element.getAttribute("alt") || "", status: element.getAttribute("data-smart-status") || "ready", ...(element.getAttribute("data-smart-decorative") === "true" ? { decorative: true } : {}), ...(element.getAttribute("data-smart-align") ? { align: element.getAttribute("data-smart-align") } : {}), ...(number("width") ? { width: number("width") } : {}), ...(number("height") ? { height: number("height") } : {}) } };
84
+ const radius = Number(element.getAttribute("data-smart-radius"));
85
+ return { type, id, attrs: {
86
+ src, alt: element.getAttribute("alt") || "", status: element.getAttribute("data-smart-status") || "ready",
87
+ ...(element.getAttribute("data-smart-decorative") === "true" ? { decorative: true } : {}),
88
+ ...(element.getAttribute("data-smart-align") ? { align: element.getAttribute("data-smart-align") } : {}),
89
+ ...(number("width") ? { width: number("width") } : {}), ...(number("height") ? { height: number("height") } : {}),
90
+ ...(Number.isFinite(radius) && radius > 0 ? { borderRadius: radius } : {}),
91
+ ...(element.getAttribute("data-smart-href") ? { href: element.getAttribute("data-smart-href") } : {}),
92
+ ...(element.getAttribute("data-smart-target") ? { target: element.getAttribute("data-smart-target") } : {}),
93
+ ...(element.getAttribute("data-smart-license-description") ? { licenseDescription: element.getAttribute("data-smart-license-description") } : {}),
94
+ ...(element.getAttribute("data-smart-license-source-url") ? { licenseSourceUrl: element.getAttribute("data-smart-license-source-url") } : {}),
95
+ ...(element.getAttribute("data-smart-license-type") ? { licenseType: element.getAttribute("data-smart-license-type") } : {}),
96
+ ...(element.getAttribute("data-smart-license-version") ? { licenseVersion: element.getAttribute("data-smart-license-version") } : {}),
97
+ ...(element.getAttribute("data-smart-license-attribution") ? { licenseAttribution: element.getAttribute("data-smart-license-attribution") } : {}),
98
+ } };
61
99
  }
62
100
  if (type === "formula" || type === "block_formula") {
63
101
  return { type, id, attrs: { source: element.getAttribute("data-smart-formula") || "", notation: element.getAttribute("data-smart-notation") === "mathml" ? "mathml" : "latex" } };
@@ -1,3 +1,4 @@
1
+ import { normalizeLinkInput } from "../security/urlPolicy.js";
1
2
  const optionalString = { validate: (value) => typeof value === "string" };
2
3
  const requiredString = { required: true, validate: (value) => typeof value === "string" };
3
4
  const dimension = { validate: (value) => Number.isFinite(value) && Number(value) > 0 && Number(value) <= 100000 };
@@ -6,6 +7,34 @@ const imageAttrs = {
6
7
  src: requiredString, alt: requiredString, width: dimension, height: dimension,
7
8
  status, uploadId: optionalString, error: optionalString, decorative: { validate: (value) => typeof value === "boolean" },
8
9
  align: { validate: (value) => value === "center" || value === "left" || value === "right" },
10
+ /**
11
+ * docs/bugs/media-details-old-editor-field-parity.md: the old editor's
12
+ * "Link"/"Target" fields have no equivalent yet - the `link` mark already
13
+ * has this exact href/target shape (marks/commands.ts), but marks are
14
+ * explicitly disallowed on every atom node (`marks: ""` below), so this
15
+ * is a dedicated pair of atom-level attrs instead of reusing the mark.
16
+ */
17
+ /** Same normalizeLinkInput validation the "link" mark's own schema uses (marks/schema.ts) - rejects at the command layer, not just sanitizing later at render/export time. */
18
+ href: { validate: (value) => typeof value === "string" && normalizeLinkInput(value).href !== null },
19
+ target: optionalString,
20
+ /** Corner rounding (px) - confirmed absent anywhere in this schema before now, unlike align/alt/width which already existed with no UI. */
21
+ borderRadius: { validate: (value) => Number.isFinite(value) && Number(value) >= 0 && Number(value) <= 1000 },
22
+ /**
23
+ * Mirrors MediaItem.license's own shape (packages/react/src/mediaProvider.ts)
24
+ * field-for-field where a real equivalent exists, so a library-sourced
25
+ * image's already-fetched license metadata can be copied straight across
26
+ * at insert time (CanonicalAuthorityEditor.tsx's selectFromMediaManager)
27
+ * instead of being read, shown in the library browser, and then silently
28
+ * discarded the way it was before this - the atom schema had nowhere to
29
+ * persist it. Renamed to this project's own vocabulary (the old editor's
30
+ * "Attribution text" ~ MediaItem.license.author, "License description" ~
31
+ * workName/licenseText) and kept independently user-editable, since a
32
+ * freshly uploaded image has no provider-supplied license data at all.
33
+ * `licenseVersion` has no MediaItem.license equivalent (its shape has no
34
+ * separate version field) - always user-entered, never auto-populated.
35
+ */
36
+ licenseDescription: optionalString, licenseSourceUrl: optionalString,
37
+ licenseType: optionalString, licenseVersion: optionalString, licenseAttribution: optionalString,
9
38
  };
10
39
  const formulaAttrs = {
11
40
  source: requiredString,
@@ -17,9 +17,26 @@ const isEditorUiNode = (node) => attr(node, "data-smart-ui") !== undefined || at
17
17
  * meaningful block content - most often a <table> - in one or more <div>s
18
18
  * (table-wrapper divs, layout divs) that carry no semantic meaning of
19
19
  * their own. parseBlock has no case for any of these tags; see
20
- * parseBlockList below for how they're unwrapped instead of swallowed.
20
+ * parseMixedBlockContent below for how they're unwrapped instead of swallowed.
21
21
  */
22
22
  const TRANSPARENT_CONTAINER_TAGS = ["div", "section", "article", "figure"];
23
+ /** Shared by both the inline `image` and block `block_image` parsing branches below - radius/license round-trip identically for both, only href/target's link-mark-fallback behavior deliberately differs between them (see each call site's own comment). */
24
+ const imageStyleAndLicenseAttrs = (node) => {
25
+ const radius = Number(attr(node, "data-smart-radius"));
26
+ const licenseDescription = attr(node, "data-smart-license-description");
27
+ const licenseSourceUrl = attr(node, "data-smart-license-source-url");
28
+ const licenseType = attr(node, "data-smart-license-type");
29
+ const licenseVersion = attr(node, "data-smart-license-version");
30
+ const licenseAttribution = attr(node, "data-smart-license-attribution");
31
+ return {
32
+ ...(Number.isFinite(radius) && radius > 0 ? { borderRadius: radius } : {}),
33
+ ...(licenseDescription ? { licenseDescription } : {}),
34
+ ...(licenseSourceUrl ? { licenseSourceUrl } : {}),
35
+ ...(licenseType ? { licenseType } : {}),
36
+ ...(licenseVersion ? { licenseVersion } : {}),
37
+ ...(licenseAttribution ? { licenseAttribution } : {}),
38
+ };
39
+ };
23
40
  /** See serializeCanonicalListHtml's own doc comment for why this is a module-scoped flag rather than a threaded parameter. */
24
41
  let renderFormulaHtmlMode = false;
25
42
  const serializeInline = (node) => {
@@ -227,12 +244,27 @@ const textWithMarks = (node, inherited = []) => {
227
244
  // synchronous and the image hasn't loaded yet at this point.
228
245
  const width = parsePixelWidth(styleValue(node, "width")) ?? parsePixelWidth(attr(node, "width"));
229
246
  const height = parsePixelWidth(styleValue(node, "height")) ?? parsePixelWidth(attr(node, "height"));
247
+ // <a><img></a> (this project's own export - list/formats.ts's own
248
+ // serializeBlock/atom/formats.ts's atomToHtml wrap a linked image in a
249
+ // real anchor - or any real-world source doing the same): the link
250
+ // mark from that ancestor <a> arrives here via `inherited`, but
251
+ // SmartElementNode has no `marks` field at all (only SmartTextNode
252
+ // does) - without this, the href would be silently discarded the
253
+ // moment it reached this atomic-image branch. The live editor's own
254
+ // DOM instead encodes this as plain data-smart-href/-target
255
+ // attributes directly on the <img> (surface/renderer.ts) with no
256
+ // wrapping <a> at all, so both encodings are checked.
257
+ const inheritedLink = inherited.find((mark) => mark.type === "link");
258
+ const href = attr(node, "data-smart-href") || inheritedLink?.attrs?.href;
259
+ const target = attr(node, "data-smart-target") || inheritedLink?.attrs?.target;
230
260
  return [{ type: "image", id: generatedId(node, "image"), attrs: {
231
261
  src, alt: attr(node, "alt") || "", status: attr(node, "data-smart-status") || "ready",
232
262
  ...(attr(node, "data-smart-decorative") === "true" ? { decorative: true } : {}),
233
263
  ...(attr(node, "title") ? { title: attr(node, "title") } : {}),
234
264
  ...(attr(node, "data-smart-align") ? { align: attr(node, "data-smart-align") } : {}),
235
265
  ...(width !== null ? { width } : {}), ...(height !== null ? { height } : {}),
266
+ ...(href ? { href } : {}), ...(target ? { target } : {}),
267
+ ...imageStyleAndLicenseAttrs(node),
236
268
  } }];
237
269
  }
238
270
  }
@@ -375,6 +407,19 @@ const parseBlock = (node) => {
375
407
  src, alt: attr(node, "alt") || "", status: attr(node, "data-smart-status") || "ready",
376
408
  ...(attr(node, "data-smart-decorative") === "true" ? { decorative: true } : {}),
377
409
  ...(width !== null ? { width } : {}), ...(height !== null ? { height } : {}),
410
+ // Only from data-smart-href/-target directly on this element (our own
411
+ // round-tripped export, per this whole branch's own `declaredAtom`
412
+ // gate) - deliberately NOT also checking for a wrapping <a> the way
413
+ // the inline `image` atom's parsing does just above in textWithMarks.
414
+ // The bare-<img>-at-block-level fallback a few lines down has its own
415
+ // long-standing, evidence-based decision (a real Wikipedia "copy
416
+ // image" paste) to unwrap and drop an incidental <a> wrapper rather
417
+ // than treat it as a user-authored link - extending that to also
418
+ // read href here would silently reverse that decision for every
419
+ // third-party block-level image paste, not just this app's own.
420
+ ...(attr(node, "data-smart-href") ? { href: attr(node, "data-smart-href") } : {}),
421
+ ...(attr(node, "data-smart-target") ? { target: attr(node, "data-smart-target") } : {}),
422
+ ...imageStyleAndLicenseAttrs(node),
378
423
  } };
379
424
  }
380
425
  if (declaredAtom === "block_formula")
@@ -440,17 +485,10 @@ const parseBlock = (node) => {
440
485
  if (tag === "blockquote") {
441
486
  // Real-world exports (Sootr among them) put inline content - <span>
442
487
  // wrapper runs, bare text, <br> - directly inside <blockquote> with no
443
- // wrapping <p>. Without this split, elementChildren+parseBlock alone
444
- // sent every such child through the generic "unrecognized tag" fallback
445
- // at the bottom of this function, producing an unknown block node per
446
- // span (rendered as "[Unsupported: span]") instead of parsed text/marks
447
- // - the same directInline pattern td/li already use below.
488
+ // wrapping <p>. See parseMixedBlockContent's own comment for why this
489
+ // must be grouped run-by-run rather than collected into one paragraph.
448
490
  const blockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "table"];
449
- const isBlockLike = (child) => blockTags.includes(child.tagName || "") || TRANSPARENT_CONTAINER_TAGS.includes(child.tagName || "");
450
- const children = parseBlockList(elementChildren(node).filter(isBlockLike));
451
- const directInline = (node.childNodes || []).filter((child) => !child.tagName || !isBlockLike(child)).flatMap((child) => textWithMarks(child));
452
- if (directInline.length)
453
- children.unshift({ type: "paragraph", id: createNodeId(), children: directInline });
491
+ const children = parseMixedBlockContent(node.childNodes || [], blockTags);
454
492
  return {
455
493
  type: "blockquote", id: generatedId(node, "quote"),
456
494
  ...(Object.keys(parsedBlockAttrs(node)).length ? { attrs: parsedBlockAttrs(node) } : {}),
@@ -539,11 +577,7 @@ const parseBlock = (node) => {
539
577
  if (verticalAlign)
540
578
  cellAttrs.verticalAlign = verticalAlign;
541
579
  const blockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "table"];
542
- const isBlockLike = (child) => blockTags.includes(child.tagName || "") || TRANSPARENT_CONTAINER_TAGS.includes(child.tagName || "");
543
- const children = parseBlockList(elementChildren(node).filter(isBlockLike));
544
- const directInline = (node.childNodes || []).filter((child) => !child.tagName || !isBlockLike(child)).flatMap((child) => textWithMarks(child));
545
- if (directInline.length)
546
- children.unshift({ type: "paragraph", id: createNodeId(), children: directInline });
580
+ const children = parseMixedBlockContent(node.childNodes || [], blockTags);
547
581
  if (!children.length)
548
582
  children.push({ type: "paragraph", id: createNodeId(), children: [] });
549
583
  return { type: "table_cell", id: generatedId(node, "cell"), attrs: cellAttrs, children };
@@ -591,17 +625,8 @@ const parseBlock = (node) => {
591
625
  attrs.htmlStyle = htmlStyle;
592
626
  if (Number.isInteger(value) && value >= 1)
593
627
  attrs.numberOverride = value;
594
- const children = [];
595
- const blockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "div", "table", "figure"];
596
- const inlineNodes = (node.childNodes || []).filter((child) => !child.tagName || !blockTags.includes(child.tagName));
597
- const directText = inlineNodes.flatMap((child) => textWithMarks(child));
598
- if (directText.length)
599
- children.push({ type: "paragraph", id: createNodeId(), children: directText });
600
- // blockTags already lists "div"/"figure" as block-worthy, but parseBlock
601
- // itself has no case for either - parseBlockList is what actually
602
- // unwraps them (rather than swallowing their content into one opaque
603
- // `unknown` node) instead of parsing them directly.
604
- children.push(...parseBlockList(elementChildren(node).filter((child) => blockTags.includes(child.tagName || ""))));
628
+ const blockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "table"];
629
+ const children = parseMixedBlockContent(node.childNodes || [], blockTags);
605
630
  if (!children.length)
606
631
  children.push({ type: "paragraph", id: createNodeId(), children: [] });
607
632
  return { type: "list_item", id: generatedId(node, "item"), ...(Object.keys(attrs).length ? { attrs } : {}), children };
@@ -614,24 +639,117 @@ const parseBlock = (node) => {
614
639
  return null;
615
640
  };
616
641
  /**
617
- * A thin wrapper around parseBlock for "list of block-level children"
618
- * call sites: a transparent container (TRANSPARENT_CONTAINER_TAGS) is
619
- * recursed into and its own children spliced in flat, rather than parsed
620
- * as one opaque node - a div wrapping N real blocks (a table, N
621
- * paragraphs, or nothing at all) produces exactly those N blocks instead
622
- * of a single `unknown` placeholder that hides all of them.
642
+ * Generic inline-formatting tags recognized by textWithMarks. Used by
643
+ * parseMixedBlockContent to decide when a bare (non-<p>-wrapped) element
644
+ * sitting at block level should be reinterpreted as inline text instead of
645
+ * left as parseBlock's own "unknown" round-trip-preserving placeholder.
646
+ * Deliberately narrow: an arbitrary/custom element (e.g. a host's own safe
647
+ * custom widget surviving the clipboard security boundary, see
648
+ * clipboard/pipeline.test.ts's "preserves a sanitized safe custom element
649
+ * as an unknown node") is NOT in this list, so it keeps going through the
650
+ * exact same "unknown" fallback as before - only tags that are
651
+ * unambiguously plain inline formatting ever get reinterpreted as text.
623
652
  */
624
- const parseBlockList = (nodes) => nodes.flatMap((node) => {
625
- if (TRANSPARENT_CONTAINER_TAGS.includes(node.tagName || ""))
626
- return parseBlockList(elementChildren(node));
627
- const parsed = parseBlock(node);
628
- return parsed ? [parsed] : [];
629
- });
653
+ const GENERIC_INLINE_TAGS = ["span", "strong", "b", "em", "i", "u", "s", "strike", "del", "code", "sup", "sub", "a", "br"];
654
+ /**
655
+ * Parses a mixed sequence of block and inline DOM children in place. A
656
+ * transparent container (TRANSPARENT_CONTAINER_TAGS) is recursed into and
657
+ * its own children spliced in flat, rather than parsed as one opaque node -
658
+ * a div wrapping N real blocks (a table, N paragraphs, or nothing at all)
659
+ * produces exactly those N blocks instead of a single `unknown` placeholder
660
+ * that hides all of them.
661
+ *
662
+ * Real-world exports (Sootr among them - both the document root and inside
663
+ * a <blockquote>/<td>/<li>) routinely put inline content - bare <span>
664
+ * wrapper runs, plain text, stray <br> - directly alongside real block
665
+ * children, with no wrapping <p>. Each *run* of consecutive non-block-like
666
+ * children is grouped into its own synthetic paragraph in place, preserving
667
+ * both the order of the surrounding real blocks and which inline runs were
668
+ * actually adjacent in the source. An earlier version of this function
669
+ * collected every inline child across the whole container into ONE
670
+ * paragraph and prepended it - correct only when inline content never
671
+ * appears between two blocks. Real legacy content interleaves several bare
672
+ * <span> lines between a <blockquote>, images, and paragraphs; merging them
673
+ * all into one paragraph at the front silently reordered content and
674
+ * mashed together lines that were never adjacent, and any other stray
675
+ * generic inline tag (e.g. a lone <b>) still fell through to the generic
676
+ * unrecognized-tag fallback and rendered as "[Unsupported: b]" wherever no
677
+ * caller-specific inline handling existed for it (previously: the document
678
+ * root, which had none at all).
679
+ *
680
+ * "a" is the one GENERIC_INLINE_TAGS entry with a real block-level meaning
681
+ * too - parseBlock's own "<a> wrapping only an <img>" case unwraps to a
682
+ * block image - so it's tried as a block first and only folded into the
683
+ * inline run if that doesn't apply.
684
+ */
685
+ const parseMixedBlockContent = (nodes, blockTags) => {
686
+ const result = [];
687
+ let inlineRun = [];
688
+ const flushInlineRun = () => {
689
+ if (!inlineRun.length)
690
+ return;
691
+ const children = inlineRun.flatMap((node) => textWithMarks(node));
692
+ inlineRun = [];
693
+ // A whitespace-only text node (a newline/indentation between block-level
694
+ // tags in pretty-printed source HTML - e.g. Word/Excel's real clipboard
695
+ // HTML) is insignificant, exactly like a browser's own whitespace
696
+ // collapsing between block elements - it must not produce its own
697
+ // spurious empty-looking paragraph. A run with any real text or any
698
+ // non-text child (an atom, a <br>) is kept in full, including any
699
+ // whitespace mixed into it (e.g. a genuine word-separating space).
700
+ const hasContent = children.some((child) => !isTextNode(child) || child.text.trim() !== "");
701
+ if (hasContent)
702
+ result.push({ type: "paragraph", id: createNodeId(), children });
703
+ };
704
+ const pushBlock = (node) => {
705
+ flushInlineRun();
706
+ const parsed = parseBlock(node);
707
+ if (parsed)
708
+ result.push(parsed);
709
+ };
710
+ nodes.forEach((node) => {
711
+ // Matches textWithMarks's own first check - a UI-only element (e.g. a
712
+ // checklist item's check-control <button>) has no tag this function
713
+ // would otherwise recognize as block-like, and previously reached
714
+ // parseBlock's generic "unknown" fallback instead of being silently
715
+ // dropped like every other reparse path already does for these.
716
+ if (isEditorUiNode(node))
717
+ return;
718
+ const tag = node.tagName;
719
+ if (!tag) {
720
+ inlineRun.push(node);
721
+ return;
722
+ }
723
+ if (TRANSPARENT_CONTAINER_TAGS.includes(tag)) {
724
+ flushInlineRun();
725
+ result.push(...parseMixedBlockContent(node.childNodes || [], blockTags));
726
+ return;
727
+ }
728
+ if (blockTags.includes(tag))
729
+ return pushBlock(node);
730
+ if (GENERIC_INLINE_TAGS.includes(tag)) {
731
+ if (tag === "a") {
732
+ const parsed = parseBlock(node);
733
+ if (parsed && parsed.type !== "unknown") {
734
+ flushInlineRun();
735
+ result.push(parsed);
736
+ return;
737
+ }
738
+ }
739
+ inlineRun.push(node);
740
+ return;
741
+ }
742
+ pushBlock(node);
743
+ });
744
+ flushInlineRun();
745
+ return result;
746
+ };
630
747
  export const parseCanonicalListHtml = (html) => {
631
748
  const fragment = parseFragment(html);
632
749
  const wrapper = elementChildren(fragment).find((node) => attr(node, "data-smart-document") === "true");
633
750
  const source = wrapper || fragment;
634
- const children = parseBlockList(elementChildren(source));
751
+ const rootBlockTags = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "ul", "ol", "blockquote", "pre", "table", "hr", "img", "video", "audio"];
752
+ const children = parseMixedBlockContent(source.childNodes || [], rootBlockTags);
635
753
  return { type: "doc", id: attr(source, "data-smart-id") || createNodeId(), children: children.length ? children : [{ type: "paragraph", id: createNodeId(), children: [] }] };
636
754
  };
637
755
  const plainText = (node) => (node.children || []).map((child) => isTextNode(child) ? child.text : child.type === "hard_break" ? "\n" : "").join("");
@@ -2,6 +2,7 @@ import { isTextNode } from "./identity.js";
2
2
  import { nodeAtPath } from "./positions.js";
3
3
  import { renderMarkedText } from "./marks/dom.js";
4
4
  import { sanitizeAtomSource } from "./atom/security.js";
5
+ import { sanitizeLinkHref, sanitizeLinkTarget } from "./security/urlPolicy.js";
5
6
  const atomTypes = new Set(["image", "block_image", "formula", "block_formula", "video", "audio", "divider"]);
6
7
  export const SMART_UI_ATTRIBUTE = "data-smart-ui";
7
8
  export const SMART_PROJECTION_ATTRIBUTE = "data-smart-projection";
@@ -93,6 +94,23 @@ const renderNode = (node, ownerDocument) => {
93
94
  if (source)
94
95
  element.setAttribute("src", source);
95
96
  element.setAttribute("alt", typeof node.attrs?.alt === "string" ? node.attrs.alt : "");
97
+ if (node.attrs?.borderRadius)
98
+ element.style.borderRadius = `${Number(node.attrs.borderRadius)}px`;
99
+ const linkHref = sanitizeLinkHref(typeof node.attrs?.href === "string" ? node.attrs.href : undefined);
100
+ if (linkHref) {
101
+ // A real <a> wrapper, unlike the live surface renderer's plain data
102
+ // attributes - this is a one-shot tree builder (no per-node element
103
+ // identity to preserve across renders), and clipboard/print output
104
+ // needs a genuine anchor for the link to survive outside this editor
105
+ // at all.
106
+ const link = ownerDocument.createElement("a");
107
+ link.setAttribute("href", linkHref);
108
+ const linkTarget = sanitizeLinkTarget(typeof node.attrs?.target === "string" ? node.attrs.target : undefined);
109
+ if (linkTarget)
110
+ link.setAttribute("target", linkTarget);
111
+ link.appendChild(element);
112
+ return link;
113
+ }
96
114
  }
97
115
  if (node.type === "formula" || node.type === "block_formula") {
98
116
  const source = String(node.attrs?.source || "");
@@ -405,6 +405,62 @@ export class FoundationSubtreeRenderer {
405
405
  element.style.float = imageAlign;
406
406
  element.style.margin = imageAlign === "left" ? "0 8px 8px 0" : "0 0 8px 8px";
407
407
  }
408
+ else {
409
+ // No align set - reset every property the two branches above can
410
+ // set, not just skip setting new ones. A pre-existing gap (found
411
+ // while verifying docs/bugs/media-details-old-editor-field-parity.md's
412
+ // new "clear align back to None" UI path): this renderer's other
413
+ // conditional style branches (e.g. borderRadius just below) always
414
+ // pair their "set" case with an explicit "else remove" case for
415
+ // exactly this reason - align's own else branches never existed
416
+ // before this UI made "align was set, then explicitly cleared" a
417
+ // real, reachable transition for the first time.
418
+ element.style.removeProperty("display");
419
+ element.style.removeProperty("float");
420
+ element.style.removeProperty("margin");
421
+ }
422
+ if (node.attrs?.borderRadius)
423
+ element.style.borderRadius = `${Number(node.attrs.borderRadius)}px`;
424
+ else
425
+ element.style.removeProperty("border-radius");
426
+ // Plain data attributes, not a wrapping <a> - this element's identity
427
+ // is stable across renders (see this class's own diffing contract);
428
+ // introducing a conditional DOM-structure change here for something
429
+ // input.ts's click handling already reads directly is unnecessary
430
+ // risk. atomToHtml/modelDom.ts DO wrap in a real <a> for the
431
+ // string/one-shot export paths, which have no such stability
432
+ // constraint and need a real anchor for the link to work outside
433
+ // this live editor at all.
434
+ if (typeof node.attrs?.href === "string" && node.attrs.href)
435
+ this.setAttribute(element, "data-smart-href", node.attrs.href, node.id);
436
+ else
437
+ this.removeAttribute(element, "data-smart-href", node.id);
438
+ if (typeof node.attrs?.target === "string" && node.attrs.target)
439
+ this.setAttribute(element, "data-smart-target", node.attrs.target, node.id);
440
+ else
441
+ this.removeAttribute(element, "data-smart-target", node.id);
442
+ // Mirrored as data attributes too, matching href/target above - a
443
+ // host inspecting the live DOM (or this project's own e2e suite)
444
+ // should be able to see this data the same way it can see href, not
445
+ // just find it by reading the model directly. Purely descriptive
446
+ // metadata, so unlike align/borderRadius there's no matching CSS
447
+ // side effect to apply here.
448
+ if (node.attrs?.borderRadius)
449
+ this.setAttribute(element, "data-smart-radius", String(node.attrs.borderRadius), node.id);
450
+ else
451
+ this.removeAttribute(element, "data-smart-radius", node.id);
452
+ [
453
+ ["licenseDescription", "data-smart-license-description"],
454
+ ["licenseSourceUrl", "data-smart-license-source-url"],
455
+ ["licenseType", "data-smart-license-type"],
456
+ ["licenseVersion", "data-smart-license-version"],
457
+ ["licenseAttribution", "data-smart-license-attribution"],
458
+ ].forEach(([key, attrName]) => {
459
+ if (typeof node.attrs?.[key] === "string" && node.attrs[key])
460
+ this.setAttribute(element, attrName, String(node.attrs[key]), node.id);
461
+ else
462
+ this.removeAttribute(element, attrName, node.id);
463
+ });
408
464
  }
409
465
  else if (node.type === "formula" || node.type === "block_formula") {
410
466
  const source = String(node.attrs?.source || "");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "smartrte-core",
3
- "version": "1.0.0-beta.2",
3
+ "version": "1.0.0-beta.4",
4
4
  "description": "Framework-agnostic document model and editing primitives for Smart RTE",
5
5
  "repository": {
6
6
  "type": "git",