@37signals/lexxy 0.9.31-beta → 0.9.32

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/lexxy.esm.js CHANGED
@@ -12,7 +12,7 @@ import { registerPlainText } from '@lexical/plain-text';
12
12
  import { RichTextExtension, $isQuoteNode, QuoteNode, $isHeadingNode, $createHeadingNode, $createQuoteNode, HeadingNode, registerRichText } from '@lexical/rich-text';
13
13
  import { $generateNodesFromDOM, $generateHtmlFromNodes } from '@lexical/html';
14
14
  import { HistoryExtension } from '@lexical/history';
15
- import { $isCodeNode, CodeHighlightNode, PrismTokenizer, $isCodeHighlightNode, $createCodeNode, $createCodeHighlightNode, CodeNode, registerCodeHighlighting, normalizeCodeLang, CODE_LANGUAGE_FRIENDLY_NAME_MAP } from '@lexical/code';
15
+ import { $isCodeNode, CodeHighlightNode, $isCodeHighlightNode, $createCodeNode, $createCodeHighlightNode, CodeNode, registerCodeHighlighting, PrismTokenizer, normalizeCodeLang, CODE_LANGUAGE_FRIENDLY_NAME_MAP } from '@lexical/code';
16
16
  import { TRANSFORMERS, registerMarkdownShortcuts } from '@lexical/markdown';
17
17
  import { INSERT_TABLE_COMMAND, $getTableCellNodeFromLexicalNode, TableCellNode, TableNode, TableRowNode, setScrollableTablesActive, registerTablePlugin, registerTableSelectionObserver, TableCellHeaderStates, $insertTableRowAtSelection, $insertTableColumnAtSelection, $deleteTableRowAtSelection, $deleteTableColumnAtSelection, $findTableNode, $getTableRowIndexFromTableCellNode, $getTableColumnIndexFromTableCellNode, $findCellNode, $getElementForTableNode } from '@lexical/table';
18
18
  import { Marked } from 'marked';
@@ -129,7 +129,8 @@ var Lexxy = {
129
129
  // Until 3.3.2 that never came up: attributes admitted by a *functional* ADD_ATTR
130
130
  // skipped URI validation entirely (GHSA-cjmm-f4jc-qw8r), which is how data:
131
131
  // URLs worked here — and, less happily, how `url="javascript:…"` survived too.
132
- // The fix restored validation for both.
132
+ // The fix restored validation for both. The bypass was reported to us by
133
+ // @petitpois via HackerOne.
133
134
  //
134
135
  // So `url` is marked URI-safe, which hands the decision to this hook. The hook
135
136
  // only ever removes an attribute — it never force-keeps one — so scoping stays
@@ -152,9 +153,41 @@ var Lexxy = {
152
153
  // ordinary prose like "Q4: results" is not a URI any regex here would accept.
153
154
  const URI_BEARING_ATTACHMENT_ATTRIBUTES = [ "url" ];
154
155
 
155
- // DOMPurify's own IS_ALLOWED_URI, reproduced rather than narrowed, so `url` on a
156
- // non-attachment tag is treated exactly as DOMPurify would have treated it.
157
- const ALLOWED_URI = /^(?:(?:(?:f|ht)tps?|mailto|tel|callto|sms|cid|xmpp|matrix):|[^a-z]|[a-z+.-]+(?:[^a-z+.:-]|$))/i;
156
+ // DOMPurify's own IS_ALLOWED_URI scheme list, reproduced rather than narrowed, so
157
+ // `url` on a non-attachment tag is treated exactly as DOMPurify would have treated
158
+ // it. Kept as a source string so allowedUriRegexp() can widen it with an editor's
159
+ // declared schemes without re-deriving the rest of the pattern.
160
+ const BASE_URI_SCHEMES = "(?:f|ht)tps?|mailto|tel|callto|sms|cid|xmpp|matrix";
161
+
162
+ // A scheme name (not a full pattern), so a caller can't inject regexp syntax.
163
+ const SCHEME_NAME = /^[a-z][a-z0-9+.-]*$/;
164
+
165
+ // The executable schemes, by DOMPurify's own IS_SCRIPT_OR_DATA definition
166
+ // (/^(?:\w+script|data):/i): javascript, vbscript, …script, and data. Widening the
167
+ // allowlist to admit one would let it survive on href/object[data], which the
168
+ // override otherwise defeats — DOMPurify does not re-block a scheme its
169
+ // ALLOWED_URI_REGEXP accepts. A caller declaring one is dropped, so it stays
170
+ // refused. This is a closed set (the schemes a browser executes), not a growing
171
+ // denylist.
172
+ const EXECUTABLE_SCHEME = /^(?:\w+script|data)$/i;
173
+
174
+ // Builds DOMPurify's default IS_ALLOWED_URI, optionally with extra schemes folded
175
+ // into the scheme alternation. Widening the recognised-safe scheme set is how a
176
+ // custom-scheme identifier (a mention's `gid://…`) passes validation without
177
+ // exempting any attribute from it. Passing no schemes reproduces DOMPurify's
178
+ // default exactly.
179
+ function allowedUriRegexp(extraSchemes = []) {
180
+ const extra = extraSchemes
181
+ .map(scheme => String(scheme).toLowerCase())
182
+ .filter(scheme => SCHEME_NAME.test(scheme) && !EXECUTABLE_SCHEME.test(scheme))
183
+ .map(scheme => scheme.replace(/[.+-]/g, "\\$&"));
184
+
185
+ const schemes = [ BASE_URI_SCHEMES, ...extra ].join("|");
186
+
187
+ return new RegExp(`^(?:(?:${schemes}):|[^a-z]|[a-z+.-]+(?:[^a-z+.:-]|$))`, "i")
188
+ }
189
+
190
+ const ALLOWED_URI = allowedUriRegexp();
158
191
 
159
192
  // eslint-disable-next-line no-control-regex -- mirrors DOMPurify's own ATTR_WHITESPACE
160
193
  const ATTR_WHITESPACE = /[\u0000-\u0020\u00A0\u1680\u180E\u2000-\u2029\u205F\u3000]/g;
@@ -351,7 +384,17 @@ function buildConfig(allowedElements = null) {
351
384
  // default tag and attribute policy stands. `ALLOWED_TAGS: []` would not be a
352
385
  // default, it would be a refusal: it strips every tag. An editor that declares
353
386
  // an empty allowlist still gets that refusal, because it asked for it.
354
- if (allowedElements) Object.assign(config, allowlistFor(allowedElements));
387
+ //
388
+ // uriSafeSchemes widens the scheme validation rather than exempting an attribute
389
+ // from it: ALLOWED_URI_REGEXP is set only when an editor declares custom schemes,
390
+ // so a value like a mention's `gid://…` passes while javascript:/data: stay
391
+ // refused on every attribute — including href and object[data]. Nothing is taken
392
+ // out of URI checking, so there is no attribute to guard against exempting.
393
+ if (allowedElements) {
394
+ const { uriSafeSchemes, ...tagPolicy } = allowlistFor(allowedElements);
395
+ if (uriSafeSchemes.length) config.ALLOWED_URI_REGEXP = allowedUriRegexp(uriSafeSchemes);
396
+ Object.assign(config, tagPolicy);
397
+ }
355
398
 
356
399
  // Always assigned, including when we have no policy — `null` is what
357
400
  // trustedTypesPolicy() returns then, and `TRUSTED_TYPES_POLICY: null` is
@@ -374,6 +417,7 @@ function allowlistFor(allowedElements) {
374
417
  // Object.prototype key: `tagAttributes["constructor"]` would answer with a
375
418
  // function, and ADD_ATTR would call .includes on it.
376
419
  const tagAttributes = Object.create(null);
420
+ const uriSafeSchemes = [];
377
421
 
378
422
  // Lowercased, because DOMPurify lowercases ALLOWED_TAGS and calls ADD_ATTR
379
423
  // with the lowercased tag and attribute names. Keeping the caller's casing
@@ -386,6 +430,15 @@ function allowlistFor(allowedElements) {
386
430
 
387
431
  tagAttributes[tag] ||= [];
388
432
  tagAttributes[tag].push(...attributes);
433
+
434
+ // A custom scheme a caller declares is folded into the editor's URI-scheme
435
+ // allowlist (ALLOWED_URI_REGEXP in buildConfig), so a value like a mention's
436
+ // `gid="gid://…"` — which DOMPurify otherwise drops as an unknown scheme, even
437
+ // once the `gid` name is allowed — passes validation. It widens the recognised
438
+ // schemes, it does not exempt an attribute: javascript:/data: stay refused
439
+ // everywhere, so there is no navigational attribute to guard. Declared per
440
+ // element for locality, but a scheme is editor-wide once allowed.
441
+ uriSafeSchemes.push(...(element.uriSafeSchemes ?? []));
389
442
  }
390
443
 
391
444
  // Only for tags the caller already permits — this widens what an allowed
@@ -397,7 +450,8 @@ function allowlistFor(allowedElements) {
397
450
  return {
398
451
  ALLOWED_TAGS: Object.keys(tagAttributes),
399
452
  ALLOWED_ATTR: ALLOWED_HTML_ATTRIBUTES,
400
- ADD_ATTR: (attribute, tag) => tagAttributes[tag]?.includes(attribute)
453
+ ADD_ATTR: (attribute, tag) => tagAttributes[tag]?.includes(attribute),
454
+ uriSafeSchemes
401
455
  }
402
456
  }
403
457
 
@@ -1854,12 +1908,11 @@ class CustomActionTextAttachmentNode extends DecoratorNode {
1854
1908
  return Lexxy.global.get("attachmentTagName")
1855
1909
  }
1856
1910
 
1857
- constructor({ tagName, sgid, contentType, innerHtml, plainText } = {}, key) {
1911
+ constructor({ sgid, contentType, innerHtml, plainText } = {}, key) {
1858
1912
  super(key);
1859
1913
 
1860
1914
  const contentTypeNamespace = Lexxy.global.get("attachmentContentTypeNamespace");
1861
1915
 
1862
- this.tagName = tagName || CustomActionTextAttachmentNode.TAG_NAME;
1863
1916
  this.sgid = sgid;
1864
1917
  this.contentType = contentType || `application/vnd.${contentTypeNamespace}.unknown`;
1865
1918
  this.innerHtml = innerHtml;
@@ -1867,7 +1920,7 @@ class CustomActionTextAttachmentNode extends DecoratorNode {
1867
1920
  }
1868
1921
 
1869
1922
  createDOM(_config, editor) {
1870
- const figure = createElement(this.tagName, { "content-type": this.contentType, "data-lexxy-decorator": true, draggable: true });
1923
+ const figure = createElement(CustomActionTextAttachmentNode.TAG_NAME, { "content-type": this.contentType, "data-lexxy-decorator": true, draggable: true });
1871
1924
  figure.dataset.lexicalNodeKey = this.__key;
1872
1925
 
1873
1926
  // Resolved from the editor so this content is sanitized with its own
@@ -1910,7 +1963,7 @@ class CustomActionTextAttachmentNode extends DecoratorNode {
1910
1963
  }
1911
1964
 
1912
1965
  exportDOM() {
1913
- const attachment = createElement(this.tagName, {
1966
+ const attachment = createElement(CustomActionTextAttachmentNode.TAG_NAME, {
1914
1967
  sgid: this.sgid,
1915
1968
  content: this.innerHtml,
1916
1969
  "content-type": this.contentType
@@ -1923,7 +1976,6 @@ class CustomActionTextAttachmentNode extends DecoratorNode {
1923
1976
  return {
1924
1977
  type: "custom_action_text_attachment",
1925
1978
  version: 1,
1926
- tagName: this.tagName,
1927
1979
  sgid: this.sgid,
1928
1980
  contentType: this.contentType,
1929
1981
  innerHtml: this.innerHtml,
@@ -2670,10 +2722,9 @@ class ActionTextAttachmentNode extends DecoratorNode {
2670
2722
  return Lexxy.global.get("attachmentTagName")
2671
2723
  }
2672
2724
 
2673
- constructor({ tagName, sgid, src, previewSrc, previewable, previewStatusUrl, pendingPreview, altText, caption, contentType, fileName, fileSize, width, height, uploadError } = {}, key) {
2725
+ constructor({ sgid, src, previewSrc, previewable, previewStatusUrl, pendingPreview, altText, caption, contentType, fileName, fileSize, width, height, uploadError } = {}, key) {
2674
2726
  super(key);
2675
2727
 
2676
- this.tagName = tagName || ActionTextAttachmentNode.TAG_NAME;
2677
2728
  this.sgid = sgid;
2678
2729
  this.src = src;
2679
2730
  this.previewSrc = previewSrc;
@@ -2732,7 +2783,7 @@ class ActionTextAttachmentNode extends DecoratorNode {
2732
2783
  }
2733
2784
 
2734
2785
  exportDOM() {
2735
- const attachment = createElement(this.tagName, {
2786
+ const attachment = createElement(ActionTextAttachmentNode.TAG_NAME, {
2736
2787
  sgid: this.sgid,
2737
2788
  previewable: this.previewable || null,
2738
2789
  url: this.src,
@@ -2753,7 +2804,6 @@ class ActionTextAttachmentNode extends DecoratorNode {
2753
2804
  return {
2754
2805
  type: "action_text_attachment",
2755
2806
  version: 1,
2756
- tagName: this.tagName,
2757
2807
  sgid: this.sgid,
2758
2808
  src: this.src,
2759
2809
  previewable: this.previewable,
@@ -3473,6 +3523,34 @@ class UploadRequests {
3473
3523
  }
3474
3524
  }
3475
3525
 
3526
+ // Partition a stretch of text into consecutive { start, end, range } segments,
3527
+ // where offsets are relative to the text and range is the covering source
3528
+ // range — expressed in outer coordinates as { start, end, ... } — or null for
3529
+ // the stretches no range covers. Ranges must be sorted and non-overlapping.
3530
+ function segmentTextByRanges(text, textStart, ranges) {
3531
+ const segments = [];
3532
+ let cursor = 0;
3533
+
3534
+ for (const range of ranges) {
3535
+ const from = Math.max(range.start - textStart, cursor);
3536
+ const to = Math.min(range.end - textStart, text.length);
3537
+
3538
+ if (from < to) {
3539
+ if (from > cursor) {
3540
+ segments.push({ start: cursor, end: from, range: null });
3541
+ }
3542
+ segments.push({ start: from, end: to, range });
3543
+ cursor = to;
3544
+ }
3545
+ }
3546
+
3547
+ if (cursor < text.length) {
3548
+ segments.push({ start: cursor, end: text.length, range: null });
3549
+ }
3550
+
3551
+ return segments
3552
+ }
3553
+
3476
3554
  // Shared, strictly-contained element used to attach ephemeral nodes when we
3477
3555
  // need to read computed styles (e.g. canonicalizing style values, resolving
3478
3556
  // CSS custom properties). The container is created once and attached to
@@ -3776,26 +3854,9 @@ function extractHighlightStyleFromElement(element) {
3776
3854
  return css.length > 0 ? css : null
3777
3855
  }
3778
3856
 
3779
- // The code retokenizer replaces a code block's children with freshly created
3780
- // tokens that carry no styles, which would drop color highlights on every
3781
- // edit. This tokenizer wraps the stock Prism tokenizer to restore them: it
3782
- // recovers the block's highlight ranges — staged during HTML import, or read
3783
- // from the children the fresh tokens are about to replace — and reapplies
3784
- // them to the fresh tokens before the retokenizer splices them in.
3785
- function buildHighlightPreservingTokenizer(editor) {
3786
- return {
3787
- defaultLanguage: PrismTokenizer.defaultLanguage,
3788
- tokenize(code, language) {
3789
- return PrismTokenizer.tokenize(code, language)
3790
- },
3791
- $tokenize(codeNode, language) {
3792
- const tokens = PrismTokenizer.$tokenize(codeNode, language);
3793
- const highlights = $takeHighlightRanges(editor, codeNode);
3794
- return $applyHighlightRangesToTokens(tokens, highlights)
3795
- }
3796
- }
3797
- }
3798
-
3857
+ // Recover the highlight ranges to reapply after a retokenization: the ranges
3858
+ // staged during HTML import, or the ones read from the children the fresh
3859
+ // tokens are about to replace.
3799
3860
  function $takeHighlightRanges(editor, codeNode) {
3800
3861
  const pending = $getPendingHighlights(editor);
3801
3862
  const key = codeNode.getKey();
@@ -3837,52 +3898,25 @@ function $applyHighlightRangesToTokens(tokens, highlights) {
3837
3898
  // attached to the tree yet, so we create CodeHighlightNode replacements.
3838
3899
  function $splitTokenAtHighlightBoundaries(token, tokenStart, highlights) {
3839
3900
  const text = token.getTextContent();
3840
- const segments = segmentTextByHighlights(text, tokenStart, highlights);
3901
+ const segments = segmentTextByRanges(text, tokenStart, highlights);
3841
3902
 
3842
3903
  if (segments.length === 1) {
3843
3904
  const [ segment ] = segments;
3844
- if (segment.style) {
3845
- $applyHighlightStyleToToken(token, segment.style);
3905
+ if (segment.range) {
3906
+ $applyHighlightStyleToToken(token, segment.range.style);
3846
3907
  }
3847
3908
  return [ token ]
3848
3909
  } else {
3849
3910
  return segments.map((segment) => {
3850
3911
  const segmentToken = $createCodeHighlightNode(text.slice(segment.start, segment.end), token.getHighlightType());
3851
- if (segment.style) {
3852
- $applyHighlightStyleToToken(segmentToken, segment.style);
3912
+ if (segment.range) {
3913
+ $applyHighlightStyleToToken(segmentToken, segment.range.style);
3853
3914
  }
3854
3915
  return segmentToken
3855
3916
  })
3856
3917
  }
3857
3918
  }
3858
3919
 
3859
- // Partition a token's text into consecutive { start, end, style } segments,
3860
- // where offsets are relative to the token and style is null for the stretches
3861
- // no highlight covers.
3862
- function segmentTextByHighlights(text, tokenStart, highlights) {
3863
- const segments = [];
3864
- let cursor = 0;
3865
-
3866
- for (const { start, end, style } of highlights) {
3867
- const from = Math.max(start - tokenStart, cursor);
3868
- const to = Math.min(end - tokenStart, text.length);
3869
-
3870
- if (from < to) {
3871
- if (from > cursor) {
3872
- segments.push({ start: cursor, end: from, style: null });
3873
- }
3874
- segments.push({ start: from, end: to, style });
3875
- cursor = to;
3876
- }
3877
- }
3878
-
3879
- if (cursor < text.length) {
3880
- segments.push({ start: cursor, end: text.length, style: null });
3881
- }
3882
-
3883
- return segments
3884
- }
3885
-
3886
3920
  function $applyHighlightStyleToToken(token, style) {
3887
3921
  token.setStyle(style);
3888
3922
  $setCodeHighlightFormat(token, true);
@@ -3892,17 +3926,24 @@ function $buildChildRanges(codeNode) {
3892
3926
  const childRanges = [];
3893
3927
  let charOffset = 0;
3894
3928
 
3895
- for (const child of codeNode.getChildren()) {
3896
- if ($isCodeHighlightNode(child) || $isTextNode(child)) {
3897
- const text = child.getTextContent();
3898
- childRanges.push({ node: child, start: charOffset, end: charOffset + text.length });
3899
- charOffset += text.length;
3900
- } else {
3901
- // LineBreakNode, TabNode - count as 1 character each (\n, \t)
3902
- charOffset += 1;
3929
+ function walk(node) {
3930
+ for (const child of node.getChildren()) {
3931
+ if ($isElementNode(child)) {
3932
+ // e.g. a LinkNode awaiting its first retokenization
3933
+ walk(child);
3934
+ } else if ($isCodeHighlightNode(child) || $isTextNode(child)) {
3935
+ const text = child.getTextContent();
3936
+ childRanges.push({ node: child, start: charOffset, end: charOffset + text.length });
3937
+ charOffset += text.length;
3938
+ } else {
3939
+ // LineBreakNode - counts as 1 character (\n)
3940
+ charOffset += 1;
3941
+ }
3903
3942
  }
3904
3943
  }
3905
3944
 
3945
+ walk(codeNode);
3946
+
3906
3947
  return childRanges
3907
3948
  }
3908
3949
 
@@ -5022,10 +5063,17 @@ class Selection {
5022
5063
  || this.#selectInLexical(this.topLevelNodeAfterCursor)
5023
5064
  }
5024
5065
 
5066
+ // hasNodeSelection reads the committed state, while $getSelection() resolves
5067
+ // against the pending one — so the selection here can be missing, no longer a
5068
+ // NodeSelection, or reference a node that no longer exists (empty editor,
5069
+ // just-deleted node, decorator boundary). Only invoke the callback when a
5070
+ // node actually resolves; returning false lets the callers fall back —
5071
+ // #selectInLexical for plain arrows, native pass-through for shift-arrows.
5025
5072
  #withCurrentNodeSelectionNode(fn) {
5026
- if (this.hasNodeSelection) {
5027
- return fn($getSelection().getNodes()[0])
5028
- }
5073
+ const selection = this.hasNodeSelection ? $getSelection() : null;
5074
+ const currentNode = $isNodeSelection(selection) ? selection.getNodes()[0] : undefined;
5075
+
5076
+ return currentNode ? fn(currentNode) : false
5029
5077
  }
5030
5078
 
5031
5079
  #rangeSelectDecorator(node, direction = "forward") {
@@ -6212,15 +6260,18 @@ class ListItemNodeInserter extends BaseNodeInserter {
6212
6260
  // very corruption we are avoiding. The block must land at the list's level.
6213
6261
  const anchorNode = this.selection.anchor.getNode();
6214
6262
  const outerList = this.#outermostList(anchorNode);
6215
- const topItem = this.#topLevelItemFor(anchorNode, outerList);
6216
6263
 
6217
- // A blank top-level bullet is just the insertion point (e.g. the user pressed
6218
- // Enter to leave the list); break out of it entirely. A bullet with content —
6219
- // including one wrapping a nested list — splits so its content stays in the list.
6220
- const splitAfterItem = $isBlankNode(topItem) ? topItem.getPreviousSibling() : topItem;
6221
- const splitIndex = splitAfterItem ? splitAfterItem.getIndexWithinParent() + 1 : 0;
6222
- const [ listBefore, listAfter ] = $splitNode(outerList, splitIndex);
6223
- if ($isBlankNode(topItem)) { topItem.remove(); }
6264
+ // removeText() can relocate the anchor before we read it back — clean out
6265
+ // of the list when the removed range spanned it. With no list left to
6266
+ // split, default insertion applies.
6267
+ if (!outerList) {
6268
+ this.selection.insertNodes(nodes);
6269
+ return
6270
+ }
6271
+
6272
+ const topItem = this.#topLevelItemFor(anchorNode, outerList);
6273
+ const [ listBefore, listAfter ] = $splitNode(outerList, this.#splitIndexFor(topItem));
6274
+ if (topItem && $isBlankNode(topItem)) { topItem.remove(); }
6224
6275
 
6225
6276
  let anchor = listBefore ?? listAfter;
6226
6277
  for (const node of nodes) {
@@ -6233,6 +6284,18 @@ class ListItemNodeInserter extends BaseNodeInserter {
6233
6284
  nodes.at(-1).selectNext();
6234
6285
  }
6235
6286
 
6287
+ // A blank top-level bullet is just the insertion point (e.g. the user pressed
6288
+ // Enter to leave the list); break out of it entirely. A bullet with content —
6289
+ // including one wrapping a nested list — splits so its content stays in the
6290
+ // list. With no top-level bullet at all — removeText() relocated the anchor
6291
+ // onto the list node itself — the anchor offset is the boundary between items.
6292
+ #splitIndexFor(topItem) {
6293
+ if (!topItem) return this.selection.anchor.offset
6294
+
6295
+ const splitAfterItem = $isBlankNode(topItem) ? topItem.getPreviousSibling() : topItem;
6296
+ return splitAfterItem ? splitAfterItem.getIndexWithinParent() + 1 : 0
6297
+ }
6298
+
6236
6299
  #outermostList(node) {
6237
6300
  return [ node, ...node.getParents() ].reverse().find($isListNode)
6238
6301
  }
@@ -7332,11 +7395,93 @@ class Contents {
7332
7395
  }
7333
7396
  }
7334
7397
 
7398
+ function parsePastedMarkdown(text) {
7399
+ return pasteMarked.parse(text)
7400
+ }
7401
+
7402
+ // A marked instance scoped to the paste flow. A fresh Marked starts from
7403
+ // marked's stock defaults — gfm stays on and tables still parse — and is
7404
+ // isolated from the global `marked` singleton, so a host bundle sharing the
7405
+ // deduped module can't leak `marked.use(...)` customizations into pasting.
7406
+ // The additions are breaks: true, matching the render pass Clipboard always
7407
+ // used, the `html` renderer, and the dropped `code` tokenizer.
7408
+ //
7335
7409
  // Markdown reads any line indented by a tab or four spaces as a code block. Pasted
7336
7410
  // plain text is full of incidental indentation — outlines, notes, logs — and nobody
7337
7411
  // indents a note meaning "this is code"; they fence it. Keep fences, drop the
7338
7412
  // indentation rule.
7339
- const pastedMarkdown = new Marked({ breaks: true }).use({ tokenizer: { code: () => undefined } });
7413
+ //
7414
+ // Following CommonMark, marked tokenizes a bare "<tag>" in prose as raw inline
7415
+ // (or block) HTML. That is intentional for recognized tags — pasted plain text
7416
+ // carrying <span style>, <mark>, <b>… is styled on import — but for an
7417
+ // *unrecognized* element the Lexical importer has no converter and silently
7418
+ // unwraps it, dropping the tag. WEBVTT speaker cues such as
7419
+ // "<v Nabila Abdel Nabi>", where the name lives in what the HTML parser reads
7420
+ // as attributes, thus vanish entirely.
7421
+ //
7422
+ // Classifying at marked's html renderer rides its public token grammar instead
7423
+ // of shadowing it: marked routes code (`code`/`codespan`), autolinks and
7424
+ // [text](<dest>) links to *other* token types that never reach this renderer,
7425
+ // so their literal "<...>" content is preserved for free — no pre-escaping,
7426
+ // code-region collection, or lexer-option syncing required. This renderer fires
7427
+ // only for genuine raw-HTML tokens, and marked inserts its return value verbatim
7428
+ // (it does not re-encode), so we escape unsupported tags to literal text here.
7429
+ const pasteMarked = new Marked({
7430
+ breaks: true,
7431
+ renderer: { html: renderPastedHtmlToken }
7432
+ }).use({ tokenizer: { code: () => undefined } });
7433
+
7434
+ // Matches a single HTML tag lexeme: an optional "/", a tag name, then attribute
7435
+ // characters up to the closing ">". Quoted attribute values may contain ">" —
7436
+ // the HTML tokenizer consumes them as part of the value, and marked's tag
7437
+ // grammar matches them into the same raw token — so the attribute run matches
7438
+ // quoted strings whole. Otherwise a lexeme like <v title="a > b"> would end at
7439
+ // the inner ">", escaping only the prefix and leaving the tail for DOMParser
7440
+ // to decode.
7441
+ const HTML_TAG_LEXEME = /<(\/?)([a-zA-Z][a-zA-Z0-9-]*)((?:"[^"]*"|'[^']*'|[^'">])*)>/g;
7442
+
7443
+ // Escape unknown tags *within* the raw-HTML token, keeping known tags intact.
7444
+ // Lexeme-level (not whole-token) escaping matters because marked can emit a
7445
+ // single coarse block-HTML token whose first tag is known but which nests an
7446
+ // unknown one — e.g. "<div><v Name> Hello</div>" arrives as one token. We keep
7447
+ // the <div> so it still renders, and escape the nested <v Name> so it survives
7448
+ // as literal text rather than being unwrapped and dropped by the importer.
7449
+ function renderPastedHtmlToken(token) {
7450
+ return token.text.replace(HTML_TAG_LEXEME, (lexeme, _slash, name) => {
7451
+ return isSupportedHtmlElement(name) ? lexeme : escapeHtml(lexeme)
7452
+ })
7453
+ }
7454
+
7455
+ // The renderer's output is final, so escape the full lexeme — "&", "<" and ">" —
7456
+ // for an exact round-trip (e.g. "<v A&nbsp;B>" → "&lt;v A&amp;nbsp;B&gt;").
7457
+ function escapeHtml(text) {
7458
+ return text.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;")
7459
+ }
7460
+
7461
+ // "Supported" — safe to pass through for the importer to handle — is decided from
7462
+ // the element taxonomy, not the importer's converter registry. The importer both
7463
+ // converts tags it has a converter for (span→mark, b→bold, tr/td→table cells) and
7464
+ // transparently traverses standard structural wrappers it has *no* converter for
7465
+ // (thead, tbody, tfoot, colgroup, caption), keeping their children. A converter
7466
+ // registry (editor._htmlConversions) lists only the former, so classifying by it
7467
+ // escaped those wrappers to literal text and corrupted pasted tables — foster
7468
+ // parenting hoisted "&lt;thead&gt;…" out as a stray paragraph. The taxonomy covers
7469
+ // both: a standard HTML element is either converted or losslessly unwrapped.
7470
+ //
7471
+ // Two exclusions escape the tag to literal text instead:
7472
+ // - Invented tags (WEBVTT's "<v Nabila Abdel Nabi>", "<foo>"): document
7473
+ // .createElement returns HTMLUnknownElement. The importer would unwrap them,
7474
+ // dropping the tag — and for <v Name> the name lives in what parses as
7475
+ // attributes, so it vanishes entirely.
7476
+ // - Custom elements (turbo-frame, bc-attachment, action-text-attachment): a
7477
+ // hyphenated name is a custom element (HTML spec; no standard element has a
7478
+ // hyphen). They're excluded even when the importer registers a converter, so
7479
+ // a plain-text paste can never materialize an attachment/widget. Legitimate
7480
+ // attachments arrive through the rich-HTML paste path, not here.
7481
+ function isSupportedHtmlElement(name) {
7482
+ const tag = name.toLowerCase();
7483
+ return !(document.createElement(tag) instanceof HTMLUnknownElement) && !tag.includes("-")
7484
+ }
7340
7485
 
7341
7486
  class Clipboard {
7342
7487
  #listeners = new ListenerBin()
@@ -7524,7 +7669,7 @@ class Clipboard {
7524
7669
  }
7525
7670
 
7526
7671
  #pasteMarkdown(text) {
7527
- const html = pastedMarkdown.parse(text);
7672
+ const html = parsePastedMarkdown(text);
7528
7673
  const doc = parseHtml(html);
7529
7674
 
7530
7675
  if (this.#isPlainTextWithoutMarkdown(doc)) {
@@ -7723,6 +7868,134 @@ class BrowserAdapter {
7723
7868
  }
7724
7869
  }
7725
7870
 
7871
+ // The code retokenizer only knows CodeHighlightNode, LineBreakNode and
7872
+ // TabNode, so a LinkNode inside a code block — Trix-authored documents allow
7873
+ // links in code — is destroyed on the first retokenization. Links survive as
7874
+ // state on the tokens instead: this state records the link attributes on each
7875
+ // CodeHighlightNode the link covers, the extraction below reads it (or a
7876
+ // still-untokenized LinkNode) back into ranges before every retokenization,
7877
+ // and the tokenizer reapplies it to the fresh tokens.
7878
+ const codeLinkState = createState("codeLink", {
7879
+ parse: (value) => (value && typeof value.url === "string" ? value : null)
7880
+ });
7881
+
7882
+ function $getCodeLink(node) {
7883
+ return $getState(node, codeLinkState)
7884
+ }
7885
+
7886
+ function $setCodeLink(node, link) {
7887
+ $setState(node, codeLinkState, link);
7888
+ }
7889
+
7890
+ // Build the list of { start, end, link } ranges covering every link in a code
7891
+ // block, with offsets into the block's text content. Links appear either as
7892
+ // LinkNode descendants (imported or just created, before their first
7893
+ // retokenization) or as link state on tokens (after it). The walk recurses
7894
+ // through nested elements because `<pre><code>` imports of multi-line content
7895
+ // briefly produce a CodeNode nested inside another before the outer one's
7896
+ // retokenization flattens them.
7897
+ function $extractLinkRangesFromCodeNode(codeNode) {
7898
+ const ranges = [];
7899
+ let offset = 0;
7900
+
7901
+ function walk(node) {
7902
+ for (const child of node.getChildren()) {
7903
+ if ($isLinkNode(child)) {
7904
+ const size = child.getTextContent().length;
7905
+ appendRange(ranges, { start: offset, end: offset + size, link: linkAttributesFrom(child) });
7906
+ offset += size;
7907
+ } else if ($isElementNode(child)) {
7908
+ walk(child);
7909
+ } else {
7910
+ const size = child.getTextContent().length;
7911
+ if ($isTextNode(child)) {
7912
+ const link = $getCodeLink(child);
7913
+ if (link) {
7914
+ appendRange(ranges, { start: offset, end: offset + size, link });
7915
+ }
7916
+ }
7917
+ offset += size;
7918
+ }
7919
+ }
7920
+ }
7921
+
7922
+ walk(codeNode);
7923
+
7924
+ return ranges
7925
+ }
7926
+
7927
+ function $applyLinkRangesToTokens(tokens, ranges) {
7928
+ if (ranges.length === 0) return tokens
7929
+
7930
+ const linkedTokens = [];
7931
+ let offset = 0;
7932
+
7933
+ for (const token of tokens) {
7934
+ if ($isCodeHighlightNode(token)) {
7935
+ linkedTokens.push(...$splitTokenAtLinkBoundaries(token, offset, ranges));
7936
+ } else {
7937
+ linkedTokens.push(token);
7938
+ }
7939
+ offset += token.getTextContentSize();
7940
+ }
7941
+
7942
+ return linkedTokens
7943
+ }
7944
+
7945
+ function $splitTokenAtLinkBoundaries(token, tokenStart, ranges) {
7946
+ const text = token.getTextContent();
7947
+ const segments = segmentTextByRanges(text, tokenStart, ranges);
7948
+
7949
+ if (segments.length === 1) {
7950
+ const [ segment ] = segments;
7951
+ if (segment.range) {
7952
+ $setCodeLink(token, segment.range.link);
7953
+ }
7954
+ return [ token ]
7955
+ } else {
7956
+ return segments.map((segment) => {
7957
+ const segmentToken = $cloneTokenSlice(token, text.slice(segment.start, segment.end));
7958
+ if (segment.range) {
7959
+ $setCodeLink(segmentToken, segment.range.link);
7960
+ }
7961
+ return segmentToken
7962
+ })
7963
+ }
7964
+ }
7965
+
7966
+ function $cloneTokenSlice(token, text) {
7967
+ const segmentToken = $createCodeHighlightNode(text, token.getHighlightType());
7968
+ segmentToken.setStyle(token.getStyle());
7969
+ // CodeHighlightNode.setFormat is a no-op, so carry the format over directly
7970
+ segmentToken.getWritable().__format = token.getFormat();
7971
+ return segmentToken
7972
+ }
7973
+
7974
+ // Consecutive children carrying the same link merge into a single range, so a
7975
+ // link split across tokens by an earlier retokenization keeps covering fresh
7976
+ // tokens as one contiguous stretch however they retokenize.
7977
+ function appendRange(ranges, range) {
7978
+ const previous = ranges[ranges.length - 1];
7979
+
7980
+ if (previous && previous.end === range.start && sameLink(previous.link, range.link)) {
7981
+ previous.end = range.end;
7982
+ } else {
7983
+ ranges.push(range);
7984
+ }
7985
+ }
7986
+
7987
+ function sameLink(a, b) {
7988
+ return a.url === b.url && a.target === b.target && a.rel === b.rel && a.title === b.title
7989
+ }
7990
+
7991
+ function linkAttributesFrom(linkNode) {
7992
+ const link = { url: linkNode.getURL() };
7993
+ if (linkNode.getTarget()) link.target = linkNode.getTarget();
7994
+ if (linkNode.getRel()) link.rel = linkNode.getRel();
7995
+ if (linkNode.getTitle()) link.title = linkNode.getTitle();
7996
+ return link
7997
+ }
7998
+
7726
7999
  // Custom TextNode exportDOM that avoids redundant wrapping.
7727
8000
  //
7728
8001
  // Lexical's built-in TextNode.exportDOM() calls createDOM() which produces semantic tags
@@ -7737,6 +8010,30 @@ class BrowserAdapter {
7737
8010
  // Any <span> elements produced by createDOM() are unwrapped, since they only carry
7738
8011
  // editor classes that aren't meaningful in exported HTML.
7739
8012
 
8013
+ // A CodeHighlightNode exports like any other text node, plus an <a> wrapper
8014
+ // when it carries link state — that's how a link survives inside a code
8015
+ // block, whose retokenizer would destroy an actual LinkNode child.
8016
+ function exportCodeHighlightNodeDOM(editor, codeHighlightNode) {
8017
+ const output = exportTextNodeDOM(editor, codeHighlightNode);
8018
+ const link = $getCodeLink(codeHighlightNode);
8019
+
8020
+ if (link) {
8021
+ return { element: wrapWithAnchor(output.element, link) }
8022
+ } else {
8023
+ return output
8024
+ }
8025
+ }
8026
+
8027
+ function wrapWithAnchor(element, link) {
8028
+ const anchor = document.createElement("a");
8029
+ anchor.setAttribute("href", link.url);
8030
+ if (link.target) anchor.setAttribute("target", link.target);
8031
+ if (link.rel) anchor.setAttribute("rel", link.rel);
8032
+ if (link.title) anchor.setAttribute("title", link.title);
8033
+ anchor.appendChild(element);
8034
+ return anchor
8035
+ }
8036
+
7740
8037
  function exportTextNodeDOM(editor, textNode) {
7741
8038
  const element = textNode.createDOM(editor._config, editor);
7742
8039
  element.style.whiteSpace = "pre-wrap";
@@ -7968,12 +8265,34 @@ class CodeHighlightingExtension extends LexxyExtension {
7968
8265
  return defineExtension({
7969
8266
  name: "lexxy/code-highlighting",
7970
8267
  register(editor) {
7971
- return registerCodeHighlighting(editor, buildHighlightPreservingTokenizer(editor))
8268
+ return registerCodeHighlighting(editor, buildMarkupPreservingTokenizer(editor))
7972
8269
  }
7973
8270
  })
7974
8271
  }
7975
8272
  }
7976
8273
 
8274
+ // The code retokenizer replaces a code block's children with freshly created
8275
+ // tokens that carry no styles or links, which would drop color highlights and
8276
+ // hyperlinks on every edit. This tokenizer wraps the stock Prism tokenizer to
8277
+ // restore both: it recovers the block's highlight and link ranges — staged
8278
+ // during HTML import, or read from the children the fresh tokens are about to
8279
+ // replace — and reapplies them to the fresh tokens before the retokenizer
8280
+ // splices them in.
8281
+ function buildMarkupPreservingTokenizer(editor) {
8282
+ return {
8283
+ defaultLanguage: PrismTokenizer.defaultLanguage,
8284
+ tokenize(code, language) {
8285
+ return PrismTokenizer.tokenize(code, language)
8286
+ },
8287
+ $tokenize(codeNode, language) {
8288
+ const linkRanges = $extractLinkRangesFromCodeNode(codeNode);
8289
+ const highlightRanges = $takeHighlightRanges(editor, codeNode);
8290
+ const tokens = PrismTokenizer.$tokenize(codeNode, language);
8291
+ return $applyLinkRangesToTokens($applyHighlightRangesToTokens(tokens, highlightRanges), linkRanges)
8292
+ }
8293
+ }
8294
+ }
8295
+
7977
8296
  const TRIX_LANGUAGE_ATTR = "language";
7978
8297
 
7979
8298
  class TrixContentExtension extends LexxyExtension {
@@ -9760,7 +10079,7 @@ class LexicalEditorElement extends HTMLElement {
9760
10079
  theme: theme,
9761
10080
  nodes: this.#lexicalNodes,
9762
10081
  html: {
9763
- export: new Map([ [ TextNode, exportTextNodeDOM ], [ CodeHighlightNode, exportTextNodeDOM ] ])
10082
+ export: new Map([ [ TextNode, exportTextNodeDOM ], [ CodeHighlightNode, exportCodeHighlightNodeDOM ] ])
9764
10083
  },
9765
10084
  $initialEditorState: (editor) => {
9766
10085
  this.#configureSanitizer(editor);
@@ -44,12 +44,13 @@ function highlightElement(preElement) {
44
44
  const grammar = Prism.languages?.[language];
45
45
  if (!grammar) return
46
46
 
47
- // Read the source text and <mark> ranges in a single walk, before Prism
48
- // rewrites the element. Sharing one traversal keeps the highlight offsets
49
- // aligned with the code string and preserves leading whitespace — deriving
50
- // either of them separately (e.g. textContent through DOMParser) collapses
51
- // leading whitespace and shifts every range, re-indenting the rendered block.
52
- const { code, highlights } = extractCodeAndHighlights(preElement);
47
+ // Read the source text and the <mark> and <a> ranges in a single walk,
48
+ // before Prism rewrites the element. Sharing one traversal keeps the range
49
+ // offsets aligned with the code string and preserves leading whitespace —
50
+ // deriving either of them separately (e.g. textContent through DOMParser)
51
+ // collapses leading whitespace and shifts every range, re-indenting the
52
+ // rendered block.
53
+ const { code, highlights, links } = extractCodeAndMarkup(preElement);
53
54
 
54
55
  const highlightedHtml = Prism.highlight(code, grammar, language);
55
56
  preElement.innerHTML = highlightedHtml;
@@ -58,18 +59,24 @@ function highlightElement(preElement) {
58
59
  applyHighlightRanges(preElement, highlights);
59
60
  }
60
61
 
62
+ if (links.length > 0) {
63
+ applyLinkRanges(preElement, links);
64
+ }
65
+
61
66
  preElement.dataset.highlighted = "true";
62
67
  }
63
68
 
64
- // Walk the <pre> once, building Prism's source text and the <mark> ranges
65
- // together: a text node contributes its text verbatim, a <br> contributes a
66
- // newline, and a <mark> records the slice of code it covers. Because both
67
- // outputs come from the same walk, every range offset is just a position in
68
- // `code` — so the highlights can't drift out of sync with the source, and the
69
- // block's leading whitespace survives (HTML parsing would collapse it).
70
- function extractCodeAndHighlights(preElement) {
69
+ // Walk the <pre> once, building Prism's source text and the <mark> and <a>
70
+ // ranges together: a text node contributes its text verbatim, a <br>
71
+ // contributes a newline, and a <mark> or <a> records the slice of code it
72
+ // covers. Because all outputs come from the same walk, every range offset is
73
+ // just a position in `code` — so the markup can't drift out of sync with the
74
+ // source, and the block's leading whitespace survives (HTML parsing would
75
+ // collapse it).
76
+ function extractCodeAndMarkup(preElement) {
71
77
  const root = preElement.querySelector("code") || preElement;
72
78
  const highlights = [];
79
+ const links = [];
73
80
  let code = "";
74
81
 
75
82
  function walk(node) {
@@ -87,6 +94,12 @@ function extractCodeAndHighlights(preElement) {
87
94
  if (style) {
88
95
  highlights.push({ start, end: code.length, style });
89
96
  }
97
+ } else if (node.tagName === "A" && node.getAttribute("href")) {
98
+ const start = code.length;
99
+ for (const child of node.childNodes) {
100
+ walk(child);
101
+ }
102
+ links.push({ start, end: code.length, attributes: linkAttributes(node) });
90
103
  } else {
91
104
  for (const child of node.childNodes) {
92
105
  walk(child);
@@ -99,7 +112,17 @@ function extractCodeAndHighlights(preElement) {
99
112
  walk(child);
100
113
  }
101
114
 
102
- return { code, highlights }
115
+ return { code, highlights, links }
116
+ }
117
+
118
+ function linkAttributes(element) {
119
+ const attributes = { href: element.getAttribute("href") };
120
+ for (const name of [ "target", "rel", "title" ]) {
121
+ if (element.getAttribute(name)) {
122
+ attributes[name] = element.getAttribute(name);
123
+ }
124
+ }
125
+ return attributes
103
126
  }
104
127
 
105
128
  function extractStyle(element) {
@@ -109,16 +132,32 @@ function extractStyle(element) {
109
132
  return parts.length > 0 ? parts.join(" ") : null
110
133
  }
111
134
 
112
- // Wrap character ranges in <mark> elements within a Prism-highlighted DOM tree.
113
- // Each range is applied independently, re-collecting text nodes each time to
114
- // account for splits from previous ranges.
135
+ // Wrap character ranges in <mark> or <a> elements within a Prism-highlighted
136
+ // DOM tree. Each range is applied independently, re-collecting text nodes
137
+ // each time to account for splits from previous ranges.
115
138
  function applyHighlightRanges(element, highlights) {
116
139
  for (const { start, end, style } of highlights) {
117
- wrapRange(element, start, end, style);
140
+ wrapRange(element, start, end, () => {
141
+ const mark = document.createElement("mark");
142
+ mark.setAttribute("style", style);
143
+ return mark
144
+ });
145
+ }
146
+ }
147
+
148
+ function applyLinkRanges(element, links) {
149
+ for (const { start, end, attributes } of links) {
150
+ wrapRange(element, start, end, () => {
151
+ const anchor = document.createElement("a");
152
+ for (const [ name, value ] of Object.entries(attributes)) {
153
+ anchor.setAttribute(name, value);
154
+ }
155
+ return anchor
156
+ });
118
157
  }
119
158
  }
120
159
 
121
- function wrapRange(container, rangeStart, rangeEnd, style) {
160
+ function wrapRange(container, rangeStart, rangeEnd, createWrapper) {
122
161
  const textNodes = collectTextNodes(container);
123
162
 
124
163
  // Process in reverse so DOM mutations don't shift earlier text node offsets
@@ -133,14 +172,13 @@ function wrapRange(container, rangeStart, rangeEnd, style) {
133
172
  const text = node.textContent;
134
173
  const parent = node.parentNode;
135
174
 
136
- const mark = document.createElement("mark");
137
- mark.setAttribute("style", style);
138
- mark.textContent = text.slice(relStart, relEnd);
175
+ const wrapper = createWrapper();
176
+ wrapper.textContent = text.slice(relStart, relEnd);
139
177
 
140
178
  if (relEnd < text.length) {
141
179
  parent.insertBefore(document.createTextNode(text.slice(relEnd)), node.nextSibling);
142
180
  }
143
- parent.insertBefore(mark, node.nextSibling);
181
+ parent.insertBefore(wrapper, node.nextSibling);
144
182
 
145
183
  if (relStart > 0) {
146
184
  node.textContent = text.slice(0, relStart);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@37signals/lexxy",
3
- "version": "0.9.31-beta",
3
+ "version": "0.9.32",
4
4
  "description": "Lexxy - A modern rich text editor for Rails.",
5
5
  "module": "dist/lexxy.esm.js",
6
6
  "type": "module",