@stll/folio-core 0.47.6 → 0.48.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. package/dist/ai-edits/word-diff.d.ts +25 -6
  2. package/dist/ai-edits/word-diff.js +210 -38
  3. package/dist/controller/fontReadiness.js +3 -3
  4. package/dist/controller/hyphenationReadiness.d.ts +19 -0
  5. package/dist/controller/hyphenationReadiness.js +47 -0
  6. package/dist/controller/layoutPipeline.d.ts +3 -0
  7. package/dist/controller/layoutPipeline.js +59 -17
  8. package/dist/display-list/build/buildDisplayList.js +39 -12
  9. package/dist/display-list/build/pageFurniture.d.ts +7 -1
  10. package/dist/display-list/build/pageFurniture.js +21 -8
  11. package/dist/display-list/build/paragraphPrimitives.js +5 -4
  12. package/dist/display-list/build/tablePrimitives.js +10 -6
  13. package/dist/docx/attributeRemainder.js +10 -2
  14. package/dist/docx/blockContentParser.d.ts +2 -0
  15. package/dist/docx/blockContentParser.js +77 -62
  16. package/dist/docx/commentParser.d.ts +2 -1
  17. package/dist/docx/commentParser.js +57 -45
  18. package/dist/docx/containerChildren.d.ts +21 -8
  19. package/dist/docx/containerChildren.js +39 -9
  20. package/dist/docx/documentParser.d.ts +5 -1
  21. package/dist/docx/documentParser.js +8 -5
  22. package/dist/docx/documentSectionFacts.d.ts +22 -0
  23. package/dist/docx/documentSectionFacts.js +43 -0
  24. package/dist/docx/fieldParser.d.ts +8 -1
  25. package/dist/docx/fieldParser.js +47 -3
  26. package/dist/docx/fontTableParser.js +34 -27
  27. package/dist/docx/footnoteParser.d.ts +3 -2
  28. package/dist/docx/footnoteParser.js +15 -12
  29. package/dist/docx/groupDrawingParser.d.ts +2 -1
  30. package/dist/docx/groupDrawingParser.js +3 -6
  31. package/dist/docx/headerFooterParser.d.ts +7 -2
  32. package/dist/docx/headerFooterParser.js +15 -5
  33. package/dist/docx/hyperlinkParser.d.ts +48 -36
  34. package/dist/docx/hyperlinkParser.js +97 -82
  35. package/dist/docx/numberingParser.js +108 -84
  36. package/dist/docx/paragraphParser.d.ts +5 -1
  37. package/dist/docx/paragraphParser.js +381 -362
  38. package/dist/docx/paragraphProperties.js +123 -109
  39. package/dist/docx/paragraphPropertySource.d.ts +6 -1
  40. package/dist/docx/paragraphPropertySource.js +9 -1
  41. package/dist/docx/paragraphTextBoxEnrichment.d.ts +2 -1
  42. package/dist/docx/paragraphTextBoxEnrichment.js +17 -6
  43. package/dist/docx/parser.d.ts +7 -1
  44. package/dist/docx/parser.js +33 -16
  45. package/dist/docx/previewBudget.d.ts +55 -33
  46. package/dist/docx/previewBudget.js +59 -65
  47. package/dist/docx/relsParser.js +9 -1
  48. package/dist/docx/rezip.js +41 -58
  49. package/dist/docx/runParser.d.ts +4 -1
  50. package/dist/docx/runParser.js +172 -184
  51. package/dist/docx/sdtProperties.js +87 -84
  52. package/dist/docx/sectionParser.js +169 -160
  53. package/dist/docx/serializer/paragraphSerializer.js +42 -38
  54. package/dist/docx/serializer/settingsSerializer.js +19 -1
  55. package/dist/docx/settingsParser.js +8 -0
  56. package/dist/docx/streamingXmlParser.d.ts +6 -2
  57. package/dist/docx/streamingXmlParser.js +12 -7
  58. package/dist/docx/styleParser.js +34 -8
  59. package/dist/docx/tableParser.d.ts +21 -12
  60. package/dist/docx/tableParser.js +518 -441
  61. package/dist/docx/textBoxParser.d.ts +8 -3
  62. package/dist/docx/textBoxParser.js +3 -3
  63. package/dist/docx/verbatimCapture.d.ts +10 -8
  64. package/dist/docx/verbatimCapture.js +45 -2
  65. package/dist/docx/vmlImageParser.d.ts +15 -13
  66. package/dist/docx/vmlImageParser.js +75 -60
  67. package/dist/docx/vmlPreview.d.ts +3 -3
  68. package/dist/docx/vmlPreview.js +3 -4
  69. package/dist/docx/xmlNamespaceContext.d.ts +9 -0
  70. package/dist/docx/xmlNamespaceContext.js +54 -0
  71. package/dist/docx/xmlParser.d.ts +6 -3
  72. package/dist/docx/xmlParser.js +11 -41
  73. package/dist/docx/xmlResourceLimits.d.ts +2 -1
  74. package/dist/docx/xmlResourceLimits.js +4 -1
  75. package/dist/fonts/headlessMeasure.js +0 -0
  76. package/dist/headless-layout.d.ts +11 -2
  77. package/dist/headless-layout.js +153 -107
  78. package/dist/internal/paragraphFormattingSerialization.js +3 -3
  79. package/dist/layout-bridge/convert/endnoteLayout.d.ts +65 -0
  80. package/dist/layout-bridge/convert/endnoteLayout.js +227 -0
  81. package/dist/layout-bridge/convert/footnoteLayout.d.ts +19 -3
  82. package/dist/layout-bridge/convert/footnoteLayout.js +40 -31
  83. package/dist/layout-bridge/convert/headerFooterLayout.d.ts +1 -7
  84. package/dist/layout-bridge/convert/headerFooterLayout.js +5 -20
  85. package/dist/layout-bridge/convert/paragraphMarkFormatting.d.ts +10 -0
  86. package/dist/layout-bridge/convert/paragraphMarkFormatting.js +59 -0
  87. package/dist/layout-bridge/convert/toFlowBlocks.d.ts +29 -1
  88. package/dist/layout-bridge/convert/toFlowBlocks.js +258 -35
  89. package/dist/layout-bridge/engine/hitTest.js +5 -3
  90. package/dist/layout-bridge/engine/selectionRects.js +5 -3
  91. package/dist/layout-engine/anchorLayoutInCellCompatibility.d.ts +10 -0
  92. package/dist/layout-engine/anchorLayoutInCellCompatibility.js +22 -0
  93. package/dist/layout-engine/index.d.ts +2 -2
  94. package/dist/layout-engine/index.js +70 -11
  95. package/dist/layout-engine/justificationCompatibility.d.ts +1 -1
  96. package/dist/layout-engine/justificationCompatibility.js +9 -5
  97. package/dist/layout-engine/keep-together.d.ts +11 -5
  98. package/dist/layout-engine/keep-together.js +39 -16
  99. package/dist/layout-engine/layoutInstrumentation.d.ts +8 -2
  100. package/dist/layout-engine/layoutInstrumentation.js +8 -1
  101. package/dist/layout-engine/measure/advanceComposition.js +4 -7
  102. package/dist/layout-engine/measure/hyphenationDictionaries.d.ts +74 -0
  103. package/dist/layout-engine/measure/hyphenationDictionaries.js +138 -0
  104. package/dist/layout-engine/measure/hyphenationPreload.d.ts +11 -0
  105. package/dist/layout-engine/measure/hyphenationPreload.js +24 -0
  106. package/dist/layout-engine/measure/lineBreakProvider.d.ts +12 -2
  107. package/dist/layout-engine/measure/lineBreakProvider.js +30 -30
  108. package/dist/layout-engine/measure/lineBreaks.d.ts +6 -1
  109. package/dist/layout-engine/measure/lineBreaks.js +10 -3
  110. package/dist/layout-engine/measure/listMarkerWidth.d.ts +9 -6
  111. package/dist/layout-engine/measure/listMarkerWidth.js +20 -22
  112. package/dist/layout-engine/measure/measureBlocks.js +42 -8
  113. package/dist/layout-engine/measure/measureContainer.js +5 -5
  114. package/dist/layout-engine/measure/measureHelpers.d.ts +15 -3
  115. package/dist/layout-engine/measure/measureHelpers.js +21 -5
  116. package/dist/layout-engine/measure/measureParagraph.js +109 -30
  117. package/dist/layout-engine/measure/measureTypes.d.ts +5 -0
  118. package/dist/layout-engine/measure/tabCalculator.d.ts +2 -0
  119. package/dist/layout-engine/measure/tabCalculator.js +2 -1
  120. package/dist/layout-engine/measure/tableCellFloating.d.ts +2 -0
  121. package/dist/layout-engine/measure/tableCellFloating.js +51 -19
  122. package/dist/layout-engine/measure/tableCellGrid.d.ts +28 -3
  123. package/dist/layout-engine/measure/tableCellGrid.js +45 -5
  124. package/dist/layout-engine/measure/tableFragmentBorderGeometry.js +2 -1
  125. package/dist/layout-engine/noteAreaFlow.d.ts +19 -0
  126. package/dist/layout-engine/noteAreaFlow.js +83 -0
  127. package/dist/layout-engine/paginator.d.ts +10 -1
  128. package/dist/layout-engine/paginator.js +13 -1
  129. package/dist/layout-engine/tableRowBreak.js +3 -1
  130. package/dist/layout-engine/types.d.ts +89 -3
  131. package/dist/layout-engine/types.js +96 -1
  132. package/dist/layout-painter/index.d.ts +3 -0
  133. package/dist/layout-painter/renderPage.js +32 -83
  134. package/dist/layout-painter/renderParagraph.d.ts +6 -1
  135. package/dist/layout-painter/renderParagraph.js +34 -22
  136. package/dist/layout-painter/renderTable.js +26 -13
  137. package/dist/prosemirror/attrs/index.d.ts +9 -9
  138. package/dist/prosemirror/attrs/index.js +43 -11
  139. package/dist/prosemirror/conversion/fromProseDoc.js +8 -1
  140. package/dist/prosemirror/conversion/markInterner.d.ts +21 -0
  141. package/dist/prosemirror/conversion/markInterner.js +90 -0
  142. package/dist/prosemirror/conversion/toProseDoc.js +77 -41
  143. package/dist/prosemirror/extensions/core/DocExtension.js +2 -1
  144. package/dist/prosemirror/extensions/core/ParagraphExtension.js +1 -0
  145. package/dist/prosemirror/extensions/marks/markUtils.d.ts +6 -2
  146. package/dist/prosemirror/extensions/marks/markUtils.js +39 -51
  147. package/dist/prosemirror/extensions/nodes/TextBoxExtension.js +2 -1
  148. package/dist/prosemirror/listMarker.d.ts +1 -0
  149. package/dist/prosemirror/listMarker.js +2 -0
  150. package/dist/prosemirror/listRenderingAttrs.d.ts +1 -0
  151. package/dist/prosemirror/listRenderingAttrs.js +3 -0
  152. package/dist/prosemirror/schema/nodes.d.ts +13 -1
  153. package/dist/prosemirror/textBoxHostParagraph.d.ts +13 -0
  154. package/dist/prosemirror/textBoxHostParagraph.js +21 -0
  155. package/dist/prosemirror/validation.js +91 -80
  156. package/dist/utils/createDocument.js +10 -1
  157. package/dist/utils/scriptSegments.d.ts +33 -4
  158. package/dist/utils/scriptSegments.js +39 -6
  159. package/dist/utils/textFormattingMerge.d.ts +17 -3
  160. package/dist/utils/textFormattingMerge.js +18 -12
  161. package/dist/utils/trailingText.d.ts +22 -0
  162. package/dist/utils/trailingText.js +32 -0
  163. package/package.json +2 -2
@@ -2,8 +2,9 @@
2
2
  /**
3
3
  * Diff between two strings, as the segments a redline is drawn from.
4
4
  *
5
- * Tokenises (by default on whitespace boundaries, preserving the whitespace as
6
- * part of each token), runs an LCS, and returns a left-to-right ordered list of
5
+ * Tokenises (by default into words and the punctuation marks at their edges,
6
+ * preserving the whitespace as part of each token; see {@link pushRunTokens}),
7
+ * runs an LCS, and returns a left-to-right ordered list of
7
8
  * segments where shared runs render as `equal`, removed runs as `del`, and
8
9
  * added runs as `ins`. Used by the panel (to render minimal-change redlines),
9
10
  * the version comparison, and the apply engine (so tracked changes mark only
@@ -18,10 +19,25 @@
18
19
  * same thing and can actually be read. Three rules pull the output back:
19
20
  *
20
21
  * 1. A match made only of separators is not a match ({@link isSeparatorOnly}).
21
- * 2. A match too short to carry meaning is dropped unless it opens the string,
22
- * where it is the reader's anchor rather than an island.
22
+ * 2. A match too short to carry meaning is dropped when changes sit on both
23
+ * sides of it and at least one of them changes words. At either end of the
24
+ * string it is the reader's anchor, and
25
+ * between two punctuation or whitespace edits it is what they were made
26
+ * around; neither is an island.
23
27
  * 3. When what survives is still too fragmented for its length, the whole
24
- * paragraph is one replacement ({@link isTooFragmented}).
28
+ * paragraph is one replacement ({@link isTooFragmented}). Punctuation and
29
+ * whitespace edits alone never are.
30
+ *
31
+ * At word granularity the LCS itself prefers words: one matched word outweighs
32
+ * any number of matched marks, and a unique mark is never an anchor.
33
+ *
34
+ * A word token carries the whitespace before it, so a string's first word has
35
+ * none while the same word inside the other string does. Word tokens are
36
+ * therefore matched on their text alone; the rules above judge that
37
+ * alignment, and a whitespace difference around a matched word is marked
38
+ * afterwards as a change of the whitespace only
39
+ * ({@link separateWhitespaceChanges}). Otherwise prefixing "(1) " would strike
40
+ * through the unchanged first word.
25
41
  *
26
42
  * Common affixes and unique-token anchors split the input into independent
27
43
  * gaps before any quadratic work. The residual gaps share one
@@ -61,6 +77,9 @@ type WordDiffOptions = {
61
77
  /** Default: nothing normalized, so both strings reconstruct exactly. */
62
78
  normalization?: WordDiffNormalization;
63
79
  };
80
+ /** Keep token storage bounded when a run has an unusually long punctuation edge. */
81
+ declare const MAX_EDGE_PUNCTUATION_TOKENS = 64;
82
+ declare const tokenizeWords: (value: string) => string[];
64
83
  /**
65
84
  * One internal comparison/apply scope. Every diff shares the same quadratic
66
85
  * allowance; public standalone calls below still receive a fresh allowance.
@@ -75,4 +94,4 @@ declare const createScopedWordDiffOptions: <Options extends WordDiffOptions>(opt
75
94
  declare const wordDiffSessionFromOptions: (options: WordDiffOptions | undefined) => ReturnType<typeof createWordDiffSession>;
76
95
  declare const diffWordSegments: (before: string, after: string, options?: WordDiffOptions) => WordDiffSegment[];
77
96
  //#endregion
78
- export { WORD_DIFF_GRANULARITIES, WordDiffGranularity, WordDiffNormalization, WordDiffOptions, WordDiffSegment, createScopedWordDiffOptions, createWordDiffSession, diffWordSegments, wordDiffSessionFromOptions };
97
+ export { MAX_EDGE_PUNCTUATION_TOKENS, WORD_DIFF_GRANULARITIES, WordDiffGranularity, WordDiffNormalization, WordDiffOptions, WordDiffSegment, createScopedWordDiffOptions, createWordDiffSession, diffWordSegments, tokenizeWords, wordDiffSessionFromOptions };
@@ -7,10 +7,11 @@
7
7
  */
8
8
  const WORD_DIFF_GRANULARITIES = Object.freeze(["word", "character"]);
9
9
  const WHITESPACE = /\s/u;
10
- /** Punctuation, symbols and whitespace: everything that is not content. */
11
- const SEPARATOR_ONLY = /^[\s\p{P}\p{S}]*$/u;
10
+ /** Punctuation and whitespace carry no content except legal section/paragraph signs. */
11
+ const SEPARATOR_ONLY = /^[\s\p{P}]*$/u;
12
+ const CONTENT_PUNCTUATION = /[§¶]/u;
12
13
  /**
13
- * A match of at most this many tokens that carries no letters or digits is
14
+ * A match of at most this many tokens that carries only punctuation or space is
14
15
  * noise: a lone space, a comma, a stray closing bracket. Matching it splits
15
16
  * two rewrites into four.
16
17
  */
@@ -33,6 +34,59 @@ const MINIMUM_ISOLATED_MATCH_UNITS = 2;
33
34
  const FRAGMENTATION_SCALE = 32;
34
35
  /** Below this length every match is a large fraction of the string, so the floor says nothing. */
35
36
  const FRAGMENTATION_MINIMUM_AVERAGE_LENGTH = 8;
37
+ /** Unicode general category P: punctuation marks, dashes, brackets and quotes. */
38
+ const PUNCTUATION = /^\p{P}$/u;
39
+ /** Keep token storage bounded when a run has an unusually long punctuation edge. */
40
+ const MAX_EDGE_PUNCTUATION_TOKENS = 64;
41
+ /** The code point starting at `index`, as a string of one or two UTF-16 units. */
42
+ const codePointAt = (value, index) => String.fromCodePoint(value.codePointAt(index) ?? 0);
43
+ /** The code point ending just before `end`. */
44
+ const codePointBefore = (value, end) => {
45
+ const low = value.charCodeAt(end - 1);
46
+ const high = value.charCodeAt(end - 2);
47
+ return low >= 56320 && low <= 57343 && high >= 55296 && high <= 56319 ? value.slice(end - 2, end) : value.slice(end - 1, end);
48
+ };
49
+ /**
50
+ * Split one whitespace-free run into its words and punctuation marks.
51
+ *
52
+ * Punctuation marks (general category P) at either edge of the run are
53
+ * tokens of their own, as Word's compare treats them: `jmění.` becoming `jmění,`
54
+ * changes the mark, not the word. Punctuation inside the run stays in the
55
+ * word, so `d.o.o`, `1.1.2026`, `3.5`, `well-known` and `don't` are single
56
+ * tokens; only their edge marks split off (`d.o.o.` is `d.o.o` + `.`, `b)` is
57
+ * `b` + `)`). A run made only of punctuation is split up to the edge cap.
58
+ * `§` and `¶` are content markers despite their Unicode punctuation category;
59
+ * symbols (general category S, such as `€` or `+`) are not punctuation.
60
+ */
61
+ const pushRunTokens = (tokens, value, run, prefixStart) => {
62
+ let wordStart = run.start;
63
+ let pending = value.slice(prefixStart, run.start);
64
+ let leadingMarks = 0;
65
+ while (wordStart < run.end && leadingMarks < 64) {
66
+ const mark = codePointAt(value, wordStart);
67
+ if (!PUNCTUATION.test(mark) || CONTENT_PUNCTUATION.test(mark)) break;
68
+ tokens.push(pending + mark);
69
+ pending = "";
70
+ wordStart += mark.length;
71
+ leadingMarks++;
72
+ }
73
+ let wordEnd = run.end;
74
+ const trailing = [];
75
+ while (wordEnd > wordStart && trailing.length < 64) {
76
+ const mark = codePointBefore(value, wordEnd);
77
+ if (!PUNCTUATION.test(mark) || CONTENT_PUNCTUATION.test(mark)) break;
78
+ trailing.push(mark);
79
+ wordEnd -= mark.length;
80
+ }
81
+ if (wordStart < wordEnd) {
82
+ tokens.push(pending + value.slice(wordStart, wordEnd));
83
+ pending = "";
84
+ }
85
+ for (let index = trailing.length - 1; index >= 0; index--) {
86
+ tokens.push(pending + (trailing[index] ?? ""));
87
+ pending = "";
88
+ }
89
+ };
36
90
  const tokenizeWords = (value) => {
37
91
  const tokens = [];
38
92
  let tokenStart = 0;
@@ -40,8 +94,12 @@ const tokenizeWords = (value) => {
40
94
  while (cursor < value.length) {
41
95
  while (cursor < value.length && WHITESPACE.test(value.charAt(cursor))) cursor++;
42
96
  if (cursor === value.length) break;
97
+ const runStart = cursor;
43
98
  while (cursor < value.length && !WHITESPACE.test(value.charAt(cursor))) cursor++;
44
- tokens.push(value.slice(tokenStart, cursor));
99
+ pushRunTokens(tokens, value, {
100
+ start: runStart,
101
+ end: cursor
102
+ }, tokenStart);
45
103
  tokenStart = cursor;
46
104
  }
47
105
  const last = tokens.at(-1);
@@ -61,7 +119,7 @@ const comparisonKey = (token, normalization) => {
61
119
  const collapsed = normalization.whitespace === true ? token.replaceAll(/\s+/gu, " ").trim() : token;
62
120
  return normalization.case === true ? collapsed.toLowerCase() : collapsed;
63
121
  };
64
- const isSeparatorOnly = (text) => SEPARATOR_ONLY.test(text);
122
+ const isSeparatorOnly = (text) => SEPARATOR_ONLY.test(text) && !CONTENT_PUNCTUATION.test(text);
65
123
  /**
66
124
  * Cell budget shared by the residual LCS gaps inside one comparison or apply
67
125
  * scope. `before`/`after` come from attacker-controlled document text (a
@@ -85,11 +143,21 @@ const MAX_WORD_DIFF_ANCHOR_TOKENS = 16384;
85
143
  * and case or whitespace normalization creates another string for it.
86
144
  */
87
145
  const MAX_WORD_DIFF_COMPARISON_KEY_CODE_UNITS = 1048576;
146
+ /**
147
+ * How the LCS scores a match. `uniform` counts every matched token once;
148
+ * `content-first` (word granularity) never trades a matched word for matched
149
+ * punctuation, so moving a mark across a word cannot strike the word through.
150
+ */
151
+ const MATCH_WEIGHTING = {
152
+ Uniform: "uniform",
153
+ ContentFirst: "content-first"
154
+ };
88
155
  const ALIGNMENT_OPERATION = {
89
156
  Equal: 1,
90
157
  Delete: 2,
91
158
  Insert: 3
92
159
  };
160
+ const hasOppositeChange = (runs, direction) => runs.some(({ type }) => type === (direction === "insertion" ? "del" : "ins"));
93
161
  const fitsCellBudget = (beforeLength, afterLength, budget) => beforeLength === 0 || afterLength === 0 || beforeLength <= Math.floor(budget.remainingCells / afterLength);
94
162
  const pushRun = (runs, run) => {
95
163
  const last = runs.at(-1);
@@ -128,13 +196,7 @@ const pushChangedRange = (runs, before, after, beforeRange, afterRange) => {
128
196
  text: joinTokenRange(after, afterRange)
129
197
  });
130
198
  };
131
- /**
132
- * Unique tokens common to both ranges, reduced to a monotone subsequence of
133
- * target positions. Repeated boilerplate is deliberately ineligible: a unique
134
- * clause number or name is stronger lineage evidence than another occurrence
135
- * of "the" chosen by an arbitrary LCS tie.
136
- */
137
- const findPatienceAnchors = (beforeKeys, afterKeys, beforeRange, afterRange) => {
199
+ const findPatienceAnchors = ({ beforeKeys, afterKeys, beforeRange, afterRange, weighting }) => {
138
200
  const beforeOccurrences = /* @__PURE__ */ new Map();
139
201
  for (let index = beforeRange.start; index < beforeRange.end; index++) {
140
202
  const key = beforeKeys[index] ?? "";
@@ -155,7 +217,7 @@ const findPatienceAnchors = (beforeKeys, afterKeys, beforeRange, afterRange) =>
155
217
  const key = beforeKeys[beforeIndex] ?? "";
156
218
  const beforeCount = beforeOccurrences.get(key);
157
219
  const afterOccurrence = afterOccurrences.get(key);
158
- if (beforeCount === 1 && afterOccurrence?.count === 1) candidates.push({
220
+ if ((weighting === MATCH_WEIGHTING.Uniform || !isSeparatorOnly(key)) && beforeCount === 1 && afterOccurrence?.count === 1) candidates.push({
159
221
  beforeIndex,
160
222
  afterIndex: afterOccurrence.index
161
223
  });
@@ -275,7 +337,7 @@ const alignMonotoneChange = (beforeText, afterText, before, after, normalization
275
337
  return runs;
276
338
  };
277
339
  /** Longest-common-subsequence alignment for one residual, bounded gap. */
278
- const alignDenseGap = ({ before, after, beforeKeys, afterKeys, beforeRange, afterRange, runs, budget }) => {
340
+ const alignDenseGap = ({ before, after, beforeKeys, afterKeys, beforeRange, afterRange, runs, budget, weighting }) => {
279
341
  const beforeLength = beforeRange.end - beforeRange.start;
280
342
  const afterLength = afterRange.end - afterRange.start;
281
343
  if (beforeLength === 0 || afterLength === 0) {
@@ -289,15 +351,24 @@ const alignDenseGap = ({ before, after, beforeKeys, afterKeys, beforeRange, afte
289
351
  const cellCount = beforeLength * afterLength;
290
352
  budget.remainingCells -= cellCount;
291
353
  const lengths = new Uint32Array(cellCount);
292
- for (let beforeOffset = 0; beforeOffset < beforeLength; beforeOffset++) for (let afterOffset = 0; afterOffset < afterLength; afterOffset++) {
293
- const index = beforeOffset * afterLength + afterOffset;
294
- if (beforeKeys[beforeRange.start + beforeOffset] === afterKeys[afterRange.start + afterOffset]) {
295
- lengths[index] = (beforeOffset > 0 && afterOffset > 0 ? lengths[(beforeOffset - 1) * afterLength + afterOffset - 1] ?? 0 : 0) + 1;
296
- continue;
354
+ const contentWeight = weighting === MATCH_WEIGHTING.ContentFirst ? Math.min(beforeLength, afterLength) + 1 : 1;
355
+ const matchWeights = new Uint32Array(beforeLength);
356
+ for (let beforeOffset = 0; beforeOffset < beforeLength; beforeOffset++) {
357
+ const key = beforeKeys[beforeRange.start + beforeOffset] ?? "";
358
+ matchWeights[beforeOffset] = isSeparatorOnly(key) ? 1 : contentWeight;
359
+ }
360
+ for (let beforeOffset = 0; beforeOffset < beforeLength; beforeOffset++) {
361
+ const matchWeight = matchWeights[beforeOffset] ?? 1;
362
+ for (let afterOffset = 0; afterOffset < afterLength; afterOffset++) {
363
+ const index = beforeOffset * afterLength + afterOffset;
364
+ if (beforeKeys[beforeRange.start + beforeOffset] === afterKeys[afterRange.start + afterOffset]) {
365
+ lengths[index] = (beforeOffset > 0 && afterOffset > 0 ? lengths[(beforeOffset - 1) * afterLength + afterOffset - 1] ?? 0 : 0) + matchWeight;
366
+ continue;
367
+ }
368
+ const above = beforeOffset > 0 ? lengths[(beforeOffset - 1) * afterLength + afterOffset] ?? 0 : 0;
369
+ const left = afterOffset > 0 ? lengths[index - 1] ?? 0 : 0;
370
+ lengths[index] = Math.max(above, left);
297
371
  }
298
- const above = beforeOffset > 0 ? lengths[(beforeOffset - 1) * afterLength + afterOffset] ?? 0 : 0;
299
- const left = afterOffset > 0 ? lengths[index - 1] ?? 0 : 0;
300
- lengths[index] = Math.max(above, left);
301
372
  }
302
373
  const reversedOperations = new Uint8Array(beforeLength + afterLength);
303
374
  let operationCount = 0;
@@ -404,7 +475,7 @@ const alignGap = (options) => {
404
475
  * repeated boilerplate from winning a tie over a unique legal term, and makes
405
476
  * a small edit inside a long paragraph pay for only its changed gap.
406
477
  */
407
- const alignTokens = ({ beforeText, afterText, before, after, normalization, budget }) => {
478
+ const alignTokens = ({ beforeText, afterText, before, after, normalization, budget, weighting }) => {
408
479
  let commonPrefixLength = 0;
409
480
  let beforePrefixLength = 0;
410
481
  let afterPrefixLength = 0;
@@ -491,7 +562,13 @@ const alignTokens = ({ beforeText, afterText, before, after, normalization, budg
491
562
  start: 0,
492
563
  end: afterMiddle.length
493
564
  };
494
- const anchors = findPatienceAnchors(beforeKeys, afterKeys, middleRangeBefore, middleRangeAfter);
565
+ const anchors = findPatienceAnchors({
566
+ beforeKeys,
567
+ afterKeys,
568
+ beforeRange: middleRangeBefore,
569
+ afterRange: middleRangeAfter,
570
+ weighting
571
+ });
495
572
  if (anchors.length === 0 && fullInputFits && fullInputFitsStorage) {
496
573
  alignDenseGap({
497
574
  before,
@@ -507,7 +584,8 @@ const alignTokens = ({ beforeText, afterText, before, after, normalization, budg
507
584
  end: after.length
508
585
  },
509
586
  runs,
510
- budget
587
+ budget,
588
+ weighting
511
589
  });
512
590
  return {
513
591
  runs,
@@ -538,7 +616,8 @@ const alignTokens = ({ beforeText, afterText, before, after, normalization, budg
538
616
  end: anchor.afterIndex
539
617
  },
540
618
  runs,
541
- budget
619
+ budget,
620
+ weighting
542
621
  });
543
622
  pushRun(runs, {
544
623
  type: "equal",
@@ -563,7 +642,8 @@ const alignTokens = ({ beforeText, afterText, before, after, normalization, budg
563
642
  end: afterMiddle.length
564
643
  },
565
644
  runs,
566
- budget
645
+ budget,
646
+ weighting
567
647
  });
568
648
  pushEqualRange(runs, before, after, {
569
649
  start: beforeMiddleEnd,
@@ -587,6 +667,16 @@ const isTooFragmented = (runs, averageLength) => {
587
667
  for (const run of runs) if (run.type === "equal") sumOfSquares += run.before.length * run.before.length;
588
668
  return sumOfSquares * FRAGMENTATION_SCALE < averageLength * averageLength;
589
669
  };
670
+ /** True when a run changes words, not only punctuation or whitespace. */
671
+ const isContentChange = (run) => run.type !== "equal" && !isSeparatorOnly(run.text);
672
+ /** True when the change region starting at `from`, walking by `step`, changes words. */
673
+ const changeCarriesContent = (runs, { from, step }) => {
674
+ for (let index = from;; index += step) {
675
+ const run = runs[index];
676
+ if (run === void 0 || run.type === "equal") return false;
677
+ if (isContentChange(run)) return true;
678
+ }
679
+ };
590
680
  /**
591
681
  * Turn a match the quality rules rejected back into the change it interrupts.
592
682
  *
@@ -602,7 +692,13 @@ const demoteRejectedMatches = (runs) => {
602
692
  continue;
603
693
  }
604
694
  const interruptsAChange = runs[index - 1] !== void 0 || runs[index + 1] !== void 0;
605
- const isIsland = runs[index - 1] !== void 0 && runs[index + 1] !== void 0;
695
+ const isIsland = runs[index - 1] !== void 0 && runs[index + 1] !== void 0 && (changeCarriesContent(runs, {
696
+ from: index - 1,
697
+ step: -1
698
+ }) || changeCarriesContent(runs, {
699
+ from: index + 1,
700
+ step: 1
701
+ }));
606
702
  if (!(interruptsAChange && run.units <= MAX_SEPARATOR_ONLY_MATCH_UNITS && isSeparatorOnly(run.before) || isIsland && run.units < MINIMUM_ISOLATED_MATCH_UNITS)) {
607
703
  kept.push(run);
608
704
  continue;
@@ -617,6 +713,70 @@ const demoteRejectedMatches = (runs) => {
617
713
  }
618
714
  return kept;
619
715
  };
716
+ const TOKEN_WHITESPACE = /^(\s*)(.*?)(\s*)$/su;
717
+ const splitTokenWhitespace = (token) => {
718
+ const [, leading = "", text = "", trailing = ""] = TOKEN_WHITESPACE.exec(token) ?? [];
719
+ return {
720
+ leading,
721
+ text,
722
+ trailing
723
+ };
724
+ };
725
+ const pushChangedText = (runs, before, after) => {
726
+ if (before.length > 0) pushRun(runs, {
727
+ type: "del",
728
+ text: before
729
+ });
730
+ if (after.length > 0) pushRun(runs, {
731
+ type: "ins",
732
+ text: after
733
+ });
734
+ };
735
+ const pushWhitespace = (runs, before, after) => {
736
+ if (before !== after) {
737
+ pushChangedText(runs, before, after);
738
+ return;
739
+ }
740
+ if (before.length > 0) pushRun(runs, {
741
+ type: "equal",
742
+ before,
743
+ after,
744
+ units: 0
745
+ });
746
+ };
747
+ /**
748
+ * Mark the whitespace a matched word gained, lost or changed, leaving the word
749
+ * itself equal. Word tokens are aligned on their text alone, so an `equal` run
750
+ * may pair `"Závislá"` with `" Závislá"`; both sides of such a run tokenize
751
+ * to the same number of words. Every piece of both texts is emitted exactly
752
+ * once and in order, so both strings reconstruct whatever the pairing.
753
+ */
754
+ const separateWhitespaceChanges = (runs) => {
755
+ const separated = [];
756
+ for (const run of runs) {
757
+ if (run.type !== "equal" || run.before === run.after) {
758
+ pushRun(separated, run);
759
+ continue;
760
+ }
761
+ const beforeTokens = tokenizeWords(run.before);
762
+ const afterTokens = tokenizeWords(run.after);
763
+ const tokenCount = Math.max(beforeTokens.length, afterTokens.length);
764
+ for (let index = 0; index < tokenCount; index++) {
765
+ const before = splitTokenWhitespace(beforeTokens[index] ?? "");
766
+ const after = splitTokenWhitespace(afterTokens[index] ?? "");
767
+ pushWhitespace(separated, before.leading, after.leading);
768
+ if (before.text.length > 0 && after.text.length > 0) pushRun(separated, {
769
+ type: "equal",
770
+ before: before.text,
771
+ after: after.text,
772
+ units: 1
773
+ });
774
+ else pushChangedText(separated, before.text, after.text);
775
+ pushWhitespace(separated, before.trailing, after.trailing);
776
+ }
777
+ }
778
+ return separated;
779
+ };
620
780
  const toSegments = (runs) => {
621
781
  const segments = [];
622
782
  const push = (type, text) => {
@@ -694,7 +854,13 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
694
854
  text: before
695
855
  }];
696
856
  const granularity = options.granularity ?? "word";
697
- const normalization = options.normalization ?? {};
857
+ const requestedNormalization = options.normalization ?? {};
858
+ const normalization = granularity === "word" ? {
859
+ ...requestedNormalization,
860
+ whitespace: true
861
+ } : requestedNormalization;
862
+ const whitespaceIsSignificant = granularity === "word" && requestedNormalization.whitespace !== true;
863
+ const weighting = granularity === "word" ? MATCH_WEIGHTING.ContentFirst : MATCH_WEIGHTING.Uniform;
698
864
  const beforeTokens = tokenize(before, granularity);
699
865
  const afterTokens = tokenize(after, granularity);
700
866
  if (beforeTokens.length === 0 && afterTokens.length === 0) return [];
@@ -705,12 +871,12 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
705
871
  before: beforeTokens,
706
872
  after: afterTokens,
707
873
  normalization,
708
- budget
874
+ budget,
875
+ weighting
709
876
  });
710
877
  let surviving = demoteRejectedMatches(aligned.runs);
711
- const inventedOppositeChange = aligned.monotoneDirection === "insertion" && surviving.some(({ type }) => type === "del") || aligned.monotoneDirection === "deletion" && surviving.some(({ type }) => type === "ins");
712
878
  const monotoneDirection = aligned.monotoneDirection;
713
- if (inventedOppositeChange && monotoneDirection !== null) if (beforeTokens.length + afterTokens.length <= MAX_WORD_DIFF_ANCHOR_TOKENS && before.length + after.length <= MAX_WORD_DIFF_COMPARISON_KEY_CODE_UNITS && fitsCellBudget(beforeTokens.length, afterTokens.length, budget)) {
879
+ if (monotoneDirection !== null && hasOppositeChange(surviving, monotoneDirection)) if (beforeTokens.length + afterTokens.length <= MAX_WORD_DIFF_ANCHOR_TOKENS && before.length + after.length <= MAX_WORD_DIFF_COMPARISON_KEY_CODE_UNITS && fitsCellBudget(beforeTokens.length, afterTokens.length, budget)) {
714
880
  const unanchored = [];
715
881
  alignDenseGap({
716
882
  before: beforeTokens,
@@ -726,12 +892,18 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
726
892
  end: afterTokens.length
727
893
  },
728
894
  runs: unanchored,
729
- budget
895
+ budget,
896
+ weighting
730
897
  });
731
- surviving = unanchored;
732
- } else surviving = monotoneDirection === "insertion" && aligned.runs.some(({ type }) => type === "del") || monotoneDirection === "deletion" && aligned.runs.some(({ type }) => type === "ins") ? alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection) : aligned.runs;
733
- if (aligned.monotoneDirection === null && isTooFragmented(surviving, (before.length + after.length) / 2)) return wholeStringReplacement(before, after);
734
- return toSegments(orderDeletionsFirst(surviving));
898
+ surviving = hasOppositeChange(unanchored, monotoneDirection) ? alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection) : unanchored;
899
+ } else surviving = hasOppositeChange(aligned.runs, monotoneDirection) ? alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection) : aligned.runs;
900
+ if (aligned.monotoneDirection === null && surviving.some(isContentChange) && isTooFragmented(surviving, (before.length + after.length) / 2)) return wholeStringReplacement(before, after);
901
+ let marked = whitespaceIsSignificant ? separateWhitespaceChanges(surviving) : surviving;
902
+ if (monotoneDirection !== null && hasOppositeChange(marked, monotoneDirection)) {
903
+ const monotoneRuns = alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection);
904
+ marked = whitespaceIsSignificant ? separateWhitespaceChanges(monotoneRuns) : monotoneRuns;
905
+ }
906
+ return toSegments(orderDeletionsFirst(marked));
735
907
  };
736
908
  /**
737
909
  * One internal comparison/apply scope. Every diff shares the same quadratic
@@ -760,4 +932,4 @@ const wordDiffSessionFromOptions = (options) => {
760
932
  };
761
933
  const diffWordSegments = (before, after, options = {}) => diffWordSegmentsWithBudget(before, after, options, { remainingCells: MAX_WORD_DIFF_CELLS });
762
934
  //#endregion
763
- export { WORD_DIFF_GRANULARITIES, createScopedWordDiffOptions, createWordDiffSession, diffWordSegments, wordDiffSessionFromOptions };
935
+ export { MAX_EDGE_PUNCTUATION_TOKENS, WORD_DIFF_GRANULARITIES, createScopedWordDiffOptions, createWordDiffSession, diffWordSegments, tokenizeWords, wordDiffSessionFromOptions };
@@ -1,4 +1,5 @@
1
1
  import { buildFontAlternates, getFontAlternate } from "../fonts/fontAlternates.js";
2
+ import { DEFAULT_FONT_FAMILY } from "../layout-engine/measure/measureHelpers.js";
2
3
  import { expectFontFamilyMarkAttrs, expectParagraphAttrs } from "../prosemirror/attrs/index.js";
3
4
  import { parseFontFamilyList, resolveFontFamily } from "../utils/fontResolver.js";
4
5
  //#region src/controller/fontReadiness.ts
@@ -11,7 +12,6 @@ function documentFontsAreLoaded() {
11
12
  return !fontSet || fontSet.status === "loaded";
12
13
  }
13
14
  const INITIAL_LAYOUT_FONT_TIMEOUT_MS = 2e3;
14
- const DEFAULT_LAYOUT_FONT_FAMILY = "Calibri";
15
15
  const CSS_GENERIC_FONT_FAMILIES = /* @__PURE__ */ new Set([
16
16
  "serif",
17
17
  "sans-serif",
@@ -54,7 +54,7 @@ function collectInitialLayoutFontFamilies(documentModel, pmDoc) {
54
54
  function collectInitialLayoutFontFaces(documentModel, pmDoc) {
55
55
  const faces = /* @__PURE__ */ new Map();
56
56
  const fontAlternates = buildFontAlternates(documentModel?.package.fontTable);
57
- addLayoutFontFamilyFace(faces, DEFAULT_LAYOUT_FONT_FAMILY, REGULAR_LAYOUT_FONT_DESCRIPTOR, fontAlternates);
57
+ addLayoutFontFamilyFace(faces, DEFAULT_FONT_FAMILY, REGULAR_LAYOUT_FONT_DESCRIPTOR, fontAlternates);
58
58
  for (const family of documentModel?.requiredFonts ?? []) addLayoutFontFamilyFace(faces, family, REGULAR_LAYOUT_FONT_DESCRIPTOR, fontAlternates);
59
59
  addLayoutFontFamilyFace(faces, documentModel?.package.theme?.fontScheme?.majorFont?.latin, REGULAR_LAYOUT_FONT_DESCRIPTOR, fontAlternates);
60
60
  addLayoutFontFamilyFace(faces, documentModel?.package.theme?.fontScheme?.minorFont?.latin, REGULAR_LAYOUT_FONT_DESCRIPTOR, fontAlternates);
@@ -75,7 +75,7 @@ function collectProseMirrorFontFaces(faces, node, inheritedTextFormatting, fontA
75
75
  if (node.type.name === "paragraph") addTextFormattingFontFaces(faces, expectParagraphAttrs(node).listMarkerFormatting, fontAlternates);
76
76
  if (node.isText) {
77
77
  const descriptor = layoutDescriptorFromFormattingAndMarks(textFormatting, node.marks);
78
- addLayoutFontFamilyFace(faces, readFontFamilyMarkAttrs(node.marks) ?? textFormatting?.fontFamily ?? DEFAULT_LAYOUT_FONT_FAMILY, descriptor, fontAlternates);
78
+ addLayoutFontFamilyFace(faces, readFontFamilyMarkAttrs(node.marks) ?? textFormatting?.fontFamily ?? "Times New Roman", descriptor, fontAlternates);
79
79
  }
80
80
  node.forEach((child) => {
81
81
  collectProseMirrorFontFaces(faces, child, textFormatting, fontAlternates);
@@ -0,0 +1,19 @@
1
+ import { HyphenationDictionaryError, HyphenationDictionaryId } from "../layout-engine/measure/hyphenationDictionaries.js";
2
+ //#region src/controller/hyphenationReadiness.d.ts
3
+ type HyphenationReadiness = {
4
+ /** Follow up on the dictionaries one layout run lacked. */
5
+ track: (missing: ReadonlySet<HyphenationDictionaryId>) => void;
6
+ /**
7
+ * Ignore loads still pending, as on unmount. A later `track` follows up
8
+ * afresh, so an adapter whose lifecycle re-attaches (React Strict Mode
9
+ * effects) keeps working.
10
+ */
11
+ cancel: () => void;
12
+ };
13
+ type HyphenationReadinessOptions = {
14
+ relayout: () => void;
15
+ onError: (error: HyphenationDictionaryError) => void;
16
+ };
17
+ declare const createHyphenationReadiness: ({ relayout, onError }: HyphenationReadinessOptions) => HyphenationReadiness;
18
+ //#endregion
19
+ export { HyphenationReadiness, HyphenationReadinessOptions, createHyphenationReadiness };
@@ -0,0 +1,47 @@
1
+ import { requestHyphenationDictionary } from "../layout-engine/measure/hyphenationDictionaries.js";
2
+ //#region src/controller/hyphenationReadiness.ts
3
+ /**
4
+ * Per-editor follow-up for hyphenation dictionaries a layout run lacked.
5
+ *
6
+ * Framework-neutral so both adapters share one implementation. The layout
7
+ * pipeline hands {@link HyphenationReadiness.track} the dictionaries its run
8
+ * requested but could not use; this editor then re-runs layout when one loads
9
+ * (the load has already invalidated measured paragraphs) and reports a failed
10
+ * one once. Editors that never asked for a dictionary hear nothing about it,
11
+ * and an editor mounted after a failure still learns of it on its first run.
12
+ */
13
+ const createHyphenationReadiness = ({ relayout, onError }) => {
14
+ let epoch = 0;
15
+ const pending = /* @__PURE__ */ new Set();
16
+ const reported = /* @__PURE__ */ new Set();
17
+ const follow = async (dictionary) => {
18
+ const startedIn = epoch;
19
+ pending.add(dictionary);
20
+ const settled = await requestHyphenationDictionary(dictionary);
21
+ if (startedIn !== epoch) return;
22
+ pending.delete(dictionary);
23
+ switch (settled.status) {
24
+ case "loaded":
25
+ relayout();
26
+ return;
27
+ case "failed":
28
+ if (!reported.has(dictionary)) {
29
+ reported.add(dictionary);
30
+ onError(settled.error);
31
+ }
32
+ return;
33
+ default:
34
+ }
35
+ };
36
+ return {
37
+ track: (missing) => {
38
+ for (const dictionary of missing) if (!pending.has(dictionary)) follow(dictionary);
39
+ },
40
+ cancel: () => {
41
+ epoch += 1;
42
+ pending.clear();
43
+ }
44
+ };
45
+ };
46
+ //#endregion
47
+ export { createHyphenationReadiness };
@@ -3,6 +3,7 @@ import { TemplatePreviewEntry, TemplatePreviewHiddenRange } from "../prosemirror
3
3
  import { ColumnLayout, FlowBlock, FootnoteContent, HeaderFooterContent, Layout, Measure, PageHeaderFooterRefs, PageMargins } from "../layout-engine/types.js";
4
4
  import { LayoutRunReason } from "../layout-engine/layoutInstrumentation.js";
5
5
  import { DirtyRange } from "../paged-layout/incrementalMeasure.js";
6
+ import { HyphenationReadiness } from "./hyphenationReadiness.js";
6
7
  import { ConvertHeaderFooterOptions, HeaderFooterMetrics } from "../layout-bridge/convert/headerFooterLayout.js";
7
8
  import "../layout-engine/index.js";
8
9
  import { FootnoteRenderItem } from "../layout-painter/renderPage.js";
@@ -74,6 +75,8 @@ type LayoutPipelineDeps<THfPMs> = {
74
75
  pageRenderer?: PageRendererName;
75
76
  emptyTemplatePreviewEntries: readonly TemplatePreviewEntry[];
76
77
  emptyTemplatePreviewHidden: readonly TemplatePreviewHiddenRange[];
78
+ /** Follows up on the hyphenation dictionaries a run lacked (relayout or error). */
79
+ hyphenationReadiness: HyphenationReadiness;
77
80
  };
78
81
  declare function runLayoutPipeline<THfPMs>(deps: LayoutPipelineDeps<THfPMs>, state: EditorState, options?: LayoutRunOptions): LayoutOutcome;
79
82
  //#endregion