@usejunior/docx-core 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. package/dist/.tsbuildinfo +1 -1
  2. package/dist/generation/index.d.ts +0 -1
  3. package/dist/generation/index.d.ts.map +1 -1
  4. package/dist/generation/index.js +0 -1
  5. package/dist/generation/index.js.map +1 -1
  6. package/dist/index.d.ts +7 -25
  7. package/dist/index.d.ts.map +1 -1
  8. package/dist/index.js +7 -47
  9. package/dist/index.js.map +1 -1
  10. package/dist/primitives/accept_ai_edits.d.ts +87 -0
  11. package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
  12. package/dist/primitives/accept_ai_edits.js +253 -0
  13. package/dist/primitives/accept_ai_edits.js.map +1 -0
  14. package/dist/primitives/accept_changes.d.ts +13 -1
  15. package/dist/primitives/accept_changes.d.ts.map +1 -1
  16. package/dist/primitives/accept_changes.js +44 -29
  17. package/dist/primitives/accept_changes.js.map +1 -1
  18. package/dist/primitives/document.d.ts +51 -0
  19. package/dist/primitives/document.d.ts.map +1 -1
  20. package/dist/primitives/document.js +144 -1
  21. package/dist/primitives/document.js.map +1 -1
  22. package/dist/primitives/index.d.ts +1 -0
  23. package/dist/primitives/index.d.ts.map +1 -1
  24. package/dist/primitives/index.js +1 -0
  25. package/dist/primitives/index.js.map +1 -1
  26. package/dist/primitives/reject_changes.d.ts +4 -1
  27. package/dist/primitives/reject_changes.d.ts.map +1 -1
  28. package/dist/primitives/reject_changes.js +54 -30
  29. package/dist/primitives/reject_changes.js.map +1 -1
  30. package/dist/primitives/relationships.d.ts +33 -0
  31. package/dist/primitives/relationships.d.ts.map +1 -1
  32. package/dist/primitives/relationships.js +84 -1
  33. package/dist/primitives/relationships.js.map +1 -1
  34. package/package.json +8 -9
  35. package/dist/atomizer.d.ts +0 -273
  36. package/dist/atomizer.d.ts.map +0 -1
  37. package/dist/atomizer.js +0 -1002
  38. package/dist/atomizer.js.map +0 -1
  39. package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
  40. package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
  41. package/dist/baselines/atomizer/atomLcs.js +0 -376
  42. package/dist/baselines/atomizer/atomLcs.js.map +0 -1
  43. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
  44. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
  45. package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
  46. package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
  47. package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
  48. package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
  49. package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
  50. package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
  51. package/dist/baselines/atomizer/debug.d.ts +0 -41
  52. package/dist/baselines/atomizer/debug.d.ts.map +0 -1
  53. package/dist/baselines/atomizer/debug.js +0 -85
  54. package/dist/baselines/atomizer/debug.js.map +0 -1
  55. package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
  56. package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
  57. package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
  58. package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
  59. package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
  60. package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
  61. package/dist/baselines/atomizer/formattingFidelity.js +0 -449
  62. package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
  63. package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
  64. package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
  65. package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
  66. package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
  67. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
  68. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
  69. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
  70. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
  71. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
  72. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
  73. package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
  74. package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
  75. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
  76. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
  77. package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
  78. package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
  79. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
  80. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
  81. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
  82. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
  83. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
  84. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
  85. package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
  86. package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
  87. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
  88. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
  89. package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
  90. package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
  91. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
  92. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
  93. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
  94. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
  95. package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
  96. package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
  97. package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
  98. package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
  99. package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
  100. package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
  101. package/dist/baselines/atomizer/numberingIntegration.js +0 -209
  102. package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
  103. package/dist/baselines/atomizer/pipeline.d.ts +0 -103
  104. package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
  105. package/dist/baselines/atomizer/pipeline.js +0 -1160
  106. package/dist/baselines/atomizer/pipeline.js.map +0 -1
  107. package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
  108. package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
  109. package/dist/baselines/atomizer/premergeRuns.js +0 -153
  110. package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
  111. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
  112. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
  113. package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
  114. package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
  115. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
  116. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
  117. package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
  118. package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
  119. package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
  120. package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
  121. package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
  122. package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
  123. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
  124. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
  125. package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
  126. package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
  127. package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
  128. package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
  129. package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
  130. package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
  131. package/dist/cli/compare-two.d.ts +0 -28
  132. package/dist/cli/compare-two.d.ts.map +0 -1
  133. package/dist/cli/compare-two.js +0 -112
  134. package/dist/cli/compare-two.js.map +0 -1
  135. package/dist/cli/index.d.ts +0 -3
  136. package/dist/cli/index.d.ts.map +0 -1
  137. package/dist/cli/index.js +0 -25
  138. package/dist/cli/index.js.map +0 -1
  139. package/dist/compare-types.d.ts +0 -197
  140. package/dist/compare-types.d.ts.map +0 -1
  141. package/dist/compare-types.js +0 -2
  142. package/dist/compare-types.js.map +0 -1
  143. package/dist/format-detection.d.ts +0 -120
  144. package/dist/format-detection.d.ts.map +0 -1
  145. package/dist/format-detection.js +0 -339
  146. package/dist/format-detection.js.map +0 -1
  147. package/dist/generation/recipes.d.ts +0 -144
  148. package/dist/generation/recipes.d.ts.map +0 -1
  149. package/dist/generation/recipes.js +0 -358
  150. package/dist/generation/recipes.js.map +0 -1
  151. package/dist/move-detection.d.ts +0 -211
  152. package/dist/move-detection.d.ts.map +0 -1
  153. package/dist/move-detection.js +0 -390
  154. package/dist/move-detection.js.map +0 -1
package/dist/atomizer.js DELETED
@@ -1,1002 +0,0 @@
1
- /**
2
- * Atomizer Module
3
- *
4
- * Provides factory functions for creating ComparisonUnitAtom instances.
5
- * Implements the core atomization logic from WmlComparer.
6
- *
7
- * @see WmlComparer.cs ComparisonUnitAtom constructor (lines 2314-2343)
8
- */
9
- import { createHash } from 'crypto';
10
- import { parseXml } from './primitives/xml.js';
11
- import { CorrelationStatus, } from './core-types.js';
12
- import { getLeafText, setLeafText, childElements, findChildByTagName, } from './primitives/index.js';
13
- import { OOXML } from './primitives/namespaces.js';
14
- // =============================================================================
15
- // Shared synthetic document for creating virtual elements
16
- // =============================================================================
17
- /**
18
- * A shared document used to create synthetic/virtual DOM elements.
19
- * These elements are not part of any real parsed document.
20
- */
21
- const SYNTHETIC_DOC = parseXml('<root/>');
22
- // =============================================================================
23
- // SHA1 Hashing
24
- // =============================================================================
25
- /**
26
- * Calculate SHA1 hash of a string.
27
- *
28
- * Used for quick equality checking of comparison units.
29
- *
30
- * @param content - The string content to hash
31
- * @returns Hexadecimal SHA1 hash string
32
- */
33
- export function sha1(content) {
34
- return createHash('sha1').update(content, 'utf8').digest('hex');
35
- }
36
- /**
37
- * Attributes that should be excluded from hashing for certain elements.
38
- *
39
- * - xml:space: A whitespace preservation hint that doesn't affect content.
40
- * Documents may have this attribute present on some w:t elements and absent
41
- * on others with identical text, causing spurious hash mismatches.
42
- */
43
- const IGNORED_HASH_ATTRIBUTES = new Set(['xml:space']);
44
- /**
45
- * Calculate SHA1 hash for a WmlElement.
46
- *
47
- * Includes tag name, attributes, and text content for uniqueness.
48
- * Excludes presentation-only attributes like xml:space that don't affect content.
49
- *
50
- * @param element - The element to hash
51
- * @returns Hexadecimal SHA1 hash string
52
- */
53
- export function hashElement(element) {
54
- const parts = [element.tagName];
55
- // Sort attributes for deterministic hashing, excluding presentation-only attributes
56
- const attrs = [];
57
- for (let i = 0; i < element.attributes.length; i++) {
58
- const attr = element.attributes[i];
59
- attrs.push([attr.name, attr.value]);
60
- }
61
- const sortedAttrs = attrs
62
- .filter(([key]) => !IGNORED_HASH_ATTRIBUTES.has(key))
63
- .sort(([a], [b]) => a.localeCompare(b));
64
- for (const [key, value] of sortedAttrs) {
65
- parts.push(`${key}=${value}`);
66
- }
67
- const leafText = getLeafText(element);
68
- if (leafText !== undefined) {
69
- parts.push(leafText);
70
- }
71
- return sha1(parts.join('|'));
72
- }
73
- // =============================================================================
74
- // Revision Tracking Detection
75
- // =============================================================================
76
- /**
77
- * Revision tracking element tag names.
78
- */
79
- const REVISION_TRACKING_TAGS = new Set(['w:ins', 'w:del', 'w:moveFrom', 'w:moveTo']);
80
- /**
81
- * Find a revision tracking element in the ancestor chain.
82
- *
83
- * Searches ancestors from nearest to root for w:ins, w:del, w:moveFrom, or w:moveTo.
84
- *
85
- * @param ancestors - Ancestor elements from root to parent
86
- * @returns The revision tracking element if found, undefined otherwise
87
- */
88
- export function findRevisionTrackingElement(ancestors) {
89
- // Search from nearest ancestor to root
90
- for (let i = ancestors.length - 1; i >= 0; i--) {
91
- const ancestor = ancestors[i];
92
- if (ancestor && REVISION_TRACKING_TAGS.has(ancestor.tagName)) {
93
- return ancestor;
94
- }
95
- }
96
- return undefined;
97
- }
98
- /**
99
- * Determine initial correlation status from revision tracking element.
100
- *
101
- * @param revTrackElement - The revision tracking element (if any)
102
- * @returns Initial correlation status
103
- */
104
- export function getStatusFromRevisionTracking(revTrackElement) {
105
- if (!revTrackElement) {
106
- return CorrelationStatus.Unknown;
107
- }
108
- switch (revTrackElement.tagName) {
109
- case 'w:ins':
110
- return CorrelationStatus.Inserted;
111
- case 'w:del':
112
- return CorrelationStatus.Deleted;
113
- case 'w:moveFrom':
114
- return CorrelationStatus.MovedSource;
115
- case 'w:moveTo':
116
- return CorrelationStatus.MovedDestination;
117
- default:
118
- return CorrelationStatus.Unknown;
119
- }
120
- }
121
- // =============================================================================
122
- // Ancestor Unid Extraction
123
- // =============================================================================
124
- /**
125
- * Extract Unid attributes from ancestor elements.
126
- *
127
- * WmlComparer uses w:Unid attributes to correlate elements between documents.
128
- *
129
- * @param ancestors - Ancestor elements from root to parent
130
- * @returns Array of Unid values found in ancestors
131
- */
132
- export function extractAncestorUnids(ancestors) {
133
- const unids = [];
134
- for (const ancestor of ancestors) {
135
- const unid = ancestor.getAttribute('w:Unid');
136
- if (unid) {
137
- unids.push(unid);
138
- }
139
- }
140
- return unids;
141
- }
142
- // =============================================================================
143
- // Leaf Node Detection
144
- // =============================================================================
145
- /**
146
- * Tag names that represent leaf nodes in the atomization tree.
147
- */
148
- const LEAF_NODE_TAGS = new Set([
149
- 'w:t', // Text
150
- 'w:br', // Break
151
- 'w:cr', // Carriage return
152
- 'w:tab', // Tab character
153
- 'w:sym', // Symbol
154
- 'w:softHyphen', // Soft hyphen
155
- 'w:noBreakHyphen', // Non-breaking hyphen
156
- 'w:fldChar', // Field character
157
- 'w:instrText', // Field instruction text
158
- 'w:delText', // Deleted text
159
- 'w:dayShort', // Date field short day
160
- 'w:dayLong', // Date field long day
161
- 'w:monthShort', // Date field short month
162
- 'w:monthLong', // Date field long month
163
- 'w:yearShort', // Date field short year
164
- 'w:yearLong', // Date field long year
165
- 'w:annotationRef', // Annotation reference
166
- 'w:footnoteRef', // Footnote reference marker
167
- 'w:endnoteRef', // Endnote reference marker
168
- 'w:footnoteReference', // Footnote reference
169
- 'w:endnoteReference', // Endnote reference
170
- 'w:commentReference', // Comment reference anchor (run-level child).
171
- // Note: w:commentRangeStart / w:commentRangeEnd / w:bookmarkStart /
172
- // w:bookmarkEnd / w:moveFromRangeStart/End / w:moveToRangeStart/End /
173
- // w:permStart / w:permEnd are paragraph-level markers
174
- // (siblings of <w:r>, not children). They are
175
- // tracked in PARAGRAPH_LEVEL_TAGS below and atomized via a separate branch in
176
- // atomizeTreeInternal so the reconstructor can emit them outside synthetic
177
- // <w:r> wrappers.
178
- 'w:separator', // Separator
179
- 'w:continuationSeparator', // Continuation separator
180
- 'w:pgNum', // Page number
181
- 'w:drawing', // Drawing (treat as atomic)
182
- 'w:pict', // Picture (VML)
183
- 'w:object', // Embedded object
184
- 'mc:AlternateContent', // Alternate content
185
- ]);
186
- /**
187
- * Tag names that are paragraph-level OOXML markers.
188
- *
189
- * These elements are valid as direct children of <w:p> (and revision wrappers
190
- * like <w:ins>/<w:del>/<w:moveFrom>/<w:moveTo>) but never inside <w:r>. The
191
- * rebuild reconstructor emits them as siblings of <w:r>, not leaves wrapped in
192
- * a synthetic run.
193
- *
194
- * Scope: commentRange, bookmark, moveFromRange / moveToRange, and
195
- * range-permission (permStart / permEnd) markers. Explicit move-range markers
196
- * coexist with the synthetic emission in wrapWithMoveFrom and wrapWithMoveTo:
197
- * the reconstructor suppresses synthesis for paragraphs whose atom stream
198
- * already carries explicit markers of the same kind, so the two paths never
199
- * double-emit.
200
- *
201
- * @conformance ECMA-376 edition 5, Part 1 § 17.13.5
202
- * @see https://github.com/UseJunior/safe-docx/issues/110
203
- * @see https://github.com/UseJunior/safe-docx/issues/111
204
- */
205
- export const PARAGRAPH_LEVEL_TAGS = new Set([
206
- 'w:commentRangeStart',
207
- 'w:commentRangeEnd',
208
- 'w:bookmarkStart',
209
- 'w:bookmarkEnd',
210
- 'w:moveFromRangeStart',
211
- 'w:moveFromRangeEnd',
212
- 'w:moveToRangeStart',
213
- 'w:moveToRangeEnd',
214
- 'w:permStart',
215
- 'w:permEnd',
216
- ]);
217
- /**
218
- * Special tag name for empty paragraph boundary atoms.
219
- * These atoms are created for paragraphs that have no content (only w:pPr).
220
- */
221
- export const EMPTY_PARAGRAPH_TAG = '__emptyParagraph__';
222
- /**
223
- * Check if an element is a leaf node for atomization.
224
- *
225
- * Leaf nodes are the smallest units that can be compared.
226
- *
227
- * @param element - The element to check
228
- * @returns True if this is a leaf node
229
- */
230
- export function isLeafNode(element) {
231
- return LEAF_NODE_TAGS.has(element.tagName);
232
- }
233
- /**
234
- * Check if an element is a paragraph-level OOXML marker.
235
- *
236
- * Paragraph-level markers (PARAGRAPH_LEVEL_TAGS: commentRange*, bookmark*,
237
- * moveFromRange*, moveToRange*, perm*) are
238
- * atomized only when they sit inside a <w:p> ancestor — body/table-sibling
239
- * placements stay out of the atom stream and are handled by the scaffold-strip
240
- * block in the reconstructor.
241
- */
242
- export function isParagraphLevelLeaf(element) {
243
- return PARAGRAPH_LEVEL_TAGS.has(element.tagName);
244
- }
245
- /**
246
- * Create a ComparisonUnitAtom from a leaf element.
247
- *
248
- * Replicates the C# ComparisonUnitAtom constructor logic:
249
- * 1. Finds revision tracking elements in ancestors
250
- * 2. Sets initial correlation status based on revision type
251
- * 3. Extracts ancestor Unids for correlation
252
- * 4. Calculates SHA1 hash for equality checking
253
- *
254
- * @param options - Options containing element, ancestors, and part
255
- * @returns A new ComparisonUnitAtom
256
- *
257
- * @see WmlComparer.cs lines 2314-2343
258
- */
259
- export function createComparisonUnitAtom(options) {
260
- const { contentElement, ancestors, part } = options;
261
- // Find revision tracking element in ancestors
262
- const revTrackElement = findRevisionTrackingElement(ancestors);
263
- // Determine initial correlation status
264
- const correlationStatus = getStatusFromRevisionTracking(revTrackElement);
265
- // Extract Unids from ancestors
266
- const ancestorUnids = extractAncestorUnids(ancestors);
267
- // Calculate SHA1 hash for the atom
268
- const sha1Hash = hashElement(contentElement);
269
- // Extract and clone run properties for first-class rPr access
270
- const rPrElement = getRunProperties({ ancestorElements: ancestors });
271
- const rPr = rPrElement ? rPrElement.cloneNode(true) : null;
272
- return {
273
- contentElement,
274
- ancestorElements: [...ancestors], // Copy to avoid mutation
275
- ancestorUnids,
276
- part,
277
- revTrackElement,
278
- sha1Hash,
279
- correlationStatus,
280
- rPr,
281
- };
282
- }
283
- // =============================================================================
284
- // Tree Atomization
285
- // =============================================================================
286
- /**
287
- * Check if a paragraph element is empty (has no content-bearing children).
288
- *
289
- * Empty paragraphs have only paragraph properties, or proofing-error anchors.
290
- * `w:proofErr` marks spelling/grammar proofing state and carries no document
291
- * content, so a paragraph containing only those anchors is empty for
292
- * comparison.
293
- *
294
- * @conformance ECMA-376 edition 5, Part 1 § 17.13.8.1
295
- * @see https://github.com/UseJunior/safe-docx/issues/456
296
- */
297
- const EMPTY_PARAGRAPH_TRANSPARENT_TAGS = new Set(['w:pPr', 'w:proofErr']);
298
- function isEmptyParagraph(node) {
299
- if (node.tagName !== 'w:p')
300
- return false;
301
- const kids = childElements(node);
302
- if (kids.length === 0)
303
- return true;
304
- for (const child of kids) {
305
- if (!EMPTY_PARAGRAPH_TRANSPARENT_TAGS.has(child.tagName)) {
306
- return false;
307
- }
308
- }
309
- return true;
310
- }
311
- /**
312
- * Create an empty paragraph boundary atom with context-aware hash.
313
- *
314
- * These atoms represent empty paragraphs that have no text content,
315
- * ensuring they are preserved during document reconstruction.
316
- *
317
- * The hash includes a paragraph-level content signature for the previous
318
- * content-bearing paragraph and a consecutive-empty index. Text fragment
319
- * boundaries are ignored, while non-text leaves contribute stable tokens.
320
- *
321
- * @param paragraphElement - The w:p element
322
- * @param ancestors - Ancestor elements from root to parent
323
- * @param part - The OPC part
324
- * @param state - Atomization state with context information
325
- */
326
- function createEmptyParagraphAtomWithContext(paragraphElement, ancestors, part, state) {
327
- // Create a virtual element to represent the empty paragraph
328
- const virtualElement = SYNTHETIC_DOC.createElement(EMPTY_PARAGRAPH_TAG);
329
- // Find revision tracking element in ancestors
330
- const revTrackElement = findRevisionTrackingElement(ancestors);
331
- // Determine initial correlation status
332
- const correlationStatus = getStatusFromRevisionTracking(revTrackElement);
333
- const pPr = findChildByTagName(paragraphElement, 'w:pPr');
334
- const pPrHash = pPr ? hashElement(pPr) : 'no-pPr';
335
- const contextHash = state.lastContentHash || 'document-start';
336
- const hashContent = `empty-paragraph:${contextHash}:${state.consecutiveEmptyIndex}:${pPrHash}`;
337
- return {
338
- contentElement: virtualElement,
339
- ancestorElements: [...ancestors, paragraphElement],
340
- ancestorUnids: extractAncestorUnids(ancestors),
341
- part,
342
- revTrackElement,
343
- sha1Hash: sha1(hashContent),
344
- correlationStatus,
345
- isEmptyParagraph: true, // Mark this as an empty paragraph atom
346
- rPr: null, // Empty paragraphs have no run formatting
347
- };
348
- }
349
- function updateParagraphContentContext(node, atoms, state) {
350
- if (node.tagName !== 'w:p') {
351
- return;
352
- }
353
- const contentAtoms = atoms.filter((atom) => !PARAGRAPH_LEVEL_TAGS.has(atom.contentElement.tagName));
354
- if (contentAtoms.length === 0) {
355
- return;
356
- }
357
- const signature = contentAtoms
358
- .map((atom) => {
359
- if (atom.contentElement.tagName === 'w:t') {
360
- return getLeafText(atom.contentElement) ?? '';
361
- }
362
- return `\u0000${atom.contentElement.tagName}:${atom.sha1Hash}\u0000`;
363
- })
364
- .join('');
365
- state.lastContentHash = sha1(`para-content:${signature}`);
366
- state.consecutiveEmptyIndex = 0;
367
- }
368
- /**
369
- * Internal recursive atomization function with state tracking.
370
- */
371
- function atomizeTreeInternal(node, ancestors, part, state, options) {
372
- const atoms = [];
373
- if (isLeafNode(node)) {
374
- const atom = createComparisonUnitAtom({
375
- contentElement: options.cloneLeafNodes ? node.cloneNode(true) : node,
376
- ancestors,
377
- part,
378
- });
379
- atoms.push(atom);
380
- }
381
- else if (options.atomizeParagraphLevelMarkers &&
382
- isParagraphLevelLeaf(node) &&
383
- ancestors.some((a) => a.tagName === 'w:p')) {
384
- // Paragraph-level markers (commentRange*, bookmark*, perm*) inside a <w:p>
385
- // become atoms so the rebuild reconstructor can re-emit them as siblings
386
- // of <w:r>.
387
- // Body/table-sibling placements are intentionally skipped — they are
388
- // already handled by the scaffold-strip block in the reconstructor and
389
- // would otherwise misattach to the previous paragraph in
390
- // assignParagraphIndices().
391
- const atom = createComparisonUnitAtom({
392
- contentElement: options.cloneLeafNodes ? node.cloneNode(true) : node,
393
- ancestors,
394
- part,
395
- });
396
- atoms.push(atom);
397
- }
398
- else if (isEmptyParagraph(node)) {
399
- // Create empty paragraph atom with context-aware hash
400
- atoms.push(createEmptyParagraphAtomWithContext(node, ancestors, part, state));
401
- state.emptyParagraphCount++;
402
- state.consecutiveEmptyIndex++;
403
- }
404
- else {
405
- for (const child of childElements(node)) {
406
- atoms.push(...atomizeTreeInternal(child, [...ancestors, node], part, state, options));
407
- }
408
- updateParagraphContentContext(node, atoms, state);
409
- }
410
- return atoms;
411
- }
412
- /**
413
- * Atomize a document tree into a flat list of ComparisonUnitAtoms.
414
- *
415
- * Recursively traverses the tree, creating atoms for each leaf node.
416
- * Also creates special atoms for empty paragraphs to preserve document structure.
417
- *
418
- * @param node - The current node in the tree
419
- * @param ancestors - Ancestor elements from root to parent of node
420
- * @param part - The OPC part this tree belongs to
421
- * @returns Array of ComparisonUnitAtoms from leaf nodes
422
- */
423
- export function atomizeTree(node, ancestors, part, options = {}) {
424
- const normalizedOptions = {
425
- cloneLeafNodes: options.cloneLeafNodes ?? false,
426
- mergeAcrossRuns: options.mergeAcrossRuns ?? true,
427
- mergePunctuationAcrossRuns: options.mergePunctuationAcrossRuns ?? true,
428
- splitTextIntoWords: options.splitTextIntoWords ?? true,
429
- atomizeParagraphLevelMarkers: options.atomizeParagraphLevelMarkers ?? false,
430
- };
431
- const state = {
432
- emptyParagraphCount: 0,
433
- consecutiveEmptyIndex: 0,
434
- lastContentHash: '',
435
- };
436
- const rawAtoms = atomizeTreeInternal(node, ancestors, part, state, normalizedOptions);
437
- // Step 1: Collapse field sequences into single atoms based on visible text
438
- // This allows matching between hardcoded text and field references
439
- const fieldCollapsedAtoms = collapseFieldSequences(rawAtoms);
440
- // Step 2: Merge contiguous text atoms with same formatting
441
- // This normalizes different w:t split boundaries
442
- const mergedAtoms = mergeContiguousTextAtoms(fieldCollapsedAtoms, normalizedOptions);
443
- // Step 3: Split merged atoms at word boundaries for finer-grained comparison
444
- // This enables word-level diffing within paragraphs
445
- const wordSplitAtoms = normalizedOptions.splitTextIntoWords
446
- ? splitAtomsIntoWords(mergedAtoms)
447
- : mergedAtoms;
448
- // Step 4: Merge punctuation-only atoms with preceding text
449
- // This handles "Conduct" + "," vs "Conduct," split differences
450
- // Must run AFTER word split since that's when punctuation becomes separate atoms
451
- const atoms = mergePunctuationAtoms(wordSplitAtoms, normalizedOptions);
452
- console.log(`[DEBUG] atomizeTree: created ${rawAtoms.length} atoms, field-collapsed to ${fieldCollapsedAtoms.length}, merged to ${mergedAtoms.length}, word-split to ${wordSplitAtoms.length}, punct-merged to ${atoms.length}, ${state.emptyParagraphCount} empty paragraphs`);
453
- return { atoms, emptyParagraphCount: state.emptyParagraphCount };
454
- }
455
- /**
456
- * Get all ancestors of a node by following parent references.
457
- *
458
- * @param node - The node to get ancestors for
459
- * @returns Array of ancestors from root to immediate parent
460
- */
461
- export function getAncestors(node) {
462
- const ancestors = [];
463
- let current = node.parentNode;
464
- while (current && current.nodeType === 1 /* ELEMENT_NODE */) {
465
- ancestors.unshift(current);
466
- current = current.parentNode;
467
- }
468
- return ancestors;
469
- }
470
- /**
471
- * Assign paragraph indices to atoms based on their w:p ancestors.
472
- *
473
- * This enables paragraph grouping in the document reconstructor when
474
- * merging atoms from different source trees (original vs revised).
475
- *
476
- * @param atoms - Array of atoms to assign indices to
477
- */
478
- export function assignParagraphIndices(atoms) {
479
- const paragraphToIndex = new Map();
480
- let nextIndex = 0;
481
- for (const atom of atoms) {
482
- // Find the w:p ancestor
483
- const pAncestor = atom.ancestorElements.find((a) => a.tagName === 'w:p');
484
- if (pAncestor) {
485
- // Get or assign index for this paragraph
486
- let index = paragraphToIndex.get(pAncestor);
487
- if (index === undefined) {
488
- index = nextIndex++;
489
- paragraphToIndex.set(pAncestor, index);
490
- }
491
- atom.paragraphIndex = index;
492
- }
493
- }
494
- }
495
- // =============================================================================
496
- // Field Sequence Collapsing
497
- // =============================================================================
498
- /**
499
- * Special tag name for collapsed field atoms.
500
- * These represent Word field codes (REF, PAGEREF, etc.) collapsed to their visible result.
501
- */
502
- export const COLLAPSED_FIELD_TAG = '__collapsedField__';
503
- /**
504
- * Check if an atom is a field begin marker.
505
- */
506
- function isFieldBegin(atom) {
507
- return (atom.contentElement.tagName === 'w:fldChar' &&
508
- atom.contentElement.getAttribute('w:fldCharType') === 'begin');
509
- }
510
- /**
511
- * Check if an atom is a field separate marker.
512
- */
513
- function isFieldSeparate(atom) {
514
- return (atom.contentElement.tagName === 'w:fldChar' &&
515
- atom.contentElement.getAttribute('w:fldCharType') === 'separate');
516
- }
517
- /**
518
- * Check if an atom is a field end marker.
519
- */
520
- function isFieldEnd(atom) {
521
- return (atom.contentElement.tagName === 'w:fldChar' &&
522
- atom.contentElement.getAttribute('w:fldCharType') === 'end');
523
- }
524
- /**
525
- * Extract visible text from a sequence of atoms (field result portion).
526
- * Only includes w:t elements, ignoring field markers and instructions.
527
- */
528
- function extractVisibleText(atoms) {
529
- return atoms
530
- .filter((a) => a.contentElement.tagName === 'w:t')
531
- .map((a) => getLeafText(a.contentElement) ?? '')
532
- .join('');
533
- }
534
- /**
535
- * Check if a field spans multiple paragraphs.
536
- * Multi-paragraph fields (like TOC) should not be collapsed.
537
- */
538
- function fieldSpansMultipleParagraphs(fieldAtoms) {
539
- const paragraphs = new Set();
540
- for (const atom of fieldAtoms) {
541
- const para = atom.ancestorElements.find((e) => e.tagName === 'w:p');
542
- if (para) {
543
- paragraphs.add(para);
544
- if (paragraphs.size > 1) {
545
- return true;
546
- }
547
- }
548
- }
549
- return false;
550
- }
551
- /**
552
- * Collapse field sequences into single atoms based on visible text.
553
- *
554
- * Word fields consist of:
555
- * - w:fldChar[begin] - field start
556
- * - w:instrText - field instruction (e.g., "REF _Ref123 \h")
557
- * - w:fldChar[separate] - separates instruction from result
558
- * - w:t (one or more) - visible result text
559
- * - w:fldChar[end] - field end
560
- *
561
- * This function collapses each field sequence into a single atom whose hash
562
- * is based only on the visible text. This allows matching between:
563
- * - Hardcoded text: "2.6"
564
- * - Field reference: [REF field]2.6[/field]
565
- *
566
- * Both will produce atoms with the same hash if the visible text matches.
567
- *
568
- * NOTE: Multi-paragraph fields (like TOC, INDEX) are NOT collapsed because
569
- * they would lose paragraph structure information.
570
- *
571
- * @param atoms - Array of atoms from atomization
572
- * @returns Array with field sequences collapsed to single atoms
573
- */
574
- export function collapseFieldSequences(atoms) {
575
- if (atoms.length === 0)
576
- return atoms;
577
- const result = [];
578
- let i = 0;
579
- while (i < atoms.length) {
580
- const atom = atoms[i];
581
- if (isFieldBegin(atom)) {
582
- // Found field start - collect until matching end
583
- const fieldAtoms = [atom];
584
- let depth = 1;
585
- let separatorIndex = -1;
586
- i++;
587
- while (i < atoms.length && depth > 0) {
588
- const current = atoms[i];
589
- fieldAtoms.push(current);
590
- if (isFieldBegin(current)) {
591
- depth++;
592
- }
593
- else if (isFieldEnd(current)) {
594
- depth--;
595
- }
596
- else if (isFieldSeparate(current) && depth === 1) {
597
- // Track separator position for the outermost field
598
- separatorIndex = fieldAtoms.length - 1;
599
- }
600
- i++;
601
- }
602
- // Check if field spans multiple paragraphs (like TOC, INDEX)
603
- // If so, don't collapse the outer field - preserve paragraph structure.
604
- // But recursively collapse inner single-paragraph fields (e.g., PAGEREF
605
- // nested inside TOC) so they are treated as single atoms during LCS.
606
- if (fieldSpansMultipleParagraphs(fieldAtoms)) {
607
- if (separatorIndex >= 0 && separatorIndex < fieldAtoms.length - 1) {
608
- // Pass through outer field markers: begin, instrText..., separate
609
- result.push(...fieldAtoms.slice(0, separatorIndex + 1));
610
- // Recursively collapse inner content (between separator and end)
611
- const innerContent = fieldAtoms.slice(separatorIndex + 1, -1);
612
- result.push(...collapseFieldSequences(innerContent));
613
- // Pass through outer end marker
614
- result.push(fieldAtoms[fieldAtoms.length - 1]);
615
- }
616
- else {
617
- // No separator found (unusual), pass through unchanged
618
- result.push(...fieldAtoms);
619
- }
620
- continue;
621
- }
622
- // Extract visible text from the field result (after separator)
623
- let visibleText;
624
- if (separatorIndex >= 0) {
625
- // Get text between separator and end (exclusive of markers)
626
- const resultAtoms = fieldAtoms.slice(separatorIndex + 1, -1);
627
- visibleText = extractVisibleText(resultAtoms);
628
- }
629
- else {
630
- // No separator - might be a field with no result yet, use instruction
631
- visibleText = extractVisibleText(fieldAtoms);
632
- }
633
- // Create a collapsed field atom with the visible text
634
- const firstAtom = fieldAtoms[0];
635
- // Use w:t so it can merge with adjacent text
636
- const virtualElement = SYNTHETIC_DOC.createElementNS(OOXML.W_NS, 'w:t');
637
- setLeafText(virtualElement, visibleText);
638
- if (/\s/.test(visibleText)) {
639
- virtualElement.setAttributeNS('http://www.w3.org/XML/1998/namespace', 'xml:space', 'preserve');
640
- }
641
- const collapsedAtom = {
642
- contentElement: virtualElement,
643
- ancestorElements: [...firstAtom.ancestorElements],
644
- ancestorUnids: firstAtom.ancestorUnids,
645
- part: firstAtom.part,
646
- revTrackElement: firstAtom.revTrackElement,
647
- sha1Hash: hashElement(virtualElement),
648
- correlationStatus: firstAtom.correlationStatus,
649
- // Store original atoms for document reconstruction
650
- collapsedFieldAtoms: fieldAtoms,
651
- // Inherit rPr from first atom in the field sequence
652
- rPr: firstAtom.rPr,
653
- };
654
- result.push(collapsedAtom);
655
- }
656
- else {
657
- // Not a field - pass through unchanged
658
- result.push(atom);
659
- i++;
660
- }
661
- }
662
- return result;
663
- }
664
- // =============================================================================
665
- // Word-Level Splitting
666
- // =============================================================================
667
- /**
668
- * Split a w:t atom into word-level atoms.
669
- *
670
- * This enables finer-grained comparison when text is stored in single w:t elements.
671
- * For example, "Hello World" becomes ["Hello", " ", "World"].
672
- *
673
- * Preserves whitespace as separate atoms to maintain spacing.
674
- *
675
- * @param atom - A w:t atom to split
676
- * @returns Array of word-level atoms (or original atom if not w:t)
677
- */
678
- function splitAtomIntoWords(atom) {
679
- // Only split w:t elements
680
- if (atom.contentElement.tagName !== 'w:t') {
681
- return [atom];
682
- }
683
- // Don't split collapsed fields - they should stay as-is
684
- if (atom.collapsedFieldAtoms) {
685
- return [atom];
686
- }
687
- const text = getLeafText(atom.contentElement) ?? '';
688
- // Don't split short text or single words
689
- if (text.length <= 1 || !text.includes(' ')) {
690
- return [atom];
691
- }
692
- // Split into words and whitespace, preserving both
693
- // Uses regex to split on word boundaries while keeping whitespace
694
- const parts = text.split(/(\s+)/);
695
- if (parts.length <= 1) {
696
- return [atom];
697
- }
698
- const result = [];
699
- for (const part of parts) {
700
- if (part === '')
701
- continue;
702
- // Create a new element for this word/whitespace
703
- const wordElement = SYNTHETIC_DOC.createElementNS(OOXML.W_NS, 'w:t');
704
- // Copy attributes from the original content element
705
- for (let i = 0; i < atom.contentElement.attributes.length; i++) {
706
- const attr = atom.contentElement.attributes[i];
707
- wordElement.setAttribute(attr.name, attr.value);
708
- }
709
- setLeafText(wordElement, part);
710
- // Ensure OOXML renderers preserve whitespace in this fragment
711
- if (/\s/.test(part)) {
712
- wordElement.setAttributeNS('http://www.w3.org/XML/1998/namespace', 'xml:space', 'preserve');
713
- }
714
- // Create atom for this word
715
- const wordAtom = {
716
- contentElement: wordElement,
717
- ancestorElements: atom.ancestorElements,
718
- ancestorUnids: atom.ancestorUnids,
719
- part: atom.part,
720
- revTrackElement: atom.revTrackElement,
721
- sha1Hash: hashElement(wordElement),
722
- correlationStatus: atom.correlationStatus,
723
- paragraphIndex: atom.paragraphIndex,
724
- // Track that this came from a split atom for potential later merge
725
- splitFromAtom: atom,
726
- // Share rPr reference (read-only after atomization)
727
- rPr: atom.rPr,
728
- };
729
- result.push(wordAtom);
730
- }
731
- return result;
732
- }
733
- /**
734
- * Split all w:t atoms into word-level atoms.
735
- *
736
- * @param atoms - Array of atoms
737
- * @returns Array with w:t atoms split into words
738
- */
739
- export function splitAtomsIntoWords(atoms) {
740
- const result = [];
741
- for (const atom of atoms) {
742
- result.push(...splitAtomIntoWords(atom));
743
- }
744
- return result;
745
- }
746
- // =============================================================================
747
- // Atom Boundary Normalization
748
- // =============================================================================
749
- /**
750
- * Get the run properties (w:rPr) from an atom's run ancestor.
751
- */
752
- function getRunProperties(atom) {
753
- const run = atom.ancestorElements.find((e) => e.tagName === 'w:r');
754
- if (!run)
755
- return undefined;
756
- return findChildByTagName(run, 'w:rPr') ?? undefined;
757
- }
758
- /**
759
- * Compute a deep hash of an element including its children.
760
- */
761
- function hashElementDeep(element) {
762
- const parts = [element.tagName];
763
- // Sort attributes for deterministic hashing
764
- const attrs = [];
765
- for (let i = 0; i < element.attributes.length; i++) {
766
- const attr = element.attributes[i];
767
- attrs.push([attr.name, attr.value]);
768
- }
769
- const sortedAttrs = attrs.sort(([a], [b]) => a.localeCompare(b));
770
- for (const [key, value] of sortedAttrs) {
771
- parts.push(`${key}=${value}`);
772
- }
773
- const leafText = getLeafText(element);
774
- if (leafText !== undefined) {
775
- parts.push(leafText);
776
- }
777
- // Recursively hash children
778
- for (const child of childElements(element)) {
779
- parts.push(hashElementDeep(child));
780
- }
781
- return sha1(parts.join('|'));
782
- }
783
- /**
784
- * Compare two w:rPr elements for equivalence.
785
- * Returns true if they have the same formatting properties.
786
- */
787
- function runPropertiesEqual(a, b) {
788
- // Both undefined = equal (no formatting)
789
- if (!a && !b)
790
- return true;
791
- // One undefined = not equal
792
- if (!a || !b)
793
- return false;
794
- // Compare by deep hashing (includes children for w:rPr properties)
795
- return hashElementDeep(a) === hashElementDeep(b);
796
- }
797
- /**
798
- * Find the nearest `w:hyperlink` ancestor of an atom, or null.
799
- *
800
- * Boundary normalization must never merge text atoms across a hyperlink
801
- * boundary: the surviving atom keeps a single ancestor chain, so a
802
- * cross-boundary merge either absorbs adjacent plain text into the link
803
- * (formatting bleed) or detaches link text from its wrapper.
804
- *
805
- * @conformance ECMA-376 edition 5, Part 1 § 17.16.22
806
- * @see https://github.com/UseJunior/safe-docx/issues/368
807
- */
808
- export function nearestHyperlinkAncestor(atom) {
809
- for (let i = atom.ancestorElements.length - 1; i >= 0; i--) {
810
- const ancestor = atom.ancestorElements[i];
811
- if (ancestor.tagName === 'w:hyperlink')
812
- return ancestor;
813
- // w:hyperlink always sits between the run and its paragraph; once the
814
- // walk reaches w:p there is no hyperlink wrapper.
815
- if (ancestor.tagName === 'w:p')
816
- break;
817
- }
818
- return null;
819
- }
820
- /**
821
- * Check if two atoms can be merged into one.
822
- *
823
- * Atoms can be merged if they:
824
- * - Are both w:t (text) elements
825
- * - Neither is a collapsed field (fields should stay as separate atoms for finer diff)
826
- * - Are in the same paragraph
827
- * - Are inside the same w:hyperlink wrapper (or both outside one)
828
- * - Have the same run formatting (w:rPr) OR are in the same run
829
- * - Have the same revision tracking status
830
- *
831
- * @param a - First atom
832
- * @param b - Second atom (immediately following a)
833
- * @returns True if atoms can be merged
834
- */
835
- function canMergeAtoms(a, b, options) {
836
- // Only merge w:t elements
837
- if (a.contentElement.tagName !== 'w:t')
838
- return false;
839
- if (b.contentElement.tagName !== 'w:t')
840
- return false;
841
- // Never merge collapsed fields - they should stay as separate atoms for finer-grained diff
842
- if (a.collapsedFieldAtoms || b.collapsedFieldAtoms)
843
- return false;
844
- // Must be in the same paragraph
845
- const aPara = a.ancestorElements.find((e) => e.tagName === 'w:p');
846
- const bPara = b.ancestorElements.find((e) => e.tagName === 'w:p');
847
- if (aPara !== bPara)
848
- return false;
849
- // Must have same revision tracking status
850
- const aRevTag = a.revTrackElement?.tagName;
851
- const bRevTag = b.revTrackElement?.tagName;
852
- if (aRevTag !== bRevTag)
853
- return false;
854
- // Check if same run (fast path)
855
- const aRun = a.ancestorElements.find((e) => e.tagName === 'w:r');
856
- const bRun = b.ancestorElements.find((e) => e.tagName === 'w:r');
857
- if (aRun === bRun)
858
- return true;
859
- // Different runs - allow cross-run merge only if enabled.
860
- // (In inplace mode we disable this so each atom stays anchored to a real run.)
861
- if (!options.mergeAcrossRuns)
862
- return false;
863
- // Never merge across a w:hyperlink boundary — the merged atom keeps only
864
- // one side's ancestry, so link text would detach from (or plain text be
865
- // absorbed into) the hyperlink wrapper.
866
- if (nearestHyperlinkAncestor(a) !== nearestHyperlinkAncestor(b))
867
- return false;
868
- // Different runs - check if they have equivalent formatting
869
- const aRPr = getRunProperties(a);
870
- const bRPr = getRunProperties(b);
871
- return runPropertiesEqual(aRPr, bRPr);
872
- }
873
- /**
874
- * Merge source atom's text content into target atom.
875
- *
876
- * Concatenates text content and recomputes the hash.
877
- *
878
- * @param target - Atom to merge into
879
- * @param source - Atom to merge from
880
- */
881
- function mergeIntoAtom(target, source) {
882
- // Concatenate text content
883
- const newText = (getLeafText(target.contentElement) ?? '') +
884
- (getLeafText(source.contentElement) ?? '');
885
- setLeafText(target.contentElement, newText);
886
- // Recompute hash
887
- target.sha1Hash = hashElement(target.contentElement);
888
- }
889
- /**
890
- * Check if an atom contains only punctuation.
891
- */
892
- function isPunctuationOnlyAtom(atom) {
893
- if (atom.contentElement.tagName !== 'w:t')
894
- return false;
895
- const text = getLeafText(atom.contentElement) ?? '';
896
- // Match common punctuation that should attach to adjacent words
897
- return /^[,.:;!?'")\]}>]+$/.test(text);
898
- }
899
- /**
900
- * Check if two atoms can be merged for punctuation normalization.
901
- *
902
- * More permissive than canMergeAtoms - allows merging punctuation with
903
- * preceding text even if they're in different runs, as long as they're
904
- * in the same paragraph and have the same revision tracking status.
905
- */
906
- function canMergePunctuation(a, b, options) {
907
- // Only merge w:t elements
908
- if (a.contentElement.tagName !== 'w:t')
909
- return false;
910
- if (b.contentElement.tagName !== 'w:t')
911
- return false;
912
- // B must be punctuation-only
913
- if (!isPunctuationOnlyAtom(b))
914
- return false;
915
- // Never merge collapsed fields
916
- if (a.collapsedFieldAtoms || b.collapsedFieldAtoms)
917
- return false;
918
- // Must be in the same paragraph
919
- const aPara = a.ancestorElements.find((e) => e.tagName === 'w:p');
920
- const bPara = b.ancestorElements.find((e) => e.tagName === 'w:p');
921
- if (aPara !== bPara)
922
- return false;
923
- // Must have same revision tracking status
924
- const aRevTag = a.revTrackElement?.tagName;
925
- const bRevTag = b.revTrackElement?.tagName;
926
- if (aRevTag !== bRevTag)
927
- return false;
928
- // A must end with a word character (not whitespace or punctuation)
929
- const aText = getLeafText(a.contentElement) ?? '';
930
- if (!/\w$/.test(aText))
931
- return false;
932
- // Never merge across a w:hyperlink boundary: punctuation that follows a
933
- // link must not inherit the link run's ancestry/formatting (e.g. the
934
- // sentence period after a URL turning underlined).
935
- if (nearestHyperlinkAncestor(a) !== nearestHyperlinkAncestor(b))
936
- return false;
937
- // If cross-run punctuation merge is disabled, require same run.
938
- if (!options.mergePunctuationAcrossRuns) {
939
- const aRun = a.ancestorElements.find((e) => e.tagName === 'w:r');
940
- const bRun = b.ancestorElements.find((e) => e.tagName === 'w:r');
941
- if (aRun !== bRun)
942
- return false;
943
- }
944
- return true;
945
- }
946
- /**
947
- * Merge punctuation-only atoms with preceding text.
948
- *
949
- * This handles cases where documents have different w:t boundaries around
950
- * punctuation (e.g., "Conduct" + "," vs "Conduct,"). Punctuation is merged
951
- * with the preceding word regardless of run formatting differences.
952
- *
953
- * @param atoms - Array of atoms
954
- * @returns Atoms with punctuation merged into preceding text
955
- */
956
- export function mergePunctuationAtoms(atoms, options = { mergePunctuationAcrossRuns: true }) {
957
- if (atoms.length === 0)
958
- return atoms;
959
- const result = [];
960
- for (const atom of atoms) {
961
- const prev = result[result.length - 1];
962
- if (prev && canMergePunctuation(prev, atom, options)) {
963
- // Merge punctuation into previous atom
964
- mergeIntoAtom(prev, atom);
965
- }
966
- else {
967
- result.push(atom);
968
- }
969
- }
970
- return result;
971
- }
972
- /**
973
- * Merge contiguous w:t atoms within the same run into single atoms.
974
- *
975
- * This normalization ensures that identical text split differently across
976
- * w:t elements in original vs revised documents will produce matching hashes.
977
- *
978
- * Example:
979
- * Before: ["Def", "initions"] (2 atoms)
980
- * After: ["Definitions"] (1 atom)
981
- *
982
- * @param atoms - Array of atoms from atomization
983
- * @returns Normalized array with contiguous text atoms merged
984
- */
985
- export function mergeContiguousTextAtoms(atoms, options = { mergeAcrossRuns: true }) {
986
- if (atoms.length === 0)
987
- return atoms;
988
- const result = [];
989
- for (const atom of atoms) {
990
- const prev = result[result.length - 1];
991
- // Only merge w:t elements in the same run
992
- if (prev && canMergeAtoms(prev, atom, options)) {
993
- // Merge text content into previous atom
994
- mergeIntoAtom(prev, atom);
995
- }
996
- else {
997
- result.push(atom);
998
- }
999
- }
1000
- return result;
1001
- }
1002
- //# sourceMappingURL=atomizer.js.map