@usejunior/docx-core 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/README.md +18 -83
  2. package/dist/.tsbuildinfo +1 -1
  3. package/dist/cli/conformance-adapter.d.ts +14 -0
  4. package/dist/cli/conformance-adapter.d.ts.map +1 -1
  5. package/dist/cli/conformance-adapter.js +104 -12
  6. package/dist/cli/conformance-adapter.js.map +1 -1
  7. package/dist/footnotes.d.ts +7 -5
  8. package/dist/footnotes.d.ts.map +1 -1
  9. package/dist/footnotes.js +7 -5
  10. package/dist/footnotes.js.map +1 -1
  11. package/dist/generated/ecma-376-vocabulary.d.ts +93 -0
  12. package/dist/generated/ecma-376-vocabulary.d.ts.map +1 -0
  13. package/dist/generated/ecma-376-vocabulary.js +87 -0
  14. package/dist/generated/ecma-376-vocabulary.js.map +1 -0
  15. package/dist/generation/compile.js +2 -2
  16. package/dist/generation/compile.js.map +1 -1
  17. package/dist/generation/emit/comments-part.d.ts +2 -0
  18. package/dist/generation/emit/comments-part.d.ts.map +1 -1
  19. package/dist/generation/emit/comments-part.js +2 -0
  20. package/dist/generation/emit/comments-part.js.map +1 -1
  21. package/dist/generation/emit/paragraph.d.ts +3 -0
  22. package/dist/generation/emit/paragraph.d.ts.map +1 -1
  23. package/dist/generation/emit/paragraph.js +3 -0
  24. package/dist/generation/emit/paragraph.js.map +1 -1
  25. package/dist/generation/emit/run.d.ts.map +1 -1
  26. package/dist/generation/emit/run.js +6 -1
  27. package/dist/generation/emit/run.js.map +1 -1
  28. package/dist/generation/emit/settings-part.d.ts +12 -3
  29. package/dist/generation/emit/settings-part.d.ts.map +1 -1
  30. package/dist/generation/emit/settings-part.js +23 -5
  31. package/dist/generation/emit/settings-part.js.map +1 -1
  32. package/dist/generation/ordering.d.ts +3 -1
  33. package/dist/generation/ordering.d.ts.map +1 -1
  34. package/dist/generation/ordering.js +3 -1
  35. package/dist/generation/ordering.js.map +1 -1
  36. package/dist/generation/schema-enum-domains.d.ts +27 -0
  37. package/dist/generation/schema-enum-domains.d.ts.map +1 -0
  38. package/dist/generation/schema-enum-domains.js +69 -0
  39. package/dist/generation/schema-enum-domains.js.map +1 -0
  40. package/dist/generation/structural-checks.js +7 -1
  41. package/dist/generation/structural-checks.js.map +1 -1
  42. package/dist/generation/validate-spec.d.ts.map +1 -1
  43. package/dist/generation/validate-spec.js +149 -31
  44. package/dist/generation/validate-spec.js.map +1 -1
  45. package/dist/index.d.ts +7 -24
  46. package/dist/index.d.ts.map +1 -1
  47. package/dist/index.js +7 -46
  48. package/dist/index.js.map +1 -1
  49. package/dist/primitives/accept_ai_edits.d.ts +87 -0
  50. package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
  51. package/dist/primitives/accept_ai_edits.js +253 -0
  52. package/dist/primitives/accept_ai_edits.js.map +1 -0
  53. package/dist/primitives/accept_changes.d.ts +36 -3
  54. package/dist/primitives/accept_changes.d.ts.map +1 -1
  55. package/dist/primitives/accept_changes.js +67 -31
  56. package/dist/primitives/accept_changes.js.map +1 -1
  57. package/dist/primitives/bookmarks.d.ts +44 -0
  58. package/dist/primitives/bookmarks.d.ts.map +1 -1
  59. package/dist/primitives/bookmarks.js +149 -11
  60. package/dist/primitives/bookmarks.js.map +1 -1
  61. package/dist/primitives/document.d.ts +53 -1
  62. package/dist/primitives/document.d.ts.map +1 -1
  63. package/dist/primitives/document.js +146 -2
  64. package/dist/primitives/document.js.map +1 -1
  65. package/dist/primitives/dom-helpers.d.ts.map +1 -1
  66. package/dist/primitives/dom-helpers.js +7 -2
  67. package/dist/primitives/dom-helpers.js.map +1 -1
  68. package/dist/primitives/index.d.ts +2 -1
  69. package/dist/primitives/index.d.ts.map +1 -1
  70. package/dist/primitives/index.js +2 -1
  71. package/dist/primitives/index.js.map +1 -1
  72. package/dist/primitives/layout.d.ts.map +1 -1
  73. package/dist/primitives/layout.js +41 -1
  74. package/dist/primitives/layout.js.map +1 -1
  75. package/dist/primitives/namespaces.d.ts +2 -0
  76. package/dist/primitives/namespaces.d.ts.map +1 -1
  77. package/dist/primitives/namespaces.js +2 -0
  78. package/dist/primitives/namespaces.js.map +1 -1
  79. package/dist/primitives/reject_changes.d.ts +21 -2
  80. package/dist/primitives/reject_changes.d.ts.map +1 -1
  81. package/dist/primitives/reject_changes.js +71 -31
  82. package/dist/primitives/reject_changes.js.map +1 -1
  83. package/dist/primitives/relationships.d.ts +33 -0
  84. package/dist/primitives/relationships.d.ts.map +1 -1
  85. package/dist/primitives/relationships.js +84 -1
  86. package/dist/primitives/relationships.js.map +1 -1
  87. package/dist/primitives/sectPrAudit.d.ts +11 -2
  88. package/dist/primitives/sectPrAudit.d.ts.map +1 -1
  89. package/dist/primitives/sectPrAudit.js +148 -23
  90. package/dist/primitives/sectPrAudit.js.map +1 -1
  91. package/dist/primitives/track-changes-emitter.d.ts +4 -0
  92. package/dist/primitives/track-changes-emitter.d.ts.map +1 -1
  93. package/dist/primitives/track-changes-emitter.js +5 -2
  94. package/dist/primitives/track-changes-emitter.js.map +1 -1
  95. package/dist/primitives/validate_ai_revisions.d.ts.map +1 -1
  96. package/dist/primitives/validate_ai_revisions.js +13 -5
  97. package/dist/primitives/validate_ai_revisions.js.map +1 -1
  98. package/dist/primitives/zip.d.ts.map +1 -1
  99. package/dist/primitives/zip.js +6 -2
  100. package/dist/primitives/zip.js.map +1 -1
  101. package/dist/shared/field-structure.d.ts +6 -0
  102. package/dist/shared/field-structure.d.ts.map +1 -1
  103. package/dist/shared/field-structure.js +6 -0
  104. package/dist/shared/field-structure.js.map +1 -1
  105. package/package.json +4 -8
  106. package/dist/atomizer.d.ts +0 -273
  107. package/dist/atomizer.d.ts.map +0 -1
  108. package/dist/atomizer.js +0 -1002
  109. package/dist/atomizer.js.map +0 -1
  110. package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
  111. package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
  112. package/dist/baselines/atomizer/atomLcs.js +0 -376
  113. package/dist/baselines/atomizer/atomLcs.js.map +0 -1
  114. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
  115. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
  116. package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
  117. package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
  118. package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
  119. package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
  120. package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
  121. package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
  122. package/dist/baselines/atomizer/debug.d.ts +0 -41
  123. package/dist/baselines/atomizer/debug.d.ts.map +0 -1
  124. package/dist/baselines/atomizer/debug.js +0 -85
  125. package/dist/baselines/atomizer/debug.js.map +0 -1
  126. package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
  127. package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
  128. package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
  129. package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
  130. package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
  131. package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
  132. package/dist/baselines/atomizer/formattingFidelity.js +0 -449
  133. package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
  134. package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
  135. package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
  136. package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
  137. package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
  138. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
  139. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
  140. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
  141. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
  142. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
  143. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
  144. package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
  145. package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
  146. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
  147. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
  148. package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
  149. package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
  150. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
  151. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
  152. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
  153. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
  154. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
  155. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
  156. package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
  157. package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
  158. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
  159. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
  160. package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
  161. package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
  162. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
  163. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
  164. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
  165. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
  166. package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
  167. package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
  168. package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
  169. package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
  170. package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
  171. package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
  172. package/dist/baselines/atomizer/numberingIntegration.js +0 -209
  173. package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
  174. package/dist/baselines/atomizer/pipeline.d.ts +0 -103
  175. package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
  176. package/dist/baselines/atomizer/pipeline.js +0 -1160
  177. package/dist/baselines/atomizer/pipeline.js.map +0 -1
  178. package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
  179. package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
  180. package/dist/baselines/atomizer/premergeRuns.js +0 -153
  181. package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
  182. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
  183. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
  184. package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
  185. package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
  186. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
  187. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
  188. package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
  189. package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
  190. package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
  191. package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
  192. package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
  193. package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
  194. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
  195. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
  196. package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
  197. package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
  198. package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
  199. package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
  200. package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
  201. package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
  202. package/dist/cli/compare-two.d.ts +0 -28
  203. package/dist/cli/compare-two.d.ts.map +0 -1
  204. package/dist/cli/compare-two.js +0 -112
  205. package/dist/cli/compare-two.js.map +0 -1
  206. package/dist/cli/index.d.ts +0 -3
  207. package/dist/cli/index.d.ts.map +0 -1
  208. package/dist/cli/index.js +0 -25
  209. package/dist/cli/index.js.map +0 -1
  210. package/dist/compare-types.d.ts +0 -197
  211. package/dist/compare-types.d.ts.map +0 -1
  212. package/dist/compare-types.js +0 -2
  213. package/dist/compare-types.js.map +0 -1
  214. package/dist/format-detection.d.ts +0 -120
  215. package/dist/format-detection.d.ts.map +0 -1
  216. package/dist/format-detection.js +0 -339
  217. package/dist/format-detection.js.map +0 -1
  218. package/dist/move-detection.d.ts +0 -211
  219. package/dist/move-detection.d.ts.map +0 -1
  220. package/dist/move-detection.js +0 -390
  221. package/dist/move-detection.js.map +0 -1
@@ -1,1449 +0,0 @@
1
- /**
2
- * Document Reconstructor
3
- *
4
- * Rebuilds document.xml from marked atoms with track changes.
5
- * Generates w:ins, w:del, w:moveFrom, w:moveTo elements as appropriate.
6
- */
7
- import { XMLSerializer } from '@xmldom/xmldom';
8
- import { parseXml } from '../../primitives/xml.js';
9
- import { CorrelationStatus } from '../../core-types.js';
10
- import { getLeafText, childElements, findChildByTagName } from '../../primitives/index.js';
11
- import { allocateRevisionId, buildPPrChangeElement, convertSerializedDeletionContent, createRevisionContext, createRevisionIdState, escapeXmlAttr, formatDate, wrapSerializedContentWithDel, wrapSerializedContentWithIns, } from '../../primitives/track-changes-emitter.js';
12
- import { serializeToXml, cloneElement } from './xmlToWmlElement.js';
13
- import { EMPTY_PARAGRAPH_TAG, isParagraphLevelLeaf, nearestHyperlinkAncestor } from '../../atomizer.js';
14
- import { enforceConsumerCompatibility } from './consumerCompatibility.js';
15
- import { placeParagraphMarkRevisionMarker } from './inPlaceModifier-wrappers.js';
16
- import { areRunPropertiesEqual } from '../../format-detection.js';
17
- import { debug } from './debug.js';
18
- const SYNTHETIC_DOC = parseXml('<root xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"/>');
19
- const W_NS = 'http://schemas.openxmlformats.org/wordprocessingml/2006/main';
20
- function createEl(tag, attrs) {
21
- const el = SYNTHETIC_DOC.createElementNS(W_NS, tag);
22
- if (attrs)
23
- for (const [k, v] of Object.entries(attrs))
24
- el.setAttribute(k, v);
25
- return el;
26
- }
27
- /**
28
- * Get or allocate move range IDs for a move name.
29
- */
30
- function getMoveRangeIds(state, moveName) {
31
- let ids = state.moveRangeIds.get(moveName);
32
- if (!ids) {
33
- ids = {
34
- sourceRangeId: allocateRevisionId(state),
35
- destRangeId: allocateRevisionId(state),
36
- };
37
- state.moveRangeIds.set(moveName, ids);
38
- }
39
- return ids;
40
- }
41
- /**
42
- * Reconstruct document.xml from merged atoms with track changes.
43
- *
44
- * @param mergedAtoms - Atoms with correlation status set
45
- * @param originalXml - Original document.xml for structure preservation
46
- * @param options - Reconstruction options
47
- * @returns New document.xml with track changes
48
- */
49
- export function reconstructDocument(mergedAtoms, originalXml, options) {
50
- const { author, date } = options;
51
- const dateStr = formatDate(date);
52
- const revState = createRevisionIdState();
53
- // Group atoms by paragraph
54
- const rawParagraphGroups = groupAtomsByParagraph(mergedAtoms);
55
- // Consolidate adjacent same-status changes for better readability
56
- const paragraphGroups = consolidateAdjacentChanges(rawParagraphGroups);
57
- // Reset debug counters
58
- resetDebugCounters();
59
- resetEmptyParagraphCounters();
60
- debug('reconstructor', `${mergedAtoms.length} atoms -> ${paragraphGroups.length} paragraphs`);
61
- // Build track changes XML for each paragraph
62
- const paragraphXmls = [];
63
- for (const group of paragraphGroups) {
64
- const paragraphXml = buildParagraphXml(group, author, dateStr, revState);
65
- paragraphXmls.push(paragraphXml);
66
- }
67
- const counters = getDebugCounters();
68
- debug('reconstructor', `buildRunContent processed: ${counters.atoms} atoms, ${counters.wt} w:t elements`);
69
- const emptyCounters = getEmptyParagraphCounters();
70
- debug('reconstructor', `Empty paragraphs: inserted=${emptyCounters.inserted}, deleted=${emptyCounters.deleted}, equal=${emptyCounters.equal}, other=${emptyCounters.other}`);
71
- // Reconstruct the document, preserving original body structure (tables, SDTs, etc.)
72
- return buildDocumentPreservingStructure(originalXml, paragraphXmls, paragraphGroups, () => allocateRevisionId(revState));
73
- }
74
- /**
75
- * Group atoms by paragraph based on their ancestor chain.
76
- *
77
- * First sorts atoms by paragraphIndex to ensure all atoms belonging to the same
78
- * paragraph are contiguous, then groups them sequentially.
79
- */
80
- function groupAtomsByParagraph(atoms) {
81
- const groups = [];
82
- let currentGroup = null;
83
- let currentRunGroup = null;
84
- const uniqueIndices = new Set(atoms.map(a => a.paragraphIndex));
85
- debug('reconstructor', `groupAtomsByParagraph: ${atoms.length} atoms, ${uniqueIndices.size} unique paragraphIndices`);
86
- // Sort atoms by paragraphIndex to ensure all atoms with the same index are contiguous.
87
- // Use stable sort to preserve relative order within the same paragraph (deleted before inserted).
88
- const sortedAtoms = [...atoms].sort((a, b) => {
89
- const aIdx = a.paragraphIndex ?? Number.MAX_SAFE_INTEGER;
90
- const bIdx = b.paragraphIndex ?? Number.MAX_SAFE_INTEGER;
91
- return aIdx - bIdx;
92
- });
93
- for (const atom of sortedAtoms) {
94
- // Find paragraph ancestor
95
- const pAncestor = findAncestorByTag(atom, 'w:p');
96
- const rAncestor = findAncestorByTag(atom, 'w:r');
97
- // Check if we need a new paragraph
98
- const pPr = pAncestor ? findChildByTag(pAncestor, 'w:pPr') : null;
99
- // Pass currentRunGroup and current atom to check if we should start a new paragraph
100
- // Uses paragraphIndex for comparison instead of object references
101
- if (!currentGroup || shouldStartNewParagraph(currentGroup, currentRunGroup, atom)) {
102
- if (currentRunGroup && currentGroup) {
103
- currentGroup.runGroups.push(currentRunGroup);
104
- }
105
- currentRunGroup = null;
106
- currentGroup = {
107
- pPr: pPr ? cloneElement(pPr) : null,
108
- runGroups: [],
109
- };
110
- groups.push(currentGroup);
111
- }
112
- // Check if we need a new run group
113
- // Use the first-class rPr field from the atom when available,
114
- // falling back to ancestor walk for atoms created before rPr was populated.
115
- const atomRPr = getEffectiveAtomRPr(atom);
116
- const rPr = atomRPr ?? (rAncestor ? findChildByTag(rAncestor, 'w:rPr') : null);
117
- if (!currentRunGroup || shouldStartNewRunGroup(currentRunGroup, atom)) {
118
- if (currentRunGroup) {
119
- currentGroup.runGroups.push(currentRunGroup);
120
- }
121
- currentRunGroup = {
122
- status: atom.correlationStatus,
123
- atoms: [atom],
124
- rPr: rPr ? cloneElement(rPr) : null,
125
- moveName: atom.moveName,
126
- };
127
- }
128
- else {
129
- currentRunGroup.atoms.push(atom);
130
- }
131
- }
132
- // Don't forget the last groups
133
- if (currentRunGroup && currentGroup) {
134
- currentGroup.runGroups.push(currentRunGroup);
135
- }
136
- return groups;
137
- }
138
- /**
139
- * Check if a RunGroup contains only whitespace.
140
- */
141
- function isWhitespaceOnlyGroup(group) {
142
- return group.atoms.every(atom => {
143
- const text = getLeafText(atom.contentElement) ?? '';
144
- return text.trim() === '';
145
- });
146
- }
147
- /**
148
- * Reorder atoms within change blocks.
149
- *
150
- * Identifies "change blocks" (contiguous regions with Del/Ins) and reorders
151
- * to put all deletions first, then all insertions.
152
- * Whitespace between changes is duplicated into both groups to preserve it
153
- * regardless of accept/reject.
154
- */
155
- function reorderChangeBlocks(groups) {
156
- for (const paraGroup of groups) {
157
- const runGroups = paraGroup.runGroups;
158
- const result = [];
159
- let i = 0;
160
- while (i < runGroups.length) {
161
- const current = runGroups[i];
162
- // Check if we're entering a change block
163
- const isChange = current.status === CorrelationStatus.Deleted ||
164
- current.status === CorrelationStatus.Inserted;
165
- if (!isChange) {
166
- result.push(current);
167
- i++;
168
- continue;
169
- }
170
- // Collect the entire change block
171
- const deletions = [];
172
- const insertions = [];
173
- while (i < runGroups.length) {
174
- const group = runGroups[i];
175
- if (group.status === CorrelationStatus.Deleted) {
176
- deletions.push(...group.atoms);
177
- i++;
178
- }
179
- else if (group.status === CorrelationStatus.Inserted) {
180
- insertions.push(...group.atoms);
181
- i++;
182
- }
183
- else if (group.status === CorrelationStatus.Equal && isWhitespaceOnlyGroup(group)) {
184
- // Duplicate whitespace into both deletions and insertions
185
- // so it's preserved regardless of accept/reject
186
- for (const atom of group.atoms) {
187
- // Clone for deletions (mark as deleted)
188
- const delAtom = {
189
- ...atom,
190
- correlationStatus: CorrelationStatus.Deleted,
191
- };
192
- deletions.push(delAtom);
193
- // Clone for insertions (mark as inserted)
194
- const insAtom = {
195
- ...atom,
196
- correlationStatus: CorrelationStatus.Inserted,
197
- };
198
- insertions.push(insAtom);
199
- }
200
- i++;
201
- }
202
- else {
203
- // Non-whitespace Equal or other status - end of block
204
- break;
205
- }
206
- }
207
- // Output reordered: all deletions first, then all insertions
208
- // rPr is set to null — buildRunContent will sub-group atoms by rPr
209
- if (deletions.length > 0) {
210
- result.push({
211
- status: CorrelationStatus.Deleted,
212
- atoms: deletions,
213
- rPr: null,
214
- });
215
- }
216
- if (insertions.length > 0) {
217
- result.push({
218
- status: CorrelationStatus.Inserted,
219
- atoms: insertions,
220
- rPr: null,
221
- });
222
- }
223
- }
224
- paraGroup.runGroups = result;
225
- }
226
- return groups;
227
- }
228
- /**
229
- * Consolidate adjacent RunGroups with the same status within each paragraph.
230
- *
231
- * This makes change tracking more readable by grouping consecutive deletions
232
- * together and consecutive insertions together, rather than interleaving them
233
- * at the word level.
234
- *
235
- * For example, instead of:
236
- * <del>word1</del><ins>word2</ins> <del>word3</del><ins>word4</ins>
237
- *
238
- * We get:
239
- * <del>word1 word3</del><ins>word2 word4</ins>
240
- */
241
- function consolidateAdjacentChanges(groups) {
242
- return reorderChangeBlocks(groups);
243
- }
244
- /**
245
- * Find an ancestor element by tag name.
246
- */
247
- function findAncestorByTag(atom, tagName) {
248
- for (let i = atom.ancestorElements.length - 1; i >= 0; i--) {
249
- if (atom.ancestorElements[i].tagName === tagName) {
250
- return atom.ancestorElements[i];
251
- }
252
- }
253
- return null;
254
- }
255
- /**
256
- * Find a child element by tag name.
257
- */
258
- function findChildByTag(element, tagName) {
259
- for (let i = 0; i < element.childNodes.length; i++) {
260
- const child = element.childNodes[i];
261
- if (child.nodeType === 1 && child.tagName === tagName) {
262
- return child;
263
- }
264
- }
265
- return null;
266
- }
267
- /**
268
- * Determine if we should start a new paragraph.
269
- *
270
- * Uses paragraphIndex for comparison instead of object references, because
271
- * atoms from original and revised documents have different tree objects.
272
- *
273
- * @param currentGroup - The current paragraph group being built
274
- * @param currentRunGroup - The current run group (may not be pushed to currentGroup yet)
275
- * @param currentAtom - The current atom being processed
276
- */
277
- function shouldStartNewParagraph(currentGroup, currentRunGroup, currentAtom) {
278
- const currentParagraphIndex = currentAtom.paragraphIndex;
279
- // If no paragraph index, fall back to false (stay in current paragraph)
280
- if (currentParagraphIndex === undefined)
281
- return false;
282
- // First check currentRunGroup (which may not be pushed to runGroups yet)
283
- if (currentRunGroup && currentRunGroup.atoms.length > 0) {
284
- const lastAtom = currentRunGroup.atoms[currentRunGroup.atoms.length - 1];
285
- const lastParagraphIndex = lastAtom.paragraphIndex;
286
- // Same paragraph index means same paragraph, even if from different trees
287
- if (lastParagraphIndex !== undefined) {
288
- return currentParagraphIndex !== lastParagraphIndex;
289
- }
290
- }
291
- // Fall back to checking runGroups
292
- if (currentGroup.runGroups.length === 0) {
293
- return false;
294
- }
295
- // Check last atom's paragraph index
296
- const lastRunGroup = currentGroup.runGroups[currentGroup.runGroups.length - 1];
297
- if (!lastRunGroup || lastRunGroup.atoms.length === 0) {
298
- return false;
299
- }
300
- const lastAtom = lastRunGroup.atoms[lastRunGroup.atoms.length - 1];
301
- const lastParagraphIndex = lastAtom.paragraphIndex;
302
- if (lastParagraphIndex !== undefined) {
303
- return currentParagraphIndex !== lastParagraphIndex;
304
- }
305
- // No paragraph indices available, stay in current paragraph
306
- return false;
307
- }
308
- /**
309
- * Get the effective rPr for an atom — uses the first-class `rPr` field
310
- * when available, otherwise returns null.
311
- */
312
- function getEffectiveAtomRPr(atom) {
313
- return atom.rPr ?? null;
314
- }
315
- /**
316
- * Determine if we should start a new run group.
317
- */
318
- function shouldStartNewRunGroup(currentGroup, atom) {
319
- // Different status = new group
320
- if (currentGroup.status !== atom.correlationStatus) {
321
- return true;
322
- }
323
- // Different move name = new group
324
- if (currentGroup.moveName !== atom.moveName) {
325
- return true;
326
- }
327
- // Skip rPr splitting for MovedSource/MovedDestination: every moved run
328
- // group is wrapped by wrapWithMoveFrom/wrapWithMoveTo, so splitting one
329
- // move into several groups would emit moveFromRangeStart/End (resp.
330
- // moveToRangeStart/End) once per slice with the same w:name and range ids.
331
- // This stays required now that explicit move-range markers atomize: the
332
- // synthetic-range suppression keyed off those markers is per paragraph, so
333
- // a detected move in a marker-free paragraph still synthesizes one range
334
- // pair per moved run group.
335
- if (currentGroup.status === CorrelationStatus.MovedSource ||
336
- currentGroup.status === CorrelationStatus.MovedDestination) {
337
- return false;
338
- }
339
- // Different rPr = new group (prevents formatting bleed between runs)
340
- const currentRPr = getEffectiveAtomRPr(currentGroup.atoms[currentGroup.atoms.length - 1]);
341
- const newRPr = getEffectiveAtomRPr(atom);
342
- // Fast path: reference equality or both null
343
- if (currentRPr === newRPr)
344
- return false;
345
- if (currentRPr === null && newRPr === null)
346
- return false;
347
- return !areRunPropertiesEqual(currentRPr, newRPr);
348
- }
349
- /**
350
- * Check if a paragraph group represents an empty paragraph with a specific status.
351
- *
352
- * @param group - The paragraph group to check
353
- * @param status - The correlation status to check for
354
- * @returns True if all atoms are empty paragraph markers with the given status
355
- */
356
- function isEmptyParagraphWithStatus(group, status) {
357
- // Check if all run groups contain only empty paragraph atoms with the given status
358
- for (const runGroup of group.runGroups) {
359
- // If any atom is not an empty paragraph marker, this is not an empty paragraph
360
- const hasNonEmptyAtom = runGroup.atoms.some((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
361
- if (hasNonEmptyAtom) {
362
- return false;
363
- }
364
- // If any atom doesn't have the expected status, return false
365
- const hasWrongStatus = runGroup.atoms.some((atom) => atom.correlationStatus !== status);
366
- if (hasWrongStatus) {
367
- return false;
368
- }
369
- }
370
- // All atoms are empty paragraph markers with the expected status
371
- return group.runGroups.length > 0;
372
- }
373
- // Debug counters for empty paragraphs
374
- let debugEmptyParaInserted = 0;
375
- let debugEmptyParaDeleted = 0;
376
- let debugEmptyParaEqual = 0;
377
- let debugEmptyParaOther = 0;
378
- /**
379
- * Reset empty paragraph debug counters.
380
- */
381
- export function resetEmptyParagraphCounters() {
382
- debugEmptyParaInserted = 0;
383
- debugEmptyParaDeleted = 0;
384
- debugEmptyParaEqual = 0;
385
- debugEmptyParaOther = 0;
386
- }
387
- /**
388
- * Get empty paragraph debug counters.
389
- */
390
- export function getEmptyParagraphCounters() {
391
- return {
392
- inserted: debugEmptyParaInserted,
393
- deleted: debugEmptyParaDeleted,
394
- equal: debugEmptyParaEqual,
395
- other: debugEmptyParaOther,
396
- };
397
- }
398
- /**
399
- * Check if a paragraph group contains only empty paragraph atoms.
400
- */
401
- function isEmptyParagraphGroup(group) {
402
- for (const runGroup of group.runGroups) {
403
- const hasNonEmptyAtom = runGroup.atoms.some((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
404
- if (hasNonEmptyAtom) {
405
- return false;
406
- }
407
- }
408
- return group.runGroups.length > 0;
409
- }
410
- const NO_EXPLICIT_MOVE_MARKERS = { moveFrom: false, moveTo: false };
411
- function collectExplicitMoveMarkers(group) {
412
- let moveFrom = false;
413
- let moveTo = false;
414
- for (const runGroup of group.runGroups) {
415
- for (const atom of runGroup.atoms) {
416
- const tag = atom.contentElement.tagName;
417
- if (tag === 'w:moveFromRangeStart' || tag === 'w:moveFromRangeEnd') {
418
- moveFrom = true;
419
- }
420
- else if (tag === 'w:moveToRangeStart' || tag === 'w:moveToRangeEnd') {
421
- moveTo = true;
422
- }
423
- }
424
- }
425
- return { moveFrom, moveTo };
426
- }
427
- /**
428
- * Build XML for a single paragraph with track changes.
429
- */
430
- function buildParagraphXml(group, author, dateStr, revState) {
431
- const revisionCtx = createRevisionContext({ author, date: dateStr, idState: revState });
432
- // Track empty paragraph statuses for debugging
433
- if (isEmptyParagraphGroup(group)) {
434
- const status = group.runGroups[0]?.atoms[0]?.correlationStatus;
435
- if (status === CorrelationStatus.Inserted) {
436
- debugEmptyParaInserted++;
437
- }
438
- else if (status === CorrelationStatus.Deleted) {
439
- debugEmptyParaDeleted++;
440
- }
441
- else if (status === CorrelationStatus.Equal) {
442
- debugEmptyParaEqual++;
443
- }
444
- else {
445
- debugEmptyParaOther++;
446
- }
447
- // Debug: log the first few empty paragraphs for investigation
448
- const debugLimit = 5;
449
- const totalEmpty = debugEmptyParaInserted + debugEmptyParaDeleted + debugEmptyParaEqual + debugEmptyParaOther;
450
- if (totalEmpty <= debugLimit) {
451
- const atoms = group.runGroups.flatMap(rg => rg.atoms);
452
- const statuses = atoms.map(a => a.correlationStatus).join(', ');
453
- debug('reconstructor', `Empty paragraph #${totalEmpty}: status=${status}, atomCount=${atoms.length}, atomStatuses=[${statuses}]`);
454
- }
455
- }
456
- // Whole-paragraph insert/delete encoding must match Word/Aspose behavior.
457
- //
458
- // IMPORTANT: <w:ins> is not a container for <w:p> in WordprocessingML.
459
- // Aspose encodes a paragraph insertion like:
460
- // <w:p>
461
- // <w:pPr><w:rPr><w:ins .../></w:rPr></w:pPr>
462
- // <w:ins ...><w:r>...</w:r></w:ins>
463
- // </w:p>
464
- //
465
- // That structure both renders in Word and allows Reject All to remove the paragraph
466
- // entirely (instead of leaving behind a stub <w:p> break).
467
- if (isEntireParagraphWithStatus(group, CorrelationStatus.Inserted)) {
468
- const paraId = allocateRevisionId(revState);
469
- const insertedRunXml = paragraphHasHyperlinkAtoms(group)
470
- ? buildWholeParagraphRevisionContent(group, (runs) => wrapSerializedContentWithIns(runs, revisionCtx))
471
- : wrapSerializedContentWithIns(group.runGroups.map((runGroup) => buildRunContentAsPlainRun(runGroup)).join(''), revisionCtx);
472
- const pPrChangeEl = buildPPrChangeElement(group.pPr, revisionCtx);
473
- const parts = [];
474
- parts.push('<w:p>');
475
- parts.push(serializePPrWithParaRevisionMarker(group.pPr, 'w:ins', paraId, author, dateStr, pPrChangeEl));
476
- parts.push(insertedRunXml);
477
- parts.push('</w:p>');
478
- return parts.join('');
479
- }
480
- if (isEntireParagraphWithStatus(group, CorrelationStatus.Deleted)) {
481
- const paraId = allocateRevisionId(revState);
482
- const parts = [];
483
- parts.push('<w:p>');
484
- parts.push(serializePPrWithParaRevisionMarker(group.pPr, 'w:del', paraId, author, dateStr));
485
- parts.push(paragraphHasHyperlinkAtoms(group)
486
- ? buildWholeParagraphRevisionContent(group, (runs) => wrapSerializedContentWithDel(runs, revisionCtx))
487
- : wrapSerializedContentWithDel(group.runGroups.map((runGroup) => buildRunContentAsPlainRun(runGroup)).join(''), revisionCtx));
488
- parts.push('</w:p>');
489
- return parts.join('');
490
- }
491
- // Empty inserted paragraphs — use paragraph-mark revision marker (same as whole-paragraph).
492
- // In OOXML, <w:ins> is NOT a valid container for <w:p>. The correct encoding places the
493
- // marker inside w:pPr > w:rPr.
494
- if (isEmptyParagraphWithStatus(group, CorrelationStatus.Inserted)) {
495
- const paraId = allocateRevisionId(revState);
496
- const pPrChangeEl = buildPPrChangeElement(group.pPr, revisionCtx);
497
- const pPrXml = serializePPrWithParaRevisionMarker(group.pPr, 'w:ins', paraId, author, dateStr, pPrChangeEl);
498
- return `<w:p>${pPrXml}</w:p>`;
499
- }
500
- // Empty deleted paragraphs — use paragraph-mark revision marker.
501
- if (isEmptyParagraphWithStatus(group, CorrelationStatus.Deleted)) {
502
- const paraId = allocateRevisionId(revState);
503
- const pPrXml = serializePPrWithParaRevisionMarker(group.pPr, 'w:del', paraId, author, dateStr);
504
- return `<w:p>${pPrXml}</w:p>`;
505
- }
506
- const parts = [];
507
- parts.push('<w:p>');
508
- // Add paragraph properties
509
- if (group.pPr) {
510
- parts.push(serializeToXml(group.pPr));
511
- }
512
- // Add run groups with track changes, restoring w:hyperlink wrappers when
513
- // the paragraph contains hyperlink atoms (issue #368). Hyperlink-free
514
- // paragraphs keep the legacy per-group emission byte-identical.
515
- const explicitMoveMarkers = collectExplicitMoveMarkers(group);
516
- if (paragraphHasHyperlinkAtoms(group)) {
517
- parts.push(buildRunGroupsWithHyperlinks(group.runGroups, author, dateStr, revState, explicitMoveMarkers));
518
- }
519
- else {
520
- for (const runGroup of group.runGroups) {
521
- const runXml = buildRunGroupXml(runGroup, author, dateStr, revState, explicitMoveMarkers);
522
- parts.push(runXml);
523
- }
524
- }
525
- parts.push('</w:p>');
526
- return parts.join('');
527
- }
528
- /**
529
- * Serialize paragraph properties with a paragraph-level revision marker (w:ins or w:del)
530
- * placed inside w:pPr > w:rPr, per OOXML spec.
531
- *
532
- * DOM-based implementation — replaces the former regex-based approach.
533
- */
534
- function serializePPrWithParaRevisionMarker(pPr, markerTag, id, author, dateStr, pPrChangeEl) {
535
- // Clone pPr or synthesize empty one.
536
- const effectivePPr = pPr ? cloneElement(pPr) : createEl('w:pPr');
537
- // Find or create w:rPr at schema-correct position.
538
- let rPr = findChildByTagName(effectivePPr, 'w:rPr');
539
- if (!rPr) {
540
- rPr = createEl('w:rPr');
541
- const sectPr = findChildByTagName(effectivePPr, 'w:sectPr');
542
- const existingPPrChange = findChildByTagName(effectivePPr, 'w:pPrChange');
543
- const insertBefore = sectPr ?? existingPPrChange ?? null;
544
- if (insertBefore) {
545
- effectivePPr.insertBefore(rPr, insertBefore);
546
- }
547
- else {
548
- effectivePPr.appendChild(rPr);
549
- }
550
- }
551
- // Reuse a pre-existing paragraph-mark marker of the same kind cloned from the
552
- // source pPr (issue #452): CT_ParaRPr allows at most one of each tracked-change
553
- // child, and the source revision's metadata (author/date/id) outranks a
554
- // synthetic duplicate. Either way the marker is placed in its schema-correct
555
- // slot ahead of formatting children.
556
- const existingMarker = findChildByTagName(rPr, markerTag);
557
- const marker = existingMarker ??
558
- createEl(markerTag, {
559
- 'w:id': String(id),
560
- 'w:author': author,
561
- 'w:date': dateStr,
562
- });
563
- placeParagraphMarkRevisionMarker(rPr, marker, markerTag);
564
- // Append pPrChange at end if provided.
565
- if (pPrChangeEl) {
566
- effectivePPr.appendChild(pPrChangeEl);
567
- }
568
- return serializeToXml(effectivePPr);
569
- }
570
- /**
571
- * Returns true if every atom in the paragraph is of the specified status
572
- * (ignoring EMPTY_PARAGRAPH_TAG markers).
573
- */
574
- function isEntireParagraphWithStatus(group, status) {
575
- let sawAnyContent = false;
576
- let sawTargetStatus = false;
577
- for (const runGroup of group.runGroups) {
578
- for (const atom of runGroup.atoms) {
579
- const el = atom.contentElement;
580
- if (el.tagName === EMPTY_PARAGRAPH_TAG)
581
- continue;
582
- sawAnyContent = true;
583
- // A whole-paragraph wrap should still apply even if there are "noise" atoms
584
- // (pure whitespace runs, tabs, breaks) marked Equal due to normalization or
585
- // LCS alignment. Those atoms would otherwise prevent wrapping and Word would
586
- // leave an empty <w:p> stub on Reject All.
587
- const isWhitespaceOnlyText = el.tagName === 'w:t' && ((getLeafText(el) ?? '').trim() === '');
588
- const isWhitespaceAtom = isWhitespaceOnlyText || el.tagName === 'w:tab' || el.tagName === 'w:br' || el.tagName === 'w:cr';
589
- if (atom.correlationStatus === status) {
590
- sawTargetStatus = true;
591
- continue;
592
- }
593
- if (isWhitespaceAtom) {
594
- continue; // ignore for whole-paragraph classification
595
- }
596
- return false;
597
- }
598
- }
599
- // If there's no content at all, let the empty-paragraph handlers deal with it.
600
- // Also require at least one atom with the target status so we don't wrap equal-only paragraphs.
601
- return sawAnyContent && sawTargetStatus;
602
- }
603
- /**
604
- * Build a <w:r> without track-change wrappers. Used when the whole paragraph is already
605
- * wrapped (paragraph-level <w:ins>/<w:del>).
606
- *
607
- * When group.rPr is null, sub-groups atoms by per-atom rPr to prevent formatting bleed.
608
- */
609
- function buildRunContentAsPlainRun(group) {
610
- const contentAtoms = group.atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
611
- if (contentAtoms.length === 0)
612
- return '';
613
- // Paragraph-level markers must sit outside <w:r>; route through the
614
- // marker-aware helper which buffers run atoms and flushes on each marker.
615
- if (groupHasParagraphLevelAtoms(group)) {
616
- return buildRunContentWithParagraphMarkers(group);
617
- }
618
- // If group has explicit rPr, emit a single run
619
- if (group.rPr !== null) {
620
- return buildSingleRun(group.atoms, group.rPr);
621
- }
622
- // No group-level rPr — sub-group by per-atom rPr
623
- const subGroups = subGroupByRPr(contentAtoms);
624
- return subGroups.map(sg => buildSingleRun(sg.atoms, sg.rPr)).join('');
625
- }
626
- /**
627
- * Build XML for a run group with appropriate track changes wrapper.
628
- *
629
- * `explicitMoveMarkers` reports whether the surrounding paragraph's atom
630
- * stream already carries explicit moveFromRange / moveToRange markers; moved
631
- * groups then skip synthetic range emission (see ExplicitMoveMarkers).
632
- */
633
- function buildRunGroupXml(group, author, dateStr, revState, explicitMoveMarkers = NO_EXPLICIT_MOVE_MARKERS) {
634
- const runContent = buildRunContent(group);
635
- // If run content is empty (e.g., only empty paragraph atoms), return empty string
636
- // This avoids generating empty track changes wrappers
637
- if (!runContent) {
638
- return '';
639
- }
640
- switch (group.status) {
641
- case CorrelationStatus.Equal:
642
- case CorrelationStatus.Unknown:
643
- return runContent;
644
- case CorrelationStatus.Inserted:
645
- return wrapWithIns(runContent, author, dateStr, revState);
646
- case CorrelationStatus.Deleted:
647
- return wrapWithDel(runContent, author, dateStr, revState);
648
- case CorrelationStatus.MovedSource:
649
- return wrapWithMoveFrom(runContent, author, dateStr, group.moveName || 'move1', revState, explicitMoveMarkers.moveFrom);
650
- case CorrelationStatus.MovedDestination:
651
- return wrapWithMoveTo(runContent, author, dateStr, group.moveName || 'move1', revState, explicitMoveMarkers.moveTo);
652
- case CorrelationStatus.FormatChanged:
653
- // For format changes, we include the run with rPrChange
654
- return buildFormatChangeRun(group, author, dateStr, revState);
655
- default:
656
- return runContent;
657
- }
658
- }
659
- // Debug counter for atoms processed
660
- let debugAtomCounter = 0;
661
- let debugWtCounter = 0;
662
- /**
663
- * Reset debug counters (for testing).
664
- */
665
- export function resetDebugCounters() {
666
- debugAtomCounter = 0;
667
- debugWtCounter = 0;
668
- }
669
- /**
670
- * Get debug counters (for testing).
671
- */
672
- export function getDebugCounters() {
673
- return { atoms: debugAtomCounter, wt: debugWtCounter };
674
- }
675
- /**
676
- * Sub-group atoms by contiguous rPr — atoms with the same effective rPr
677
- * stay in one sub-group, a change in rPr starts a new sub-group.
678
- */
679
- function subGroupByRPr(atoms) {
680
- if (atoms.length === 0)
681
- return [];
682
- const result = [];
683
- let currentRPr = getEffectiveAtomRPr(atoms[0]);
684
- let currentAtoms = [atoms[0]];
685
- for (let i = 1; i < atoms.length; i++) {
686
- const atomRPr = getEffectiveAtomRPr(atoms[i]);
687
- // Fast path: reference equality or both null
688
- let same = currentRPr === atomRPr;
689
- if (!same && currentRPr === null && atomRPr === null) {
690
- same = true;
691
- }
692
- if (!same) {
693
- same = areRunPropertiesEqual(currentRPr, atomRPr);
694
- }
695
- if (same) {
696
- currentAtoms.push(atoms[i]);
697
- }
698
- else {
699
- result.push({ rPr: currentRPr, atoms: currentAtoms });
700
- currentRPr = atomRPr;
701
- currentAtoms = [atoms[i]];
702
- }
703
- }
704
- result.push({ rPr: currentRPr, atoms: currentAtoms });
705
- return result;
706
- }
707
- /**
708
- * Attribute fingerprint of a w:hyperlink element, used to recognize "the
709
- * same" hyperlink across the original and revised trees (equal/deleted atoms
710
- * reference the original tree's element, inserted atoms the revised tree's).
711
- */
712
- function hyperlinkKey(el) {
713
- const parts = [];
714
- for (let i = 0; i < el.attributes.length; i++) {
715
- const attr = el.attributes.item(i);
716
- if (attr.name.startsWith('xmlns'))
717
- continue;
718
- parts.push(`${attr.name}=${attr.value}`);
719
- }
720
- return parts.sort().join('\u0000');
721
- }
722
- /**
723
- * Resolve the hyperlink wrapper an atom belongs to, preferring the
724
- * original-tree element so the re-emitted r:id resolves against the
725
- * original-based rebuild package: deleted atoms carry original ancestry
726
- * directly; equal atoms (revised tree) reach it via comparisonUnitAtomBefore.
727
- *
728
- * @conformance ECMA-376 edition 5, Part 1 § 17.16.22
729
- * @see https://github.com/UseJunior/safe-docx/issues/368
730
- */
731
- function resolveHyperlinkForAtom(atom) {
732
- const own = nearestHyperlinkAncestor(atom);
733
- if (!own)
734
- return null;
735
- if (atom.sourceDocument === 'original') {
736
- return { element: own, key: hyperlinkKey(own), fromOriginal: true };
737
- }
738
- const before = atom.comparisonUnitAtomBefore;
739
- const beforeHyperlink = before ? nearestHyperlinkAncestor(before) : null;
740
- // Attribute to the original wrapper only when both trees agree on the
741
- // hyperlink's attributes. When they differ (e.g. the revision retargeted
742
- // the link to a new r:id), emitting the original wrapper would pin the
743
- // still-equal link text to the STALE target in the accepted document —
744
- // worse than dropping the wrapper. Such atoms fall through to the
745
- // revised-only policy below instead.
746
- // TODO(#376): the faithful tracked representation of a retargeted link
747
- // is delete-old-link + insert-new-link (what Word emits), which needs the
748
- // hyperlink fingerprint in atom identity so the LCS stops matching text
749
- // across different link targets.
750
- if (beforeHyperlink && hyperlinkKey(beforeHyperlink) === hyperlinkKey(own)) {
751
- return { element: beforeHyperlink, key: hyperlinkKey(beforeHyperlink), fromOriginal: true };
752
- }
753
- // Revised-only attribution (purely inserted hyperlink). Emitting its r:id
754
- // would dangle against the original-based package, so the caller only
755
- // wraps when the hyperlink carries no relationship reference (anchor-only).
756
- return { element: own, key: hyperlinkKey(own), fromOriginal: false };
757
- }
758
- /**
759
- * Whether a resolved hyperlink is safe to re-emit. Original-attributed
760
- * wrappers always are; revised-only wrappers are safe only without an r:id
761
- * (internal anchor links), because the rebuild package ships the ORIGINAL
762
- * document.xml.rels and a revised-only r:id would be a dangling reference
763
- * (Word treats those as a corrupt package). Revised-only r:id hyperlinks
764
- * keep today's behavior — content emitted unwrapped.
765
- */
766
- function isEmittableHyperlink(resolved) {
767
- return resolved.fromOriginal || resolved.element.getAttribute('r:id') === null;
768
- }
769
- /**
770
- * True when any atom in the paragraph sits inside a w:hyperlink. Gates the
771
- * hyperlink-aware emission paths so hyperlink-free paragraphs keep the
772
- * byte-identical legacy output.
773
- */
774
- function paragraphHasHyperlinkAtoms(group) {
775
- return group.runGroups.some((rg) => rg.atoms.some((atom) => nearestHyperlinkAncestor(atom) !== null));
776
- }
777
- /**
778
- * Split a RunGroup into contiguous hyperlink-pure sub-groups.
779
- *
780
- * Moved groups are returned whole: splitting them would emit
781
- * moveFromRangeStart/End once per slice, corrupting the move ranges. A move
782
- * spanning a hyperlink keeps today's unwrapped emission.
783
- */
784
- function splitRunGroupByHyperlink(group) {
785
- if (group.status === CorrelationStatus.MovedSource ||
786
- group.status === CorrelationStatus.MovedDestination) {
787
- return [{ group, hyperlink: null }];
788
- }
789
- const segments = [];
790
- let current = null;
791
- for (const atom of group.atoms) {
792
- // Emit-ability is decided per merged bucket, not per atom: an inserted
793
- // atom inside an otherwise-original hyperlink folds into the adjacent
794
- // original-attributed bucket via the shared key.
795
- const resolved = resolveHyperlinkForAtom(atom);
796
- const key = resolved?.key ?? null;
797
- if (current && (current.hyperlink?.key ?? null) === key) {
798
- current.group.atoms.push(atom);
799
- // Prefer an original-attributed representative within the segment.
800
- if (resolved?.fromOriginal && current.hyperlink && !current.hyperlink.fromOriginal) {
801
- current.hyperlink = resolved;
802
- }
803
- }
804
- else {
805
- current = {
806
- group: { ...group, atoms: [atom] },
807
- hyperlink: resolved,
808
- };
809
- segments.push(current);
810
- }
811
- }
812
- return segments;
813
- }
814
- /**
815
- * Serialize the opening tag of a re-emitted w:hyperlink wrapper, copying the
816
- * source element's attributes verbatim (r:id, w:anchor, w:history, ...).
817
- */
818
- function serializeHyperlinkOpenTag(el) {
819
- const attrs = [];
820
- for (let i = 0; i < el.attributes.length; i++) {
821
- const attr = el.attributes.item(i);
822
- if (attr.name.startsWith('xmlns'))
823
- continue;
824
- attrs.push(` ${attr.name}="${escapeXmlAttr(attr.value)}"`);
825
- }
826
- return `<w:hyperlink${attrs.join('')}>`;
827
- }
828
- /**
829
- * Merge adjacent segments that resolve to the same hyperlink fingerprint, so
830
- * an equal/deleted/inserted sequence inside one link shares one wrapper.
831
- */
832
- function mergeAdjacentHyperlinkSegments(segments) {
833
- const buckets = [];
834
- for (const segment of segments) {
835
- const last = buckets[buckets.length - 1];
836
- if (last && (last.hyperlink?.key ?? null) === (segment.hyperlink?.key ?? null)) {
837
- last.groups.push(segment.group);
838
- if (segment.hyperlink?.fromOriginal && last.hyperlink && !last.hyperlink.fromOriginal) {
839
- last.hyperlink = segment.hyperlink;
840
- }
841
- }
842
- else {
843
- buckets.push({ hyperlink: segment.hyperlink, groups: [segment.group] });
844
- }
845
- }
846
- return buckets;
847
- }
848
- /**
849
- * Emit a paragraph's run groups with w:hyperlink wrappers restored around the
850
- * runs whose atoms came from inside a hyperlink. Track-change wrappers nest
851
- * INSIDE the hyperlink (`<w:hyperlink><w:ins>…`): CT_Hyperlink admits
852
- * EG_RunLevelElts (w:ins / w:del / range markers), while CT_RunTrackChange
853
- * does not admit w:hyperlink.
854
- *
855
- * @conformance ECMA-376 edition 5, Part 1 § 17.16.22
856
- * @see https://github.com/UseJunior/safe-docx/issues/368
857
- */
858
- function buildRunGroupsWithHyperlinks(runGroups, author, dateStr, revState, explicitMoveMarkers = NO_EXPLICIT_MOVE_MARKERS) {
859
- const buckets = mergeAdjacentHyperlinkSegments(runGroups.flatMap(splitRunGroupByHyperlink));
860
- const parts = [];
861
- for (const bucket of buckets) {
862
- const content = bucket.groups
863
- .map((g) => buildRunGroupXml(g, author, dateStr, revState, explicitMoveMarkers))
864
- .join('');
865
- if (!content)
866
- continue;
867
- parts.push(bucket.hyperlink && isEmittableHyperlink(bucket.hyperlink)
868
- ? `${serializeHyperlinkOpenTag(bucket.hyperlink.element)}${content}</w:hyperlink>`
869
- : content);
870
- }
871
- return parts.join('');
872
- }
873
- /**
874
- * Whole-paragraph insert/delete emission with hyperlink wrappers restored.
875
- * Each bucket gets its own revision wrapper so the hyperlink can stay
876
- * OUTSIDE the w:ins / w:del (see buildRunGroupsWithHyperlinks).
877
- */
878
- function buildWholeParagraphRevisionContent(group, wrap) {
879
- const buckets = mergeAdjacentHyperlinkSegments(group.runGroups.flatMap(splitRunGroupByHyperlink));
880
- const parts = [];
881
- for (const bucket of buckets) {
882
- const runs = bucket.groups.map((g) => buildRunContentAsPlainRun(g)).join('');
883
- if (!runs)
884
- continue;
885
- const wrapped = wrap(runs);
886
- parts.push(bucket.hyperlink && isEmittableHyperlink(bucket.hyperlink)
887
- ? `${serializeHyperlinkOpenTag(bucket.hyperlink.element)}${wrapped}</w:hyperlink>`
888
- : wrapped);
889
- }
890
- return parts.join('');
891
- }
892
- /**
893
- * Returns true when any atom in the group is a paragraph-level marker
894
- * (commentRange / bookmark / moveFromRange / moveToRange / perm) that must
895
- * be emitted outside <w:r>.
896
- */
897
- function groupHasParagraphLevelAtoms(group) {
898
- for (const atom of group.atoms) {
899
- if (isParagraphLevelLeaf(atom.contentElement))
900
- return true;
901
- }
902
- return false;
903
- }
904
- /**
905
- * Marker-aware emission for run groups containing paragraph-level atoms.
906
- *
907
- * Walks atoms left-to-right. Run-level atoms accumulate in a buffer; on
908
- * encountering a paragraph-level atom (or end of group) the buffer is flushed
909
- * via subGroupByRPr + buildSingleRun (one <w:r> per contiguous rPr) and the
910
- * marker is emitted as a bare element.
911
- *
912
- * group.rPr is intentionally ignored here — RunGroup.rPr is captured from the
913
- * first atom in groupAtomsByParagraph(), and for moved groups
914
- * shouldStartNewRunGroup() suppresses rPr-based splitting. Always re-deriving
915
- * rPr per atom prevents formatting bleed and bogus rPr inheritance from
916
- * illegally nested markers.
917
- */
918
- function buildRunContentWithParagraphMarkers(group) {
919
- const parts = [];
920
- let runBuffer = [];
921
- const flush = () => {
922
- if (runBuffer.length === 0)
923
- return;
924
- for (const sg of subGroupByRPr(runBuffer)) {
925
- const run = buildSingleRun(sg.atoms, sg.rPr);
926
- if (run)
927
- parts.push(run);
928
- }
929
- runBuffer = [];
930
- };
931
- for (const atom of group.atoms) {
932
- if (atom.contentElement.tagName === EMPTY_PARAGRAPH_TAG)
933
- continue;
934
- if (isParagraphLevelLeaf(atom.contentElement)) {
935
- flush();
936
- parts.push(serializeAtomElement(atom.contentElement));
937
- }
938
- else {
939
- runBuffer.push(atom);
940
- }
941
- }
942
- flush();
943
- return parts.join('');
944
- }
945
- /**
946
- * Build a single <w:r> element from a set of atoms with the given rPr.
947
- * Preserves pendingText coalescing, collapsedFieldAtoms expansion,
948
- * and debug counter increments.
949
- */
950
- function buildSingleRun(atoms, rPr) {
951
- const contentAtoms = atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
952
- if (contentAtoms.length === 0)
953
- return '';
954
- const parts = [];
955
- parts.push('<w:r>');
956
- if (rPr)
957
- parts.push(serializeToXml(rPr));
958
- let pendingText = '';
959
- const flushPendingText = () => {
960
- if (!pendingText)
961
- return;
962
- const escaped = escapeXmlText(pendingText);
963
- const needsPreserve = pendingText.startsWith(' ') ||
964
- pendingText.endsWith(' ') ||
965
- pendingText.includes(' ');
966
- parts.push(needsPreserve
967
- ? `<w:t xml:space="preserve">${escaped}</w:t>`
968
- : `<w:t>${escaped}</w:t>`);
969
- pendingText = '';
970
- };
971
- for (const atom of contentAtoms) {
972
- debugAtomCounter++;
973
- if (atom.collapsedFieldAtoms && atom.collapsedFieldAtoms.length > 0) {
974
- flushPendingText();
975
- for (const fieldAtom of atom.collapsedFieldAtoms) {
976
- parts.push(serializeAtomElement(fieldAtom.contentElement));
977
- }
978
- continue;
979
- }
980
- const el = atom.contentElement;
981
- if (el.tagName === 'w:t') {
982
- pendingText += getLeafText(el) ?? '';
983
- continue;
984
- }
985
- flushPendingText();
986
- parts.push(serializeAtomElement(el));
987
- }
988
- flushPendingText();
989
- parts.push('</w:r>');
990
- return parts.join('');
991
- }
992
- /**
993
- * Serialize an atom's content element to XML string.
994
- */
995
- function serializeAtomElement(element) {
996
- if (element.tagName === 'w:t') {
997
- debugWtCounter++;
998
- // Text element - preserve xml:space if needed
999
- const text = escapeXmlText(getLeafText(element) ?? '');
1000
- if (text.startsWith(' ') || text.endsWith(' ') || text.includes(' ')) {
1001
- return `<w:t xml:space="preserve">${text}</w:t>`;
1002
- }
1003
- else {
1004
- return `<w:t>${text}</w:t>`;
1005
- }
1006
- }
1007
- else if (element.tagName === 'w:br') {
1008
- return '<w:br/>';
1009
- }
1010
- else if (element.tagName === 'w:tab') {
1011
- return '<w:tab/>';
1012
- }
1013
- else if (element.tagName === 'w:cr') {
1014
- return '<w:cr/>';
1015
- }
1016
- else {
1017
- // Other elements (including field chars, instrText) - serialize as-is
1018
- return serializeToXml(element);
1019
- }
1020
- }
1021
- /**
1022
- * Build the content of a run from atoms.
1023
- *
1024
- * Returns empty string if all atoms are empty paragraph markers,
1025
- * which ensures no empty <w:r> elements are generated.
1026
- *
1027
- * When group.rPr is non-null, emits a single <w:r> with that rPr.
1028
- * When group.rPr is null (e.g., after reorderChangeBlocks merges atoms
1029
- * from multiple original RunGroups), sub-groups atoms by their per-atom
1030
- * rPr and emits one <w:r> per sub-group to prevent formatting bleed.
1031
- */
1032
- function buildRunContent(group) {
1033
- // Check if this run group contains only empty paragraph atoms
1034
- const contentAtoms = group.atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
1035
- // If no content atoms, return empty string (don't generate empty run)
1036
- if (contentAtoms.length === 0) {
1037
- return '';
1038
- }
1039
- // Paragraph-level markers must sit outside <w:r>; route through the
1040
- // marker-aware helper which buffers run atoms and flushes on each marker.
1041
- if (groupHasParagraphLevelAtoms(group)) {
1042
- return buildRunContentWithParagraphMarkers(group);
1043
- }
1044
- // If group has explicit rPr, emit a single run
1045
- if (group.rPr !== null) {
1046
- return buildSingleRun(group.atoms, group.rPr);
1047
- }
1048
- // No group-level rPr — sub-group by per-atom rPr
1049
- const subGroups = subGroupByRPr(contentAtoms);
1050
- return subGroups.map(sg => buildSingleRun(sg.atoms, sg.rPr)).join('');
1051
- }
1052
- /**
1053
- * Wrap content with w:ins element.
1054
- */
1055
- function wrapWithIns(content, author, dateStr, revState) {
1056
- return wrapSerializedContentWithIns(content, createRevisionContext({ author, date: dateStr, idState: revState }));
1057
- }
1058
- /**
1059
- * Wrap content with w:del element.
1060
- */
1061
- function wrapWithDel(content, author, dateStr, revState) {
1062
- return wrapSerializedContentWithDel(content, createRevisionContext({ author, date: dateStr, idState: revState }));
1063
- }
1064
- /**
1065
- * Wrap content with w:moveFrom elements.
1066
- *
1067
- * When `suppressRangeMarkers` is true the paragraph's atom stream already
1068
- * carries explicit w:moveFromRangeStart/End markers (re-emitted by
1069
- * buildRunContentWithParagraphMarkers), so only the w:moveFrom wrapper is
1070
- * synthesized — emitting a second range pair would corrupt the move ranges.
1071
- *
1072
- * @see https://github.com/UseJunior/safe-docx/issues/110
1073
- */
1074
- function wrapWithMoveFrom(content, author, dateStr, moveName, revState, suppressRangeMarkers = false) {
1075
- if (suppressRangeMarkers) {
1076
- const moveId = allocateRevisionId(revState);
1077
- const delContent = convertSerializedDeletionContent(content);
1078
- return `<w:moveFrom w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${delContent}</w:moveFrom>`;
1079
- }
1080
- const ids = getMoveRangeIds(revState, moveName);
1081
- const moveId = allocateRevisionId(revState);
1082
- const delContent = convertSerializedDeletionContent(content);
1083
- return (`<w:moveFromRangeStart w:id="${ids.sourceRangeId}" w:name="${moveName}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}"/>` +
1084
- `<w:moveFrom w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${delContent}</w:moveFrom>` +
1085
- `<w:moveFromRangeEnd w:id="${ids.sourceRangeId}"/>`);
1086
- }
1087
- /**
1088
- * Wrap content with w:moveTo elements.
1089
- *
1090
- * When `suppressRangeMarkers` is true the paragraph's atom stream already
1091
- * carries explicit w:moveToRangeStart/End markers, so only the w:moveTo
1092
- * wrapper is synthesized (see wrapWithMoveFrom).
1093
- */
1094
- function wrapWithMoveTo(content, author, dateStr, moveName, revState, suppressRangeMarkers = false) {
1095
- if (suppressRangeMarkers) {
1096
- const moveId = allocateRevisionId(revState);
1097
- return `<w:moveTo w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${content}</w:moveTo>`;
1098
- }
1099
- const ids = getMoveRangeIds(revState, moveName);
1100
- const moveId = allocateRevisionId(revState);
1101
- return (`<w:moveToRangeStart w:id="${ids.destRangeId}" w:name="${moveName}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}"/>` +
1102
- `<w:moveTo w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${content}</w:moveTo>` +
1103
- `<w:moveToRangeEnd w:id="${ids.destRangeId}"/>`);
1104
- }
1105
- /**
1106
- * Build run with format change tracking (w:rPrChange).
1107
- */
1108
- function buildFormatChangeRun(group, author, dateStr, revState) {
1109
- const parts = [];
1110
- parts.push('<w:r>');
1111
- // Build rPr with rPrChange
1112
- const effectiveRPr = group.rPr ?? group.atoms[0]?.rPr ?? null;
1113
- if (effectiveRPr || group.atoms[0]?.formatChange) {
1114
- parts.push('<w:rPr>');
1115
- // Current properties
1116
- if (effectiveRPr) {
1117
- for (const child of childElements(effectiveRPr)) {
1118
- if (child.tagName !== 'w:rPrChange') {
1119
- parts.push(serializeToXml(child));
1120
- }
1121
- }
1122
- }
1123
- // Add rPrChange with old properties (wrapped in w:rPr per OOXML spec).
1124
- // Kept as the original per-child serialization (NOT delegated to
1125
- // buildRPrChangeElement) to preserve byte-identical output: xmldom emits
1126
- // inline `xmlns:w="..."` declarations when serializing detached children,
1127
- // and downstream consumers may pin on that exact serialized form. The
1128
- // DOM-aware buildRPrChangeElement helper exists for new primitive code
1129
- // paths (#136 onward).
1130
- const formatChange = group.atoms[0]?.formatChange;
1131
- if (formatChange?.oldRunProperties) {
1132
- const id = allocateRevisionId(revState);
1133
- parts.push(`<w:rPrChange w:id="${id}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">`);
1134
- parts.push('<w:rPr>');
1135
- for (const child of childElements(formatChange.oldRunProperties)) {
1136
- parts.push(serializeToXml(child));
1137
- }
1138
- parts.push('</w:rPr>');
1139
- parts.push('</w:rPrChange>');
1140
- }
1141
- parts.push('</w:rPr>');
1142
- }
1143
- // Add atom content
1144
- for (const atom of group.atoms) {
1145
- const element = atom.contentElement;
1146
- if (element.tagName === 'w:t') {
1147
- const text = escapeXmlText(getLeafText(element) ?? '');
1148
- if (text.startsWith(' ') || text.endsWith(' ') || text.includes(' ')) {
1149
- parts.push(`<w:t xml:space="preserve">${text}</w:t>`);
1150
- }
1151
- else {
1152
- parts.push(`<w:t>${text}</w:t>`);
1153
- }
1154
- }
1155
- else {
1156
- parts.push(serializeToXml(element));
1157
- }
1158
- }
1159
- parts.push('</w:r>');
1160
- return parts.join('');
1161
- }
1162
- /**
1163
- * Parse the original document body into a structural map.
1164
- *
1165
- * Recursively finds ALL <w:p> elements in document order, regardless of
1166
- * wrapper (tables, SDTs, customXml, nested tables, etc.). This matches
1167
- * the atomizer's recursive tree walk in atomizer.ts.
1168
- */
1169
- function parseOriginalBodyStructure(originalXml) {
1170
- const doc = parseXml(originalXml);
1171
- const bodies = doc.getElementsByTagName('w:body');
1172
- if (!bodies.length) {
1173
- throw new Error('Could not find w:body in document');
1174
- }
1175
- const body = bodies[0];
1176
- // getElementsByTagName returns ALL descendants in document order —
1177
- // this naturally recurses through tables, SDTs, customXml, nested tables, etc.
1178
- const paragraphs = body.getElementsByTagName('w:p');
1179
- const slots = [];
1180
- for (let i = 0; i < paragraphs.length; i++) {
1181
- const el = paragraphs[i];
1182
- slots.push({ index: i, element: el, parent: el.parentNode });
1183
- }
1184
- return { doc, body, slots };
1185
- }
1186
- /**
1187
- * Determine if a ParagraphGroup is "rooted" (maps to an original paragraph slot)
1188
- * or "purely inserted" (new content with no original counterpart).
1189
- *
1190
- * A group is rooted if ANY run group has a status other than Inserted or
1191
- * MovedDestination — i.e., it contains content from the original document.
1192
- * This correctly handles Equal, Deleted, MovedSource, and FormatChanged.
1193
- */
1194
- function isRootedGroup(group) {
1195
- return group.runGroups.some((rg) => rg.status !== CorrelationStatus.Inserted &&
1196
- rg.status !== CorrelationStatus.MovedDestination);
1197
- }
1198
- /**
1199
- * Build the final document preserving original body structure.
1200
- *
1201
- * Instead of replacing <w:body> content with flat paragraphs, this uses the
1202
- * original body DOM as a scaffold: rooted paragraphs replace their corresponding
1203
- * <w:p> slots, inserted paragraphs are placed adjacent to their context, and
1204
- * all structural wrappers (tables, SDTs, etc.) are preserved.
1205
- */
1206
- function buildDocumentPreservingStructure(originalXml, paragraphXmls, paragraphGroups, allocateRevisionId) {
1207
- const { doc, body, slots } = parseOriginalBodyStructure(originalXml);
1208
- let slotCursor = 0;
1209
- let lastEmittedNode = null;
1210
- // Find body-level <w:sectPr> (must stay as last child of body)
1211
- const bodyChildren = childElements(body);
1212
- const finalSectPr = bodyChildren.length > 0 &&
1213
- bodyChildren[bodyChildren.length - 1].tagName === 'w:sectPr'
1214
- ? bodyChildren[bodyChildren.length - 1]
1215
- : null;
1216
- for (let i = 0; i < paragraphGroups.length; i++) {
1217
- const group = paragraphGroups[i];
1218
- const paraXml = paragraphXmls[i];
1219
- // Parse the reconstructed paragraph XML into a DOM node
1220
- const fragDoc = parseXml(`<__wrap xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml" xmlns:w15="http://schemas.microsoft.com/office/word/2012/wordml" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships" xmlns:mc="http://schemas.openxmlformats.org/markup-compatibility/2006">${paraXml}</__wrap>`);
1221
- const newNode = doc.importNode(fragDoc.documentElement.firstChild, true);
1222
- if (isRootedGroup(group)) {
1223
- // Replace the corresponding original <w:p> slot
1224
- if (slotCursor < slots.length) {
1225
- const slot = slots[slotCursor];
1226
- slot.parent.replaceChild(newNode, slot.element);
1227
- lastEmittedNode = newNode;
1228
- slotCursor++;
1229
- }
1230
- else {
1231
- // More rooted paragraphs than slots — append to body (before sectPr)
1232
- if (finalSectPr) {
1233
- body.insertBefore(newNode, finalSectPr);
1234
- }
1235
- else {
1236
- body.appendChild(newNode);
1237
- }
1238
- lastEmittedNode = newNode;
1239
- }
1240
- }
1241
- else {
1242
- // Inserted paragraph — place in context
1243
- if (lastEmittedNode) {
1244
- // Insert after the previous paragraph in the same parent
1245
- const parent = lastEmittedNode.parentNode;
1246
- const nextSibling = lastEmittedNode.nextSibling;
1247
- // Guard: never insert after body-level <w:sectPr>
1248
- if (nextSibling === finalSectPr && parent === body) {
1249
- parent.insertBefore(newNode, finalSectPr);
1250
- }
1251
- else {
1252
- parent.insertBefore(newNode, nextSibling);
1253
- }
1254
- lastEmittedNode = newNode;
1255
- }
1256
- else if (slotCursor < slots.length) {
1257
- // No previous node — insert before the next rooted slot
1258
- const nextSlot = slots[slotCursor];
1259
- nextSlot.parent.insertBefore(newNode, nextSlot.element);
1260
- lastEmittedNode = newNode;
1261
- }
1262
- else {
1263
- // No context — append to body (before sectPr)
1264
- if (finalSectPr) {
1265
- body.insertBefore(newNode, finalSectPr);
1266
- }
1267
- else {
1268
- body.appendChild(newNode);
1269
- }
1270
- lastEmittedNode = newNode;
1271
- }
1272
- }
1273
- }
1274
- // Remove any leftover original <w:p> slots that weren't consumed
1275
- // (this happens when the original has more paragraphs than the merged result)
1276
- for (let i = slotCursor; i < slots.length; i++) {
1277
- const slot = slots[i];
1278
- slot.parent.removeChild(slot.element);
1279
- }
1280
- // Strip inter-paragraph bookmark/comment/move-range/permission markers from
1281
- // the scaffold. These are bookmarkStart/End, commentRangeStart/End,
1282
- // moveFromRange*/moveToRange*, and permStart/End elements that were siblings
1283
- // of <w:p> in the original body. The paragraph rebuilder handles its own
1284
- // bookmark logic, so keeping these orphaned markers causes unmatched
1285
- // bookmark IDs. Body-level move-range markers are likewise scaffold
1286
- // remnants: in-paragraph markers travel through the atom stream, and
1287
- // detected moves synthesize fresh range pairs inside the reconstructed
1288
- // paragraphs, so a leftover body-level pair would either dangle or double an
1289
- // emitted range.
1290
- //
1291
- // Comment range markers are treated differently: a sibling-level
1292
- // commentRangeStart/End is the legitimate shape for a comment range that
1293
- // spans whole paragraphs, and such markers never enter the atom stream
1294
- // (see isParagraphLevelLeaf in atomizer.ts), so nothing re-emits them.
1295
- // Stripping them unconditionally destroys multi-paragraph comment ranges
1296
- // (issue #103). Instead, strip a sibling-level comment range marker only
1297
- // when its counterpart (same w:id) is absent from the rebuilt body —
1298
- // i.e., it is a genuinely orphaned scaffold remnant.
1299
- const SCAFFOLD_STRIP_TAGS = new Set([
1300
- 'w:bookmarkStart', 'w:bookmarkEnd',
1301
- 'w:commentRangeStart', 'w:commentRangeEnd',
1302
- 'w:moveFromRangeStart', 'w:moveFromRangeEnd',
1303
- 'w:moveToRangeStart', 'w:moveToRangeEnd',
1304
- 'w:permStart', 'w:permEnd',
1305
- ]);
1306
- const COMMENT_RANGE_TAGS = new Set(['w:commentRangeStart', 'w:commentRangeEnd']);
1307
- const commentRangeStartIds = new Set();
1308
- const commentRangeEndIds = new Set();
1309
- for (const el of Array.from(body.getElementsByTagName('*'))) {
1310
- const id = el.getAttribute('w:id');
1311
- if (id == null)
1312
- continue;
1313
- if (el.tagName === 'w:commentRangeStart')
1314
- commentRangeStartIds.add(id);
1315
- else if (el.tagName === 'w:commentRangeEnd')
1316
- commentRangeEndIds.add(id);
1317
- }
1318
- const toRemove = [];
1319
- for (const el of Array.from(body.getElementsByTagName('*'))) {
1320
- if (SCAFFOLD_STRIP_TAGS.has(el.tagName) && el.parentNode) {
1321
- // Only strip if NOT inside a reconstructed <w:p> (i.e., it's a scaffold remnant)
1322
- let insideParagraph = false;
1323
- let ancestor = el.parentNode;
1324
- while (ancestor && ancestor !== body) {
1325
- if (ancestor.tagName === 'w:p') {
1326
- insideParagraph = true;
1327
- break;
1328
- }
1329
- ancestor = ancestor.parentNode;
1330
- }
1331
- if (insideParagraph)
1332
- continue;
1333
- if (COMMENT_RANGE_TAGS.has(el.tagName)) {
1334
- const id = el.getAttribute('w:id');
1335
- const counterpartIds = el.tagName === 'w:commentRangeStart'
1336
- ? commentRangeEndIds
1337
- : commentRangeStartIds;
1338
- if (id != null && counterpartIds.has(id))
1339
- continue;
1340
- }
1341
- toRemove.push(el);
1342
- }
1343
- }
1344
- for (const el of toRemove) {
1345
- el.parentNode.removeChild(el);
1346
- }
1347
- // Balance bookmarks and enforce consumer-compatibility invariants on the
1348
- // rebuilt body. This dedupes bookmark Names/IDs, hoists bookmarkStart/End
1349
- // out of <w:ins>/<w:del> wrappers (so they survive accept/reject), and
1350
- // synthesizes recovery markers for orphaned starts/ends. Mirrors the
1351
- // post-processing applied in inplace mode (inPlaceModifier.ts).
1352
- enforceConsumerCompatibility(body, allocateRevisionId);
1353
- // Serialize modified body and splice back into original envelope
1354
- const serializer = new XMLSerializer();
1355
- let newBodyXml = serializer.serializeToString(body);
1356
- // Strip redundant xmlns:w declarations from inner elements.
1357
- // XMLSerializer adds xmlns:w="..." on imported paragraph/rPr nodes because
1358
- // they were parsed in a separate fragment document. These redundant
1359
- // redeclarations are valid XML but confuse some OOXML consumers (Pages,
1360
- // Google Docs) and prevent them from rendering tracked changes.
1361
- // The w: namespace is already declared on the document root element.
1362
- const W_NS_DECL = ' xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"';
1363
- // Keep the first declaration (on <w:body>) but remove all others
1364
- let firstFound = false;
1365
- newBodyXml = newBodyXml.replace(new RegExp(W_NS_DECL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'), 'g'), (match) => {
1366
- if (!firstFound) {
1367
- firstFound = true;
1368
- return match;
1369
- }
1370
- return '';
1371
- });
1372
- // Replace original body in the full document string
1373
- const bodyRegex = /<w:body[^>]*>[\s\S]*?<\/w:body>/;
1374
- return originalXml.replace(bodyRegex, newBodyXml);
1375
- }
1376
- /**
1377
- * Build the final document by replacing body content (legacy flat mode).
1378
- *
1379
- * Note: sectPr elements are NOT extracted and appended separately because:
1380
- * 1. Section properties inside pPr elements are already preserved in the reconstructed paragraphs
1381
- * 2. The regex to extract "final sectPr" was incorrectly matching sectPr inside pPr elements
1382
- * and capturing large amounts of body content, causing duplicate text.
1383
- *
1384
- * @deprecated Use buildDocumentPreservingStructure instead. Retained as fallback.
1385
- */
1386
- export function buildDocument(originalXml, paragraphXmls) {
1387
- // Extract document structure
1388
- const bodyMatch = originalXml.match(/(<w:body[^>]*>)([\s\S]*?)(<\/w:body>)/);
1389
- if (!bodyMatch) {
1390
- throw new Error('Could not find w:body in document');
1391
- }
1392
- const beforeBody = originalXml.slice(0, originalXml.indexOf(bodyMatch[0]));
1393
- const bodyOpenTag = bodyMatch[1];
1394
- const bodyCloseTag = bodyMatch[3];
1395
- const afterBody = originalXml.slice(originalXml.indexOf(bodyMatch[0]) + bodyMatch[0].length);
1396
- // Build new body (no separate sectPr extraction - it's in the paragraphs' pPr)
1397
- const newBodyContent = paragraphXmls.join('\n');
1398
- return beforeBody + bodyOpenTag + '\n' + newBodyContent + '\n' + bodyCloseTag + afterBody;
1399
- }
1400
- /**
1401
- * Escape XML text content.
1402
- */
1403
- function escapeXmlText(text) {
1404
- return text
1405
- .replace(/&/g, '&amp;')
1406
- .replace(/</g, '&lt;')
1407
- .replace(/>/g, '&gt;');
1408
- }
1409
- /**
1410
- * Count statistics from merged atoms.
1411
- */
1412
- export function computeReconstructionStats(mergedAtoms) {
1413
- let insertions = 0;
1414
- let deletions = 0;
1415
- let moves = 0;
1416
- let formatChanges = 0;
1417
- const paragraphs = new Set();
1418
- for (const atom of mergedAtoms) {
1419
- // Count paragraph
1420
- const pAncestor = findAncestorByTag(atom, 'w:p');
1421
- if (pAncestor) {
1422
- paragraphs.add(pAncestor);
1423
- }
1424
- // Count by status
1425
- switch (atom.correlationStatus) {
1426
- case CorrelationStatus.Inserted:
1427
- insertions++;
1428
- break;
1429
- case CorrelationStatus.Deleted:
1430
- deletions++;
1431
- break;
1432
- case CorrelationStatus.MovedSource:
1433
- case CorrelationStatus.MovedDestination:
1434
- moves++;
1435
- break;
1436
- case CorrelationStatus.FormatChanged:
1437
- formatChanges++;
1438
- break;
1439
- }
1440
- }
1441
- return {
1442
- paragraphs: paragraphs.size,
1443
- insertions,
1444
- deletions,
1445
- moves: Math.floor(moves / 2), // Source and destination counted separately
1446
- formatChanges,
1447
- };
1448
- }
1449
- //# sourceMappingURL=documentReconstructor.js.map