@usejunior/docx-core 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/README.md +18 -83
  2. package/dist/.tsbuildinfo +1 -1
  3. package/dist/cli/conformance-adapter.d.ts +14 -0
  4. package/dist/cli/conformance-adapter.d.ts.map +1 -1
  5. package/dist/cli/conformance-adapter.js +104 -12
  6. package/dist/cli/conformance-adapter.js.map +1 -1
  7. package/dist/footnotes.d.ts +7 -5
  8. package/dist/footnotes.d.ts.map +1 -1
  9. package/dist/footnotes.js +7 -5
  10. package/dist/footnotes.js.map +1 -1
  11. package/dist/generated/ecma-376-vocabulary.d.ts +93 -0
  12. package/dist/generated/ecma-376-vocabulary.d.ts.map +1 -0
  13. package/dist/generated/ecma-376-vocabulary.js +87 -0
  14. package/dist/generated/ecma-376-vocabulary.js.map +1 -0
  15. package/dist/generation/compile.js +2 -2
  16. package/dist/generation/compile.js.map +1 -1
  17. package/dist/generation/emit/comments-part.d.ts +2 -0
  18. package/dist/generation/emit/comments-part.d.ts.map +1 -1
  19. package/dist/generation/emit/comments-part.js +2 -0
  20. package/dist/generation/emit/comments-part.js.map +1 -1
  21. package/dist/generation/emit/paragraph.d.ts +3 -0
  22. package/dist/generation/emit/paragraph.d.ts.map +1 -1
  23. package/dist/generation/emit/paragraph.js +3 -0
  24. package/dist/generation/emit/paragraph.js.map +1 -1
  25. package/dist/generation/emit/run.d.ts.map +1 -1
  26. package/dist/generation/emit/run.js +6 -1
  27. package/dist/generation/emit/run.js.map +1 -1
  28. package/dist/generation/emit/settings-part.d.ts +12 -3
  29. package/dist/generation/emit/settings-part.d.ts.map +1 -1
  30. package/dist/generation/emit/settings-part.js +23 -5
  31. package/dist/generation/emit/settings-part.js.map +1 -1
  32. package/dist/generation/ordering.d.ts +3 -1
  33. package/dist/generation/ordering.d.ts.map +1 -1
  34. package/dist/generation/ordering.js +3 -1
  35. package/dist/generation/ordering.js.map +1 -1
  36. package/dist/generation/schema-enum-domains.d.ts +27 -0
  37. package/dist/generation/schema-enum-domains.d.ts.map +1 -0
  38. package/dist/generation/schema-enum-domains.js +69 -0
  39. package/dist/generation/schema-enum-domains.js.map +1 -0
  40. package/dist/generation/structural-checks.js +7 -1
  41. package/dist/generation/structural-checks.js.map +1 -1
  42. package/dist/generation/validate-spec.d.ts.map +1 -1
  43. package/dist/generation/validate-spec.js +149 -31
  44. package/dist/generation/validate-spec.js.map +1 -1
  45. package/dist/index.d.ts +7 -24
  46. package/dist/index.d.ts.map +1 -1
  47. package/dist/index.js +7 -46
  48. package/dist/index.js.map +1 -1
  49. package/dist/primitives/accept_ai_edits.d.ts +87 -0
  50. package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
  51. package/dist/primitives/accept_ai_edits.js +253 -0
  52. package/dist/primitives/accept_ai_edits.js.map +1 -0
  53. package/dist/primitives/accept_changes.d.ts +36 -3
  54. package/dist/primitives/accept_changes.d.ts.map +1 -1
  55. package/dist/primitives/accept_changes.js +67 -31
  56. package/dist/primitives/accept_changes.js.map +1 -1
  57. package/dist/primitives/bookmarks.d.ts +44 -0
  58. package/dist/primitives/bookmarks.d.ts.map +1 -1
  59. package/dist/primitives/bookmarks.js +149 -11
  60. package/dist/primitives/bookmarks.js.map +1 -1
  61. package/dist/primitives/document.d.ts +53 -1
  62. package/dist/primitives/document.d.ts.map +1 -1
  63. package/dist/primitives/document.js +146 -2
  64. package/dist/primitives/document.js.map +1 -1
  65. package/dist/primitives/dom-helpers.d.ts.map +1 -1
  66. package/dist/primitives/dom-helpers.js +7 -2
  67. package/dist/primitives/dom-helpers.js.map +1 -1
  68. package/dist/primitives/index.d.ts +2 -1
  69. package/dist/primitives/index.d.ts.map +1 -1
  70. package/dist/primitives/index.js +2 -1
  71. package/dist/primitives/index.js.map +1 -1
  72. package/dist/primitives/layout.d.ts.map +1 -1
  73. package/dist/primitives/layout.js +41 -1
  74. package/dist/primitives/layout.js.map +1 -1
  75. package/dist/primitives/namespaces.d.ts +2 -0
  76. package/dist/primitives/namespaces.d.ts.map +1 -1
  77. package/dist/primitives/namespaces.js +2 -0
  78. package/dist/primitives/namespaces.js.map +1 -1
  79. package/dist/primitives/reject_changes.d.ts +21 -2
  80. package/dist/primitives/reject_changes.d.ts.map +1 -1
  81. package/dist/primitives/reject_changes.js +71 -31
  82. package/dist/primitives/reject_changes.js.map +1 -1
  83. package/dist/primitives/relationships.d.ts +33 -0
  84. package/dist/primitives/relationships.d.ts.map +1 -1
  85. package/dist/primitives/relationships.js +84 -1
  86. package/dist/primitives/relationships.js.map +1 -1
  87. package/dist/primitives/sectPrAudit.d.ts +11 -2
  88. package/dist/primitives/sectPrAudit.d.ts.map +1 -1
  89. package/dist/primitives/sectPrAudit.js +148 -23
  90. package/dist/primitives/sectPrAudit.js.map +1 -1
  91. package/dist/primitives/track-changes-emitter.d.ts +4 -0
  92. package/dist/primitives/track-changes-emitter.d.ts.map +1 -1
  93. package/dist/primitives/track-changes-emitter.js +5 -2
  94. package/dist/primitives/track-changes-emitter.js.map +1 -1
  95. package/dist/primitives/validate_ai_revisions.d.ts.map +1 -1
  96. package/dist/primitives/validate_ai_revisions.js +13 -5
  97. package/dist/primitives/validate_ai_revisions.js.map +1 -1
  98. package/dist/primitives/zip.d.ts.map +1 -1
  99. package/dist/primitives/zip.js +6 -2
  100. package/dist/primitives/zip.js.map +1 -1
  101. package/dist/shared/field-structure.d.ts +6 -0
  102. package/dist/shared/field-structure.d.ts.map +1 -1
  103. package/dist/shared/field-structure.js +6 -0
  104. package/dist/shared/field-structure.js.map +1 -1
  105. package/package.json +4 -8
  106. package/dist/atomizer.d.ts +0 -273
  107. package/dist/atomizer.d.ts.map +0 -1
  108. package/dist/atomizer.js +0 -1002
  109. package/dist/atomizer.js.map +0 -1
  110. package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
  111. package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
  112. package/dist/baselines/atomizer/atomLcs.js +0 -376
  113. package/dist/baselines/atomizer/atomLcs.js.map +0 -1
  114. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
  115. package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
  116. package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
  117. package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
  118. package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
  119. package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
  120. package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
  121. package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
  122. package/dist/baselines/atomizer/debug.d.ts +0 -41
  123. package/dist/baselines/atomizer/debug.d.ts.map +0 -1
  124. package/dist/baselines/atomizer/debug.js +0 -85
  125. package/dist/baselines/atomizer/debug.js.map +0 -1
  126. package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
  127. package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
  128. package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
  129. package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
  130. package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
  131. package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
  132. package/dist/baselines/atomizer/formattingFidelity.js +0 -449
  133. package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
  134. package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
  135. package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
  136. package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
  137. package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
  138. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
  139. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
  140. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
  141. package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
  142. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
  143. package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
  144. package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
  145. package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
  146. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
  147. package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
  148. package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
  149. package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
  150. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
  151. package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
  152. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
  153. package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
  154. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
  155. package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
  156. package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
  157. package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
  158. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
  159. package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
  160. package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
  161. package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
  162. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
  163. package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
  164. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
  165. package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
  166. package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
  167. package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
  168. package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
  169. package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
  170. package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
  171. package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
  172. package/dist/baselines/atomizer/numberingIntegration.js +0 -209
  173. package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
  174. package/dist/baselines/atomizer/pipeline.d.ts +0 -103
  175. package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
  176. package/dist/baselines/atomizer/pipeline.js +0 -1160
  177. package/dist/baselines/atomizer/pipeline.js.map +0 -1
  178. package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
  179. package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
  180. package/dist/baselines/atomizer/premergeRuns.js +0 -153
  181. package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
  182. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
  183. package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
  184. package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
  185. package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
  186. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
  187. package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
  188. package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
  189. package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
  190. package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
  191. package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
  192. package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
  193. package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
  194. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
  195. package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
  196. package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
  197. package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
  198. package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
  199. package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
  200. package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
  201. package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
  202. package/dist/cli/compare-two.d.ts +0 -28
  203. package/dist/cli/compare-two.d.ts.map +0 -1
  204. package/dist/cli/compare-two.js +0 -112
  205. package/dist/cli/compare-two.js.map +0 -1
  206. package/dist/cli/index.d.ts +0 -3
  207. package/dist/cli/index.d.ts.map +0 -1
  208. package/dist/cli/index.js +0 -25
  209. package/dist/cli/index.js.map +0 -1
  210. package/dist/compare-types.d.ts +0 -197
  211. package/dist/compare-types.d.ts.map +0 -1
  212. package/dist/compare-types.js +0 -2
  213. package/dist/compare-types.js.map +0 -1
  214. package/dist/format-detection.d.ts +0 -120
  215. package/dist/format-detection.d.ts.map +0 -1
  216. package/dist/format-detection.js +0 -339
  217. package/dist/format-detection.js.map +0 -1
  218. package/dist/move-detection.d.ts +0 -211
  219. package/dist/move-detection.d.ts.map +0 -1
  220. package/dist/move-detection.js +0 -390
  221. package/dist/move-detection.js.map +0 -1
@@ -1,1160 +0,0 @@
1
- /**
2
- * Atomizer Pipeline
3
- *
4
- * Main orchestration for the atomizer-based document comparison.
5
- * Integrates atomization, LCS comparison, move detection, format detection,
6
- * and document reconstruction.
7
- */
8
- import { XMLSerializer } from '@xmldom/xmldom';
9
- import { parseXml } from '../../primitives/xml.js';
10
- import { DocxArchive } from '../../shared/docx/DocxArchive.js';
11
- import { DEFAULT_MOVE_DETECTION_SETTINGS, DEFAULT_FORMAT_DETECTION_SETTINGS, CorrelationStatus, } from '../../core-types.js';
12
- import { atomizeTree, assignParagraphIndices } from '../../atomizer.js';
13
- import { detectMovesInAtomList } from '../../move-detection.js';
14
- import { detectFormatChangesInAtomList } from '../../format-detection.js';
15
- import { parseDocumentXml, findBody, backfillParentReferences, } from './xmlToWmlElement.js';
16
- import { findAllByTagName, getLeafText } from '../../primitives/index.js';
17
- import { createMergedAtomList, assignUnifiedParagraphIndices, } from './atomLcs.js';
18
- import { hierarchicalCompare, markHierarchicalCorrelationStatus, } from './hierarchicalLcs.js';
19
- import { reconstructDocument, computeReconstructionStats, } from './documentReconstructor.js';
20
- import { modifyRevisedDocument, ContainerResolutionError } from './inPlaceModifier.js';
21
- import { acceptAllChanges, rejectAllChanges, extractTextWithParagraphs, compareTexts, } from './trackChangesAcceptorAst.js';
22
- import { virtualizeNumberingLabels, DEFAULT_NUMBERING_OPTIONS, } from './numberingIntegration.js';
23
- import { premergeAdjacentRuns } from './premergeRuns.js';
24
- export { hasFldCharInsideDel, validateFieldStructure, } from '../../shared/field-structure.js';
25
- import { hasFldCharInsideDel, validateFieldStructure, } from '../../shared/field-structure.js';
26
- import { AUXILIARY_PARTS, parseEntries, renumberCollidingAuxiliaryIds, restampCollidingCommentParaIds, } from './auxiliaryIdCollision.js';
27
- import { maybeCaptureEmittedDocumentXml } from '../../primitives/schema-corpus-capture.js';
28
- function arraysEqual(a, b) {
29
- if (a.length !== b.length)
30
- return false;
31
- for (let i = 0; i < a.length; i++) {
32
- if (a[i] !== b[i])
33
- return false;
34
- }
35
- return true;
36
- }
37
- function collectReferencedBookmarkNames(root) {
38
- const refs = new Set();
39
- const refRegex = /\b(?:PAGEREF|REF)\s+([^\s\\]+)/g;
40
- for (const node of findAllByTagName(root, 'w:instrText')) {
41
- const instr = getLeafText(node) ?? '';
42
- for (const match of instr.matchAll(refRegex)) {
43
- const name = match[1]?.trim();
44
- if (name)
45
- refs.add(name);
46
- }
47
- }
48
- return Array.from(refs).sort();
49
- }
50
- function collectBookmarkDiagnostics(documentXml) {
51
- const root = parseDocumentXml(documentXml);
52
- const startSet = new Set();
53
- const endSet = new Set();
54
- const startNameSet = new Set();
55
- const duplicateStartSet = new Set();
56
- const duplicateEndSet = new Set();
57
- const duplicateStartNameSet = new Set();
58
- for (const node of findAllByTagName(root, 'w:bookmarkStart')) {
59
- const id = node.getAttribute('w:id');
60
- if (!id)
61
- continue;
62
- if (startSet.has(id))
63
- duplicateStartSet.add(id);
64
- else
65
- startSet.add(id);
66
- const name = node.getAttribute('w:name');
67
- if (name) {
68
- if (startNameSet.has(name))
69
- duplicateStartNameSet.add(name);
70
- else
71
- startNameSet.add(name);
72
- }
73
- }
74
- for (const node of findAllByTagName(root, 'w:bookmarkEnd')) {
75
- const id = node.getAttribute('w:id');
76
- if (!id)
77
- continue;
78
- if (endSet.has(id))
79
- duplicateEndSet.add(id);
80
- else
81
- endSet.add(id);
82
- }
83
- const startIds = Array.from(startSet).sort();
84
- const endIds = Array.from(endSet).sort();
85
- const startNames = Array.from(startNameSet).sort();
86
- const referencedBookmarkNames = collectReferencedBookmarkNames(root);
87
- const unresolvedReferenceNames = referencedBookmarkNames
88
- .filter((name) => !startNameSet.has(name))
89
- .sort();
90
- const unmatchedStartIds = startIds.filter((id) => !endSet.has(id));
91
- const unmatchedEndIds = endIds.filter((id) => !startSet.has(id));
92
- return {
93
- startIds,
94
- endIds,
95
- startNames,
96
- duplicateStartNames: Array.from(duplicateStartNameSet).sort(),
97
- referencedBookmarkNames,
98
- unresolvedReferenceNames,
99
- duplicateStartIds: Array.from(duplicateStartSet).sort(),
100
- duplicateEndIds: Array.from(duplicateEndSet).sort(),
101
- unmatchedStartIds,
102
- unmatchedEndIds,
103
- };
104
- }
105
- /**
106
- * Bookmark round-trip safety is semantic, not byte/ID exact:
107
- * - Bookmark IDs may be renumbered by reconstruction/Word and still be valid.
108
- * - Bookmark names and field-reference targets must stay intact.
109
- * - Structural integrity (balanced, no duplicates) must remain intact.
110
- */
111
- function bookmarkDiagnosticsSemanticallyEqual(expected, actual) {
112
- return (arraysEqual(expected.startNames, actual.startNames) &&
113
- arraysEqual(expected.duplicateStartNames, actual.duplicateStartNames) &&
114
- arraysEqual(expected.referencedBookmarkNames, actual.referencedBookmarkNames) &&
115
- arraysEqual(expected.unresolvedReferenceNames, actual.unresolvedReferenceNames) &&
116
- arraysEqual(expected.duplicateStartIds, actual.duplicateStartIds) &&
117
- arraysEqual(expected.duplicateEndIds, actual.duplicateEndIds) &&
118
- arraysEqual(expected.unmatchedStartIds, actual.unmatchedStartIds) &&
119
- arraysEqual(expected.unmatchedEndIds, actual.unmatchedEndIds));
120
- }
121
- function diffIds(expected, actual) {
122
- const expectedSet = new Set(expected);
123
- const actualSet = new Set(actual);
124
- const missing = expected.filter((id) => !actualSet.has(id));
125
- const unexpected = actual.filter((id) => !expectedSet.has(id));
126
- return { missing, unexpected };
127
- }
128
- function buildTextMismatchDetails(expectedText, actualText) {
129
- const comparison = compareTexts(expectedText, actualText);
130
- const expectedParas = expectedText.split('\n');
131
- const actualParas = actualText.split('\n');
132
- const maxLen = Math.max(expectedParas.length, actualParas.length);
133
- let firstDifferingParagraphIndex = -1;
134
- for (let i = 0; i < maxLen; i++) {
135
- if ((expectedParas[i] ?? '') !== (actualParas[i] ?? '')) {
136
- firstDifferingParagraphIndex = i;
137
- break;
138
- }
139
- }
140
- return {
141
- expectedLength: comparison.expectedLength,
142
- actualLength: comparison.actualLength,
143
- firstDifferingParagraphIndex,
144
- expectedParagraph: firstDifferingParagraphIndex >= 0 ? (expectedParas[firstDifferingParagraphIndex] ?? '') : '',
145
- actualParagraph: firstDifferingParagraphIndex >= 0 ? (actualParas[firstDifferingParagraphIndex] ?? '') : '',
146
- differenceSample: comparison.differences.slice(0, 3),
147
- };
148
- }
149
- function buildBookmarkMismatchDetails(expected, actual) {
150
- return {
151
- startNames: diffIds(expected.startNames, actual.startNames),
152
- referencedBookmarkNames: diffIds(expected.referencedBookmarkNames, actual.referencedBookmarkNames),
153
- unresolvedReferenceNames: diffIds(expected.unresolvedReferenceNames, actual.unresolvedReferenceNames),
154
- startIds: diffIds(expected.startIds, actual.startIds),
155
- endIds: diffIds(expected.endIds, actual.endIds),
156
- expectedDuplicateStartNames: expected.duplicateStartNames,
157
- actualDuplicateStartNames: actual.duplicateStartNames,
158
- expectedDuplicateStartIds: expected.duplicateStartIds,
159
- actualDuplicateStartIds: actual.duplicateStartIds,
160
- expectedDuplicateEndIds: expected.duplicateEndIds,
161
- actualDuplicateEndIds: actual.duplicateEndIds,
162
- expectedUnmatchedStartIds: expected.unmatchedStartIds,
163
- actualUnmatchedStartIds: actual.unmatchedStartIds,
164
- expectedUnmatchedEndIds: expected.unmatchedEndIds,
165
- actualUnmatchedEndIds: actual.unmatchedEndIds,
166
- };
167
- }
168
- function summarizeIdDelta(delta) {
169
- return {
170
- missingCount: delta.missing.length,
171
- unexpectedCount: delta.unexpected.length,
172
- firstMissing: delta.missing[0],
173
- firstUnexpected: delta.unexpected[0],
174
- };
175
- }
176
- function truncateForSummary(value, maxLength = 160) {
177
- if (value.length <= maxLength) {
178
- return value;
179
- }
180
- return `${value.slice(0, maxLength)}...`;
181
- }
182
- function summarizeTextMismatch(details) {
183
- return {
184
- firstDifferingParagraphIndex: details.firstDifferingParagraphIndex,
185
- expectedParagraph: truncateForSummary(details.expectedParagraph),
186
- actualParagraph: truncateForSummary(details.actualParagraph),
187
- firstDifference: details.differenceSample[0] ?? 'No diff sample',
188
- };
189
- }
190
- function summarizeBookmarkMismatch(details) {
191
- return {
192
- startNames: summarizeIdDelta(details.startNames),
193
- referencedBookmarkNames: summarizeIdDelta(details.referencedBookmarkNames),
194
- unresolvedReferenceNames: summarizeIdDelta(details.unresolvedReferenceNames),
195
- startIds: summarizeIdDelta(details.startIds),
196
- endIds: summarizeIdDelta(details.endIds),
197
- unmatchedStartCount: details.actualUnmatchedStartIds.length,
198
- unmatchedEndCount: details.actualUnmatchedEndIds.length,
199
- firstUnmatchedStartId: details.actualUnmatchedStartIds[0],
200
- firstUnmatchedEndId: details.actualUnmatchedEndIds[0],
201
- };
202
- }
203
- function buildFailureSummary(failureDetails) {
204
- if (!failureDetails) {
205
- return undefined;
206
- }
207
- const summary = {};
208
- if (failureDetails.acceptText) {
209
- summary.acceptText = summarizeTextMismatch(failureDetails.acceptText);
210
- }
211
- if (failureDetails.rejectText) {
212
- summary.rejectText = summarizeTextMismatch(failureDetails.rejectText);
213
- }
214
- if (failureDetails.acceptBookmarks) {
215
- summary.acceptBookmarks = summarizeBookmarkMismatch(failureDetails.acceptBookmarks);
216
- }
217
- if (failureDetails.rejectBookmarks) {
218
- summary.rejectBookmarks = summarizeBookmarkMismatch(failureDetails.rejectBookmarks);
219
- }
220
- return Object.keys(summary).length > 0 ? summary : undefined;
221
- }
222
- // Declared above splitStories so the function body never observes an
223
- // uninitialized binding under circular imports.
224
- const serializer = new XMLSerializer();
225
- /**
226
- * Split a docx into per-story XML fragments for field-closure validation.
227
- *
228
- * Each footnote/endnote entry is treated as an isolated story: a complex
229
- * field whose `begin` and `end` markers straddle stories breaks Word's
230
- * field state machine. We therefore validate each `<w:footnote>` and
231
- * `<w:endnote>` entry independently rather than treating the whole
232
- * `footnotes.xml`/`endnotes.xml` as one stream.
233
- *
234
- * Accepts arrays of sidecar XMLs (one per source archive) so callers can
235
- * validate the union of entries from every archive that may contribute to the
236
- * final result. Step 12 of `compareDocumentsAtomizer` merges entries from a
237
- * mode-dependent source archive into the base archive; passing both archives'
238
- * sidecars guarantees that whichever path the merge takes, the entries it
239
- * could publish have already been screened. Duplicates (same `w:id` in both
240
- * archives) yield redundant but harmless validation work.
241
- *
242
- * Header/footer stories are not yet covered — they require relationship
243
- * walking to enumerate `headerN.xml`/`footerN.xml`.
244
- *
245
- * @conformance ECMA-376 edition 5, Part 4 § 17.16.5
246
- * @see https://github.com/UseJunior/safe-docx/issues/212
247
- */
248
- export function splitStories(documentXml, footnotesXmls, endnotesXmls) {
249
- const stories = [{ label: 'document', xml: documentXml }];
250
- const collectEntries = (sidecars, entryTag, labelPrefix) => {
251
- for (let s = 0; s < sidecars.length; s++) {
252
- const sidecarXml = sidecars[s];
253
- if (!sidecarXml)
254
- continue;
255
- const doc = parseXml(sidecarXml);
256
- const entries = doc.getElementsByTagName(entryTag);
257
- for (let i = 0; i < entries.length; i++) {
258
- const entry = entries[i];
259
- const id = entry.getAttribute('w:id') ?? String(i);
260
- stories.push({
261
- label: `${labelPrefix}[${s}]:${id}`,
262
- xml: serializer.serializeToString(entry),
263
- });
264
- }
265
- }
266
- };
267
- collectEntries(footnotesXmls, 'w:footnote', 'footnote');
268
- collectEntries(endnotesXmls, 'w:endnote', 'endnote');
269
- return stories;
270
- }
271
- function evaluateSafetyChecks(originalTextForRoundTrip, revisedTextForRoundTrip, originalBookmarkDiagnostics, revisedBookmarkDiagnostics, candidateXml, auxiliarySidecars) {
272
- const acceptedXml = acceptAllChanges(candidateXml);
273
- const rejectedXml = rejectAllChanges(candidateXml);
274
- const acceptedText = extractTextWithParagraphs(acceptedXml);
275
- const rejectedText = extractTextWithParagraphs(rejectedXml);
276
- const acceptedBookmarkDiagnostics = collectBookmarkDiagnostics(acceptedXml);
277
- const rejectedBookmarkDiagnostics = collectBookmarkDiagnostics(rejectedXml);
278
- const acceptTextComparison = compareTexts(revisedTextForRoundTrip, acceptedText);
279
- const rejectTextComparison = compareTexts(originalTextForRoundTrip, rejectedText);
280
- const acceptBookmarksOk = bookmarkDiagnosticsSemanticallyEqual(revisedBookmarkDiagnostics, acceptedBookmarkDiagnostics);
281
- const rejectBookmarksOk = bookmarkDiagnosticsSemanticallyEqual(originalBookmarkDiagnostics, rejectedBookmarkDiagnostics);
282
- // Validate field structure per-story. Each footnote/endnote entry is its own
283
- // ECMA-376 story; a complex field that crosses a story boundary breaks
284
- // Word's field state machine even when global begin/end counts balance.
285
- // Sidecars from BOTH archives are validated because Step 12's auxiliary-part
286
- // merge picks its base and source archives by reconstruction mode (inplace
287
- // base = revised; rebuild base = original) and validating only one side
288
- // would miss field issues that would still ship in the merged result.
289
- // `acceptAllChanges` / `rejectAllChanges` only transform document.xml, so
290
- // the sidecar set is identical for both transforms.
291
- const acceptedStories = splitStories(acceptedXml, auxiliarySidecars.footnotesXmls, auxiliarySidecars.endnotesXmls);
292
- const rejectedStories = splitStories(rejectedXml, auxiliarySidecars.footnotesXmls, auxiliarySidecars.endnotesXmls);
293
- // Issue #217 conformance gate on the COMBINED output: w:fldChar MUST NOT
294
- // appear inside <w:del>. ECMA-376 Part 4 § 17.16.5 makes this fatal for
295
- // Word's field state machine. The full validateFieldStructure check is run
296
- // on the accept/reject projections (per-story); on the combined view we
297
- // only gate the strict no-fldChar-in-del rule because some legacy emit
298
- // paths (e.g. delInstrText inside <w:moveFrom>) are non-conformant in shape
299
- // but out of scope for #217.
300
- const combinedNoFldCharInDel = !hasFldCharInsideDel(candidateXml);
301
- const fieldStructureOk = combinedNoFldCharInDel &&
302
- validateFieldStructure(acceptedStories) &&
303
- validateFieldStructure(rejectedStories);
304
- const checks = {
305
- acceptText: acceptTextComparison.normalizedIdentical,
306
- rejectText: rejectTextComparison.normalizedIdentical,
307
- // Bookmark checks are soft: consumer compatibility pass legitimately alters
308
- // bookmarks (deduplication, orphan repair, hoisting out of revision wrappers).
309
- // Log mismatches in diagnostics but don't trigger fallback to rebuild.
310
- acceptBookmarks: true,
311
- rejectBookmarks: true,
312
- fieldStructure: fieldStructureOk,
313
- };
314
- const failedChecks = Object.entries(checks)
315
- .filter(([, ok]) => !ok)
316
- .map(([name]) => name);
317
- const failureDetails = {};
318
- if (!checks.acceptText) {
319
- failureDetails.acceptText = buildTextMismatchDetails(revisedTextForRoundTrip, acceptedText);
320
- }
321
- if (!checks.rejectText) {
322
- failureDetails.rejectText = buildTextMismatchDetails(originalTextForRoundTrip, rejectedText);
323
- }
324
- // Bookmark mismatches are always collected for diagnostics even though the
325
- // check itself is soft (doesn't trigger fallback).
326
- if (!acceptBookmarksOk) {
327
- failureDetails.acceptBookmarks = buildBookmarkMismatchDetails(revisedBookmarkDiagnostics, acceptedBookmarkDiagnostics);
328
- }
329
- if (!rejectBookmarksOk) {
330
- failureDetails.rejectBookmarks = buildBookmarkMismatchDetails(originalBookmarkDiagnostics, rejectedBookmarkDiagnostics);
331
- }
332
- return {
333
- safe: failedChecks.length === 0,
334
- checks,
335
- failedChecks,
336
- failureDetails: failedChecks.length > 0 ? failureDetails : undefined,
337
- failureSummary: failedChecks.length > 0 ? buildFailureSummary(failureDetails) : undefined,
338
- };
339
- }
340
- /**
341
- * Compare two DOCX documents using the atomizer-based approach.
342
- *
343
- * Pipeline steps:
344
- * 1. Load DOCX archives
345
- * 2. Extract document.xml
346
- * 3. Parse to WmlElement trees
347
- * 4. Atomize both documents
348
- * 5. (Optional) Apply numbering virtualization
349
- * 6. Run LCS on atom hashes
350
- * 7. Mark correlation status
351
- * 8. Run move detection
352
- * 9. Run format detection
353
- * 10. Reconstruct document with track changes
354
- * 11. Save and return result
355
- *
356
- * @param original - Original document as Buffer
357
- * @param revised - Revised document as Buffer
358
- * @param options - Pipeline options
359
- * @returns Comparison result with track changes document
360
- */
361
- export async function compareDocumentsAtomizer(original, revised, options = {}) {
362
- const { author = 'Comparison', date = new Date(), moveDetection = {}, formatDetection = {}, numbering = {}, premergeRuns = true, reconstructionMode = 'rebuild', } = options;
363
- // Merge settings with defaults
364
- const moveSettings = {
365
- ...DEFAULT_MOVE_DETECTION_SETTINGS,
366
- ...moveDetection,
367
- };
368
- const formatSettings = {
369
- ...DEFAULT_FORMAT_DETECTION_SETTINGS,
370
- ...formatDetection,
371
- };
372
- const numberingSettings = {
373
- ...DEFAULT_NUMBERING_OPTIONS,
374
- ...numbering,
375
- };
376
- // Step 1: Load DOCX archives
377
- const originalArchive = await DocxArchive.load(original);
378
- const revisedArchive = await DocxArchive.load(revised);
379
- // Step 1b: Resolve auxiliary ID collisions. When both sides define
380
- // different content under the same comment/footnote/endnote w:id or the
381
- // same comment paraId, rewrite the revised side so no anchor or ancillary
382
- // row in the merged output can bind to the other document's definition.
383
- // Must run before any document.xml extraction so every downstream step sees
384
- // the rewritten archive.
385
- await renumberCollidingAuxiliaryIds(originalArchive, revisedArchive);
386
- await restampCollidingCommentParaIds(originalArchive, revisedArchive);
387
- // Step 2: Extract document.xml
388
- const originalXml = await originalArchive.getDocumentXml();
389
- const revisedXml = await revisedArchive.getDocumentXml();
390
- // Extract numbering.xml if available
391
- const originalNumberingXml = await originalArchive.getNumberingXml() ?? undefined;
392
- const revisedNumberingXml = await revisedArchive.getNumberingXml() ?? undefined;
393
- // Extract footnote/endnote sidecars from BOTH archives for per-story
394
- // field-closure validation (issue #212). Step 12 picks the base archive by
395
- // reconstruction mode (inplace = revised, rebuild = original) and merges
396
- // missing referenced entries from the opposite archive. Validating both
397
- // archives' sidecars covers the union of entries that could ship without
398
- // having to duplicate the merge logic at safety-check time.
399
- const [originalFootnotesXml, originalEndnotesXml, revisedFootnotesXml, revisedEndnotesXml,] = await Promise.all([
400
- originalArchive.getFile('word/footnotes.xml'),
401
- originalArchive.getFile('word/endnotes.xml'),
402
- revisedArchive.getFile('word/footnotes.xml'),
403
- revisedArchive.getFile('word/endnotes.xml'),
404
- ]);
405
- const auxiliarySidecars = {
406
- footnotesXmls: [originalFootnotesXml, revisedFootnotesXml],
407
- endnotesXmls: [originalEndnotesXml, revisedEndnotesXml],
408
- };
409
- const originalPart = {
410
- uri: 'word/document.xml',
411
- contentType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml',
412
- };
413
- const revisedPart = {
414
- uri: 'word/document.xml',
415
- contentType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml',
416
- };
417
- // Project each input through the SAME accept/reject operation the candidate is
418
- // checked under, so the round-trip comparison is like-for-like even when an
419
- // input already carries its own tracked changes (pre-tracked w:ins / w:del,
420
- // comment anchors, multi-author stacks). For a clean input these equal the raw
421
- // extraction, so behavior on the common case is unchanged. (#347)
422
- const originalTextForRoundTrip = extractTextWithParagraphs(rejectAllChanges(originalXml));
423
- const revisedTextForRoundTrip = extractTextWithParagraphs(acceptAllChanges(revisedXml));
424
- const originalBookmarkDiagnostics = collectBookmarkDiagnostics(originalXml);
425
- const revisedBookmarkDiagnostics = collectBookmarkDiagnostics(revisedXml);
426
- const runComparisonPass = (atomizeOptions, outputMode) => {
427
- // Parse fresh trees for each pass because inplace reconstruction mutates revised AST.
428
- const originalTree = parseDocumentXml(originalXml);
429
- const revisedTree = parseDocumentXml(revisedXml);
430
- backfillParentReferences(originalTree);
431
- backfillParentReferences(revisedTree);
432
- const originalBody = findBody(originalTree);
433
- const revisedBody = findBody(revisedTree);
434
- if (!originalBody || !revisedBody) {
435
- throw new Error('Could not find w:body in one or both documents');
436
- }
437
- if (premergeRuns) {
438
- premergeAdjacentRuns(originalBody);
439
- premergeAdjacentRuns(revisedBody);
440
- }
441
- const { atoms: originalAtoms } = atomizeTree(originalBody, [], originalPart, atomizeOptions);
442
- const { atoms: revisedAtoms } = atomizeTree(revisedBody, [], revisedPart, atomizeOptions);
443
- // Assign paragraph indices for proper grouping during reconstruction
444
- assignParagraphIndices(originalAtoms);
445
- assignParagraphIndices(revisedAtoms);
446
- // Step 5: Apply numbering virtualization (optional)
447
- if (numberingSettings.enabled) {
448
- virtualizeNumberingLabels(originalAtoms, originalNumberingXml, numberingSettings);
449
- virtualizeNumberingLabels(revisedAtoms, revisedNumberingXml, numberingSettings);
450
- }
451
- // Step 6: Run hierarchical LCS (paragraph-level first, then atom-level within)
452
- const lcsResult = hierarchicalCompare(originalAtoms, revisedAtoms);
453
- // Step 7: Mark correlation status using hierarchical result
454
- markHierarchicalCorrelationStatus(originalAtoms, revisedAtoms, lcsResult);
455
- // Step 8: Run move detection
456
- if (moveSettings.detectMoves) {
457
- // Create a combined list for move detection
458
- // Move detection looks at the revised atoms with Inserted status
459
- // and original atoms with Deleted status
460
- const allAtoms = [...originalAtoms, ...revisedAtoms];
461
- detectMovesInAtomList(allAtoms, moveSettings);
462
- }
463
- // Step 9: Run format detection
464
- if (formatSettings.detectFormatChanges) {
465
- // Format detection operates on the revised atoms that are Equal
466
- detectFormatChangesInAtomList(revisedAtoms, formatSettings);
467
- }
468
- // Step 10: Create merged atom list for reconstruction
469
- const mergedAtoms = createMergedAtomList(originalAtoms, revisedAtoms, lcsResult);
470
- // Step 10b: Assign unified paragraph indices to handle atoms from different trees
471
- assignUnifiedParagraphIndices(originalAtoms, revisedAtoms, mergedAtoms, lcsResult);
472
- // Step 11: Reconstruct document with track changes
473
- let newDocumentXml;
474
- if (outputMode === 'inplace') {
475
- // In-place mode: modify the revised AST directly, producing revised-based output.
476
- newDocumentXml = modifyRevisedDocument(revisedTree, originalAtoms, revisedAtoms, mergedAtoms, { author, date });
477
- }
478
- else {
479
- // Rebuild mode: reconstruct from atoms using original as the structural base.
480
- newDocumentXml = reconstructDocument(mergedAtoms, originalXml, { author, date });
481
- }
482
- return { mergedAtoms, newDocumentXml, outputMode };
483
- };
484
- const evaluateRoundTripSafety = (candidateXml) => evaluateSafetyChecks(originalTextForRoundTrip, revisedTextForRoundTrip, originalBookmarkDiagnostics, revisedBookmarkDiagnostics, candidateXml, auxiliarySidecars);
485
- let comparisonResult;
486
- let fallbackReason;
487
- let fallbackDiagnostics;
488
- if (reconstructionMode === 'inplace') {
489
- // Adaptive strategy:
490
- // 1) Try no-cross-run passes first (higher run anchoring fidelity).
491
- // 2) If safety fails, retry with cross-run merging to handle run-fragmented docs.
492
- // 3) If still unsafe, reuse rebuild reconstruction as a hard safety fallback.
493
- const inplacePasses = [
494
- {
495
- pass: 'inplace_word_split',
496
- atomizeOptions: {
497
- cloneLeafNodes: true,
498
- mergeAcrossRuns: false,
499
- mergePunctuationAcrossRuns: false,
500
- splitTextIntoWords: true,
501
- },
502
- },
503
- {
504
- pass: 'inplace_run_level',
505
- atomizeOptions: {
506
- cloneLeafNodes: true,
507
- mergeAcrossRuns: false,
508
- mergePunctuationAcrossRuns: false,
509
- splitTextIntoWords: false,
510
- },
511
- },
512
- {
513
- pass: 'inplace_word_split_cross_run',
514
- atomizeOptions: {
515
- cloneLeafNodes: true,
516
- mergeAcrossRuns: true,
517
- mergePunctuationAcrossRuns: true,
518
- splitTextIntoWords: true,
519
- },
520
- },
521
- {
522
- pass: 'inplace_run_level_cross_run',
523
- atomizeOptions: {
524
- cloneLeafNodes: true,
525
- mergeAcrossRuns: true,
526
- mergePunctuationAcrossRuns: true,
527
- splitTextIntoWords: false,
528
- },
529
- },
530
- ];
531
- const failedAttempts = [];
532
- let selected;
533
- for (const { pass, atomizeOptions } of inplacePasses) {
534
- let candidate;
535
- try {
536
- candidate = runComparisonPass(atomizeOptions, 'inplace');
537
- }
538
- catch (e) {
539
- if (e instanceof ContainerResolutionError) {
540
- // Container topology mismatch — treat as failed pass (issue #65)
541
- failedAttempts.push({
542
- pass,
543
- checks: { acceptText: false, rejectText: false, acceptBookmarks: true, rejectBookmarks: true, fieldStructure: false },
544
- failedChecks: ['rejectText'],
545
- failureDetails: undefined,
546
- firstDiffSummary: undefined,
547
- });
548
- continue;
549
- }
550
- throw e;
551
- }
552
- const safety = evaluateRoundTripSafety(candidate.newDocumentXml);
553
- if (safety.safe) {
554
- selected = candidate;
555
- break;
556
- }
557
- failedAttempts.push({
558
- pass,
559
- checks: safety.checks,
560
- failedChecks: safety.failedChecks,
561
- failureDetails: safety.failureDetails,
562
- firstDiffSummary: safety.failureSummary,
563
- });
564
- }
565
- if (selected) {
566
- comparisonResult = selected;
567
- }
568
- else {
569
- comparisonResult = runComparisonPass({ atomizeParagraphLevelMarkers: true }, 'rebuild');
570
- fallbackReason = 'round_trip_safety_check_failed';
571
- fallbackDiagnostics = {
572
- attempts: failedAttempts,
573
- };
574
- }
575
- }
576
- else {
577
- comparisonResult = runComparisonPass({ atomizeParagraphLevelMarkers: true }, 'rebuild');
578
- }
579
- // Rebuild output gets the same safety screening as inplace attempts, whether
580
- // rebuild was requested directly or reached via inplace fallback. Rebuild is
581
- // the terminal strategy, so failures are surfaced in diagnostics rather than
582
- // blocking the output.
583
- // @see https://github.com/UseJunior/safe-docx/issues/226
584
- let rebuildSafetyDiagnostics;
585
- if (comparisonResult.outputMode === 'rebuild') {
586
- const safety = evaluateRoundTripSafety(comparisonResult.newDocumentXml);
587
- if (!safety.safe) {
588
- rebuildSafetyDiagnostics = {
589
- checks: safety.checks,
590
- failedChecks: safety.failedChecks,
591
- failureDetails: safety.failureDetails,
592
- firstDiffSummary: safety.failureSummary,
593
- };
594
- }
595
- }
596
- const { mergedAtoms, newDocumentXml } = comparisonResult;
597
- // Step 12: Clone appropriate archive and update document.xml.
598
- // Use the revised archive only for true inplace output.
599
- const baseArchive = comparisonResult.outputMode === 'inplace' ? revisedArchive : originalArchive;
600
- // The merge source is the *opposite* archive from the base: inplace pulls
601
- // deleted-but-still-referenced definitions from the original, rebuild pulls
602
- // added-but-still-referenced definitions from the revised. Without this,
603
- // rebuild output ships dangling references when the original lacks an
604
- // auxiliary part that the revised side introduced (issue #94).
605
- const mergeSourceArchive = comparisonResult.outputMode === 'inplace' ? originalArchive : revisedArchive;
606
- const resultArchive = await baseArchive.clone();
607
- maybeCaptureEmittedDocumentXml(newDocumentXml);
608
- resultArchive.setDocumentXml(newDocumentXml);
609
- // Step 12b: Merge auxiliary part definitions (footnotes, endnotes, comments).
610
- // Reconstruction may insert content (deleted in inplace, added in rebuild)
611
- // whose definitions are missing from the base archive.
612
- for (const descriptor of AUXILIARY_PARTS) {
613
- await mergeAuxiliaryPartDefinitions(mergeSourceArchive, resultArchive, newDocumentXml, descriptor);
614
- }
615
- // Comment-specific post-pass: walk reply threads via commentsExtended.xml.
616
- // Gated on root comment IDs in the *result* document (not on what the
617
- // generic merge appended), so the pass runs even when the original already
618
- // contains the root and revised only adds replies under it (issue #108).
619
- // Comments anchored on footnote/endnote text count as roots too.
620
- const rootCommentIds = await collectStoryReferenceIds(resultArchive, newDocumentXml, 'w:commentReference', null);
621
- if (rootCommentIds.size > 0) {
622
- await mergeCommentAncillaryParts(mergeSourceArchive, resultArchive, rootCommentIds);
623
- }
624
- // Step 13: Save result and compute stats
625
- const resultBuffer = await resultArchive.save();
626
- const stats = computeAtomizerStats(mergedAtoms);
627
- return {
628
- document: resultBuffer,
629
- stats,
630
- engine: 'atomizer',
631
- reconstructionModeRequested: reconstructionMode,
632
- reconstructionModeUsed: comparisonResult.outputMode,
633
- fallbackReason,
634
- fallbackDiagnostics,
635
- rebuildSafetyDiagnostics,
636
- };
637
- }
638
- /**
639
- * Collect reference IDs across every result story that can host anchors: the
640
- * merged document.xml plus the result archive's footnote/endnote parts (Word
641
- * allows comments anchored on note text). `excludePartPath` skips the part
642
- * whose own definitions are being merged — entries can't reference
643
- * themselves.
644
- */
645
- async function collectStoryReferenceIds(resultArchive, documentXml, referenceTag, excludePartPath) {
646
- const ids = collectReferenceIds(documentXml, referenceTag);
647
- for (const storyPath of ['word/footnotes.xml', 'word/endnotes.xml']) {
648
- if (storyPath === excludePartPath)
649
- continue;
650
- const storyXml = await resultArchive.getFile(storyPath);
651
- if (!storyXml)
652
- continue;
653
- for (const id of collectReferenceIds(storyXml, referenceTag))
654
- ids.add(id);
655
- }
656
- return ids;
657
- }
658
- /**
659
- * Collect reference IDs from document.xml using DOM parsing.
660
- */
661
- function collectReferenceIds(documentXml, referenceTag) {
662
- const ids = new Set();
663
- const doc = parseXml(documentXml);
664
- const refs = doc.getElementsByTagName(referenceTag);
665
- for (let i = 0; i < refs.length; i++) {
666
- const id = refs[i].getAttribute('w:id');
667
- if (id)
668
- ids.add(id);
669
- }
670
- return ids;
671
- }
672
- /**
673
- * Merge auxiliary part definitions (footnotes, endnotes, comments) from the
674
- * source archive into the result archive. The source archive is whichever
675
- * side reconstruction may have introduced references to: original in inplace
676
- * mode (deleted-but-referenced definitions), revised in rebuild mode
677
- * (added-but-referenced definitions).
678
- */
679
- async function mergeAuxiliaryPartDefinitions(sourceArchive, resultArchive, documentXml, descriptor) {
680
- const result = { mergedIds: new Set(), createdPart: false };
681
- // Anchors may live in the merged body or on note text in the result's
682
- // footnote/endnote stories. AUXILIARY_PARTS merges notes before comments,
683
- // so by the comment pass the note stories already carry any merged-in
684
- // comment anchors.
685
- const referencedIds = await collectStoryReferenceIds(resultArchive, documentXml, descriptor.referenceTag, descriptor.partPath);
686
- if (referencedIds.size === 0)
687
- return result;
688
- const sourcePartXml = await sourceArchive.getFile(descriptor.partPath);
689
- if (!sourcePartXml)
690
- return result;
691
- const resultPartXml = await resultArchive.getFile(descriptor.partPath);
692
- const sourceParsed = parseEntries(sourcePartXml, descriptor.entryTag);
693
- const resultParsed = resultPartXml ? parseEntries(resultPartXml, descriptor.entryTag) : null;
694
- // Find missing entries: referenced in document.xml but not in result
695
- const missingElements = [];
696
- for (const id of referencedIds) {
697
- if (!(resultParsed?.entries.has(id)) && sourceParsed.entries.has(id)) {
698
- missingElements.push(sourceParsed.entries.get(id));
699
- result.mergedIds.add(id);
700
- }
701
- }
702
- if (missingElements.length === 0)
703
- return result;
704
- if (resultPartXml && resultParsed) {
705
- // Insert missing entries into existing result part
706
- const rootEl = resultParsed.doc.getElementsByTagName(descriptor.rootTag)[0];
707
- if (rootEl) {
708
- for (const el of missingElements) {
709
- const imported = resultParsed.doc.importNode(el, true);
710
- rootEl.appendChild(imported);
711
- }
712
- resultArchive.setFile(descriptor.partPath, serializer.serializeToString(resultParsed.doc));
713
- }
714
- }
715
- else {
716
- // Create part from scratch: clone root from merge source, drop every
717
- // non-reserved entry, then append the missing referenced ones.
718
- // Reserved entries are footnote/endnote separators identified by
719
- // w:type="separator" / w:type="continuationSeparator" — Word expects
720
- // them to exist and they don't carry user content. Filtering by w:type
721
- // (not by magic w:id values) keeps this robust across authoring tools.
722
- const newDoc = parseXml(sourcePartXml);
723
- const rootEl = newDoc.getElementsByTagName(descriptor.rootTag)[0];
724
- if (rootEl) {
725
- const existingEntries = rootEl.getElementsByTagName(descriptor.entryTag);
726
- const toRemove = [];
727
- for (let i = 0; i < existingEntries.length; i++) {
728
- const el = existingEntries[i];
729
- const type = el.getAttribute('w:type');
730
- if (type !== 'separator' && type !== 'continuationSeparator') {
731
- toRemove.push(el);
732
- }
733
- }
734
- for (const el of toRemove) {
735
- rootEl.removeChild(el);
736
- }
737
- for (const el of missingElements) {
738
- const imported = newDoc.importNode(el, true);
739
- rootEl.appendChild(imported);
740
- }
741
- resultArchive.setFile(descriptor.partPath, serializer.serializeToString(newDoc));
742
- result.createdPart = true;
743
- await ensureOpcMetadata(resultArchive, descriptor);
744
- }
745
- }
746
- return result;
747
- }
748
- // =============================================================================
749
- // OPC Metadata Bootstrapping
750
- // =============================================================================
751
- const CT_NS = 'http://schemas.openxmlformats.org/package/2006/content-types';
752
- const REL_NS = 'http://schemas.openxmlformats.org/package/2006/relationships';
753
- /**
754
- * Ensure [Content_Types].xml and document.xml.rels have entries for a
755
- * newly-created auxiliary part.
756
- */
757
- async function ensureOpcMetadata(archive, descriptor) {
758
- // 1. Update [Content_Types].xml
759
- const ctXml = await archive.getFile('[Content_Types].xml');
760
- if (ctXml) {
761
- const ctDoc = parseXml(ctXml);
762
- const typesEl = ctDoc.documentElement;
763
- const overrides = typesEl.getElementsByTagNameNS(CT_NS, 'Override');
764
- const partName = `/${descriptor.partPath}`;
765
- let found = false;
766
- for (let i = 0; i < overrides.length; i++) {
767
- if (overrides[i].getAttribute('PartName') === partName) {
768
- found = true;
769
- break;
770
- }
771
- }
772
- if (!found) {
773
- const override = ctDoc.createElementNS(CT_NS, 'Override');
774
- override.setAttribute('PartName', partName);
775
- override.setAttribute('ContentType', descriptor.contentType);
776
- typesEl.appendChild(override);
777
- archive.setFile('[Content_Types].xml', serializer.serializeToString(ctDoc));
778
- }
779
- }
780
- // 2. Update word/_rels/document.xml.rels
781
- const relsPath = 'word/_rels/document.xml.rels';
782
- const relsXml = await archive.getFile(relsPath);
783
- if (relsXml) {
784
- const relsDoc = parseXml(relsXml);
785
- const relsEl = relsDoc.documentElement;
786
- const existingRels = relsEl.getElementsByTagNameNS(REL_NS, 'Relationship');
787
- let found = false;
788
- let maxId = 0;
789
- for (let i = 0; i < existingRels.length; i++) {
790
- const rel = existingRels[i];
791
- if (rel.getAttribute('Type') === descriptor.relationshipType) {
792
- found = true;
793
- }
794
- const id = rel.getAttribute('Id') ?? '';
795
- const idMatch = /^rId(\d+)$/.exec(id);
796
- if (idMatch)
797
- maxId = Math.max(maxId, parseInt(idMatch[1], 10));
798
- }
799
- if (!found) {
800
- maxId++;
801
- const rel = relsDoc.createElementNS(REL_NS, 'Relationship');
802
- rel.setAttribute('Id', `rId${maxId}`);
803
- rel.setAttribute('Type', descriptor.relationshipType);
804
- rel.setAttribute('Target', descriptor.partPath.replace('word/', ''));
805
- relsEl.appendChild(rel);
806
- archive.setFile(relsPath, serializer.serializeToString(relsDoc));
807
- }
808
- }
809
- }
810
- // =============================================================================
811
- // Comment Ancillary Parts Merging
812
- // =============================================================================
813
- /**
814
- * Walk the comment reply graph from each root referenced in the result
815
- * document, merging reply <w:comment> entries, their commentsExtended.xml
816
- * threading entries, and people.xml authors. Replies have no
817
- * <w:commentReference> in document.xml — they're discoverable only via
818
- * w15:paraIdParent in commentsExtended.xml. Without this expansion, rebuild
819
- * mode silently drops reply threads (issue #108).
820
- */
821
- async function mergeCommentAncillaryParts(sourceArchive, resultArchive, rootCommentIds) {
822
- const sourceCommentsXml = await sourceArchive.getFile('word/comments.xml');
823
- if (!sourceCommentsXml)
824
- return;
825
- const sourceDoc = parseXml(sourceCommentsXml);
826
- // Build full source comment maps. Canonical paraId is the first <w:p>
827
- // child's w14:paraId, matching getCommentElParaId() in primitives/comments.ts.
828
- const commentById = new Map();
829
- const paraIdByCommentId = new Map();
830
- const commentIdByParaId = new Map();
831
- const authorByCommentId = new Map();
832
- const allCommentEls = sourceDoc.getElementsByTagName('w:comment');
833
- for (let i = 0; i < allCommentEls.length; i++) {
834
- const el = allCommentEls[i];
835
- const id = el.getAttribute('w:id');
836
- if (!id)
837
- continue;
838
- commentById.set(id, el);
839
- const author = el.getAttribute('w:author');
840
- if (author)
841
- authorByCommentId.set(id, author);
842
- const firstP = el.getElementsByTagName('w:p')[0];
843
- const paraId = firstP?.getAttribute('w14:paraId');
844
- if (paraId) {
845
- paraIdByCommentId.set(id, paraId);
846
- commentIdByParaId.set(paraId, id);
847
- }
848
- }
849
- // Seed inclusion sets from the root IDs that appear in the result document.
850
- const includedCommentIds = new Set();
851
- const includedParaIds = new Set();
852
- const includedAuthors = new Set();
853
- for (const id of rootCommentIds) {
854
- if (!commentById.has(id))
855
- continue;
856
- includedCommentIds.add(id);
857
- const pid = paraIdByCommentId.get(id);
858
- if (pid)
859
- includedParaIds.add(pid);
860
- const author = authorByCommentId.get(id);
861
- if (author)
862
- includedAuthors.add(author);
863
- }
864
- // BFS over commentsExtended.xml's paraIdParent graph from each included
865
- // root paraId. Skip entries that don't resolve to a real source comment so
866
- // we never pull in dangling commentEx/people without a backing definition.
867
- const sourceExtendedXml = await sourceArchive.getFile('word/commentsExtended.xml');
868
- if (sourceExtendedXml) {
869
- const exDoc = parseXml(sourceExtendedXml);
870
- const exEls = exDoc.getElementsByTagName('w15:commentEx');
871
- const childrenOf = new Map();
872
- for (let i = 0; i < exEls.length; i++) {
873
- const ex = exEls[i];
874
- const childPid = ex.getAttribute('w15:paraId');
875
- const parentPid = ex.getAttribute('w15:paraIdParent');
876
- if (!childPid || !parentPid)
877
- continue;
878
- const arr = childrenOf.get(parentPid);
879
- if (arr)
880
- arr.push(childPid);
881
- else
882
- childrenOf.set(parentPid, [childPid]);
883
- }
884
- const queue = [...includedParaIds];
885
- while (queue.length > 0) {
886
- const pid = queue.shift();
887
- const children = childrenOf.get(pid);
888
- if (!children)
889
- continue;
890
- for (const childPid of children) {
891
- if (includedParaIds.has(childPid))
892
- continue;
893
- const childCommentId = commentIdByParaId.get(childPid);
894
- if (!childCommentId)
895
- continue;
896
- includedParaIds.add(childPid);
897
- includedCommentIds.add(childCommentId);
898
- const author = authorByCommentId.get(childCommentId);
899
- if (author)
900
- includedAuthors.add(author);
901
- queue.push(childPid);
902
- }
903
- }
904
- }
905
- // Append any reply <w:comment> definitions still missing from result.
906
- // The generic merge already added roots when needed; we add the replies
907
- // (and any roots not yet present in the result, defensively).
908
- await mergeMissingCommentDefinitions(resultArchive, commentById, includedCommentIds);
909
- // Merge commentsExtended and people for the expanded set.
910
- await mergeCommentsExtended(sourceArchive, resultArchive, includedParaIds);
911
- await mergePeople(sourceArchive, resultArchive, includedAuthors);
912
- }
913
- /**
914
- * Append any source <w:comment> definitions in `includedCommentIds` that
915
- * aren't already in result/word/comments.xml. Mirrors the append-with-importNode
916
- * pattern used by mergeCommentsExtended below.
917
- */
918
- async function mergeMissingCommentDefinitions(resultArchive, commentById, includedCommentIds) {
919
- if (includedCommentIds.size === 0)
920
- return;
921
- const resultXml = await resultArchive.getFile('word/comments.xml');
922
- if (!resultXml) {
923
- // If result has no comments.xml at all, the generic merge would have
924
- // bootstrapped it for any included root. Nothing to do here.
925
- return;
926
- }
927
- const resultDoc = parseXml(resultXml);
928
- const rootEl = resultDoc.documentElement;
929
- const existingIds = new Set();
930
- const existing = rootEl.getElementsByTagName('w:comment');
931
- for (let i = 0; i < existing.length; i++) {
932
- const id = existing[i].getAttribute('w:id');
933
- if (id)
934
- existingIds.add(id);
935
- }
936
- let appended = false;
937
- for (const id of includedCommentIds) {
938
- if (existingIds.has(id))
939
- continue;
940
- const sourceEl = commentById.get(id);
941
- if (!sourceEl)
942
- continue;
943
- rootEl.appendChild(resultDoc.importNode(sourceEl, true));
944
- appended = true;
945
- }
946
- if (appended) {
947
- resultArchive.setFile('word/comments.xml', serializer.serializeToString(resultDoc));
948
- }
949
- }
950
- async function mergeCommentsExtended(sourceArchive, resultArchive, mergedParaIds) {
951
- if (mergedParaIds.size === 0)
952
- return;
953
- const sourceXml = await sourceArchive.getFile('word/commentsExtended.xml');
954
- if (!sourceXml)
955
- return;
956
- const sourceDoc = parseXml(sourceXml);
957
- const sourceEntries = sourceDoc.getElementsByTagName('w15:commentEx');
958
- // Collect entries whose paraId matches a merged comment's paragraph
959
- const entriesToMerge = [];
960
- for (let i = 0; i < sourceEntries.length; i++) {
961
- const el = sourceEntries[i];
962
- const paraId = el.getAttribute('w15:paraId');
963
- if (paraId && mergedParaIds.has(paraId)) {
964
- entriesToMerge.push(el);
965
- }
966
- }
967
- if (entriesToMerge.length === 0)
968
- return;
969
- const resultXml = await resultArchive.getFile('word/commentsExtended.xml');
970
- if (resultXml) {
971
- const resultDoc = parseXml(resultXml);
972
- const rootEl = resultDoc.documentElement;
973
- const existingParaIds = new Set();
974
- const existing = rootEl.getElementsByTagName('w15:commentEx');
975
- for (let i = 0; i < existing.length; i++) {
976
- const pid = existing[i].getAttribute('w15:paraId');
977
- if (pid)
978
- existingParaIds.add(pid);
979
- }
980
- for (const el of entriesToMerge) {
981
- const pid = el.getAttribute('w15:paraId');
982
- if (pid && !existingParaIds.has(pid)) {
983
- rootEl.appendChild(resultDoc.importNode(el, true));
984
- }
985
- }
986
- resultArchive.setFile('word/commentsExtended.xml', serializer.serializeToString(resultDoc));
987
- return;
988
- }
989
- // Bootstrap: result lacks commentsExtended.xml but the merged comments
990
- // depend on it for reply threading / done state. Clone the source's root
991
- // (preserves namespaces), drop non-matching entries, then add OPC metadata.
992
- const newDoc = parseXml(sourceXml);
993
- const newRoot = newDoc.documentElement;
994
- const allEntries = newRoot.getElementsByTagName('w15:commentEx');
995
- const toRemove = [];
996
- for (let i = 0; i < allEntries.length; i++) {
997
- const el = allEntries[i];
998
- const paraId = el.getAttribute('w15:paraId');
999
- if (!paraId || !mergedParaIds.has(paraId))
1000
- toRemove.push(el);
1001
- }
1002
- for (const el of toRemove)
1003
- newRoot.removeChild(el);
1004
- resultArchive.setFile('word/commentsExtended.xml', serializer.serializeToString(newDoc));
1005
- await ensureOpcMetadata(resultArchive, COMMENTS_EXTENDED_DESCRIPTOR);
1006
- }
1007
- const COMMENTS_EXTENDED_DESCRIPTOR = {
1008
- label: 'commentsExtended',
1009
- partPath: 'word/commentsExtended.xml',
1010
- referenceTag: '',
1011
- entryTag: 'w15:commentEx',
1012
- rootTag: 'w15:commentsEx',
1013
- contentType: 'application/vnd.ms-word.commentsExtended+xml',
1014
- relationshipType: 'http://schemas.microsoft.com/office/2011/relationships/commentsExtended',
1015
- idBearingTags: [], // keyed by w15:paraId, not w:id
1016
- };
1017
- const PEOPLE_DESCRIPTOR = {
1018
- label: 'people',
1019
- partPath: 'word/people.xml',
1020
- referenceTag: '',
1021
- entryTag: 'w15:person',
1022
- rootTag: 'w15:people',
1023
- contentType: 'application/vnd.ms-word.people+xml',
1024
- relationshipType: 'http://schemas.microsoft.com/office/2011/relationships/people',
1025
- idBearingTags: [], // keyed by w15:author, not w:id
1026
- };
1027
- async function mergePeople(sourceArchive, resultArchive, mergedAuthors) {
1028
- if (mergedAuthors.size === 0)
1029
- return;
1030
- const sourceXml = await sourceArchive.getFile('word/people.xml');
1031
- if (!sourceXml)
1032
- return;
1033
- const sourceDoc = parseXml(sourceXml);
1034
- const sourcePersons = sourceDoc.getElementsByTagName('w15:person');
1035
- const personsToMerge = [];
1036
- for (let i = 0; i < sourcePersons.length; i++) {
1037
- const el = sourcePersons[i];
1038
- const author = el.getAttribute('w15:author');
1039
- if (author && mergedAuthors.has(author)) {
1040
- personsToMerge.push(el);
1041
- }
1042
- }
1043
- if (personsToMerge.length === 0)
1044
- return;
1045
- const resultXml = await resultArchive.getFile('word/people.xml');
1046
- if (resultXml) {
1047
- const resultDoc = parseXml(resultXml);
1048
- const rootEl = resultDoc.documentElement;
1049
- const existingAuthors = new Set();
1050
- const existing = rootEl.getElementsByTagName('w15:person');
1051
- for (let i = 0; i < existing.length; i++) {
1052
- const a = existing[i].getAttribute('w15:author');
1053
- if (a)
1054
- existingAuthors.add(a);
1055
- }
1056
- for (const el of personsToMerge) {
1057
- const a = el.getAttribute('w15:author');
1058
- if (a && !existingAuthors.has(a)) {
1059
- rootEl.appendChild(resultDoc.importNode(el, true));
1060
- }
1061
- }
1062
- resultArchive.setFile('word/people.xml', serializer.serializeToString(resultDoc));
1063
- return;
1064
- }
1065
- // Bootstrap: result lacks people.xml. Clone source root (preserves
1066
- // namespaces), remove non-matching authors, then add OPC metadata.
1067
- const newDoc = parseXml(sourceXml);
1068
- const newRoot = newDoc.documentElement;
1069
- const allPersons = newRoot.getElementsByTagName('w15:person');
1070
- const toRemove = [];
1071
- for (let i = 0; i < allPersons.length; i++) {
1072
- const el = allPersons[i];
1073
- const author = el.getAttribute('w15:author');
1074
- if (!author || !mergedAuthors.has(author))
1075
- toRemove.push(el);
1076
- }
1077
- for (const el of toRemove)
1078
- newRoot.removeChild(el);
1079
- resultArchive.setFile('word/people.xml', serializer.serializeToString(newDoc));
1080
- await ensureOpcMetadata(resultArchive, PEOPLE_DESCRIPTOR);
1081
- }
1082
- const fallbackParagraphStatsKeys = new WeakMap();
1083
- let nextFallbackParagraphStatsKey = 0;
1084
- function paragraphStatsKey(atom) {
1085
- if (atom.paragraphIndex !== undefined) {
1086
- return `${atom.part.uri}:${atom.paragraphIndex}`;
1087
- }
1088
- const pAncestor = atom.ancestorElements.find((a) => a.tagName === 'w:p');
1089
- if (!pAncestor)
1090
- return undefined;
1091
- let key = fallbackParagraphStatsKeys.get(pAncestor);
1092
- if (!key) {
1093
- key = `${atom.part.uri}:paragraph-ref:${nextFallbackParagraphStatsKey++}`;
1094
- fallbackParagraphStatsKeys.set(pAncestor, key);
1095
- }
1096
- return key;
1097
- }
1098
- /**
1099
- * Compute comparison statistics from merged atoms.
1100
- *
1101
- * Range counts are contiguous same-status runs in the merged atom stream, scoped
1102
- * to a paragraph. Atom counts remain available under explicit names for callers
1103
- * that need the old granular benchmark signal.
1104
- */
1105
- export function computeAtomizerStats(mergedAtoms) {
1106
- const reconstructionStats = computeReconstructionStats(mergedAtoms);
1107
- let insertedRanges = 0;
1108
- let deletedRanges = 0;
1109
- let formatChanges = 0;
1110
- let previousRangeStatus = null;
1111
- let previousRangeParagraph;
1112
- const paragraphs = new Map();
1113
- for (const atom of mergedAtoms) {
1114
- const paragraphKey = paragraphStatsKey(atom);
1115
- const status = atom.correlationStatus;
1116
- const rangeStatus = status === CorrelationStatus.Inserted ||
1117
- status === CorrelationStatus.Deleted ||
1118
- status === CorrelationStatus.FormatChanged
1119
- ? status
1120
- : null;
1121
- if (rangeStatus) {
1122
- if (rangeStatus !== previousRangeStatus || paragraphKey !== previousRangeParagraph) {
1123
- if (rangeStatus === CorrelationStatus.Inserted)
1124
- insertedRanges++;
1125
- if (rangeStatus === CorrelationStatus.Deleted)
1126
- deletedRanges++;
1127
- if (rangeStatus === CorrelationStatus.FormatChanged)
1128
- formatChanges++;
1129
- }
1130
- previousRangeStatus = rangeStatus;
1131
- previousRangeParagraph = paragraphKey;
1132
- }
1133
- else {
1134
- previousRangeStatus = null;
1135
- previousRangeParagraph = undefined;
1136
- }
1137
- if (paragraphKey && (status === CorrelationStatus.Deleted || status === CorrelationStatus.Inserted)) {
1138
- const flags = paragraphs.get(paragraphKey) ?? { hasDeleted: false, hasInserted: false };
1139
- if (status === CorrelationStatus.Deleted)
1140
- flags.hasDeleted = true;
1141
- if (status === CorrelationStatus.Inserted)
1142
- flags.hasInserted = true;
1143
- paragraphs.set(paragraphKey, flags);
1144
- }
1145
- }
1146
- const modifiedParagraphs = Array.from(paragraphs.values()).filter((flags) => flags.hasDeleted && flags.hasInserted).length;
1147
- return {
1148
- insertions: insertedRanges,
1149
- deletions: deletedRanges,
1150
- modifications: modifiedParagraphs,
1151
- insertedRanges,
1152
- deletedRanges,
1153
- insertedAtoms: reconstructionStats.insertions,
1154
- deletedAtoms: reconstructionStats.deletions,
1155
- modifiedParagraphs,
1156
- formatChanges,
1157
- formatChangeAtoms: reconstructionStats.formatChanges,
1158
- };
1159
- }
1160
- //# sourceMappingURL=pipeline.js.map