@usejunior/docx-core 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -83
- package/dist/.tsbuildinfo +1 -1
- package/dist/cli/conformance-adapter.d.ts +14 -0
- package/dist/cli/conformance-adapter.d.ts.map +1 -1
- package/dist/cli/conformance-adapter.js +104 -12
- package/dist/cli/conformance-adapter.js.map +1 -1
- package/dist/footnotes.d.ts +7 -5
- package/dist/footnotes.d.ts.map +1 -1
- package/dist/footnotes.js +7 -5
- package/dist/footnotes.js.map +1 -1
- package/dist/generated/ecma-376-vocabulary.d.ts +93 -0
- package/dist/generated/ecma-376-vocabulary.d.ts.map +1 -0
- package/dist/generated/ecma-376-vocabulary.js +87 -0
- package/dist/generated/ecma-376-vocabulary.js.map +1 -0
- package/dist/generation/compile.js +2 -2
- package/dist/generation/compile.js.map +1 -1
- package/dist/generation/emit/comments-part.d.ts +2 -0
- package/dist/generation/emit/comments-part.d.ts.map +1 -1
- package/dist/generation/emit/comments-part.js +2 -0
- package/dist/generation/emit/comments-part.js.map +1 -1
- package/dist/generation/emit/paragraph.d.ts +3 -0
- package/dist/generation/emit/paragraph.d.ts.map +1 -1
- package/dist/generation/emit/paragraph.js +3 -0
- package/dist/generation/emit/paragraph.js.map +1 -1
- package/dist/generation/emit/run.d.ts.map +1 -1
- package/dist/generation/emit/run.js +6 -1
- package/dist/generation/emit/run.js.map +1 -1
- package/dist/generation/emit/settings-part.d.ts +12 -3
- package/dist/generation/emit/settings-part.d.ts.map +1 -1
- package/dist/generation/emit/settings-part.js +23 -5
- package/dist/generation/emit/settings-part.js.map +1 -1
- package/dist/generation/ordering.d.ts +3 -1
- package/dist/generation/ordering.d.ts.map +1 -1
- package/dist/generation/ordering.js +3 -1
- package/dist/generation/ordering.js.map +1 -1
- package/dist/generation/schema-enum-domains.d.ts +27 -0
- package/dist/generation/schema-enum-domains.d.ts.map +1 -0
- package/dist/generation/schema-enum-domains.js +69 -0
- package/dist/generation/schema-enum-domains.js.map +1 -0
- package/dist/generation/structural-checks.js +7 -1
- package/dist/generation/structural-checks.js.map +1 -1
- package/dist/generation/validate-spec.d.ts.map +1 -1
- package/dist/generation/validate-spec.js +149 -31
- package/dist/generation/validate-spec.js.map +1 -1
- package/dist/index.d.ts +7 -24
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -46
- package/dist/index.js.map +1 -1
- package/dist/primitives/accept_ai_edits.d.ts +87 -0
- package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
- package/dist/primitives/accept_ai_edits.js +253 -0
- package/dist/primitives/accept_ai_edits.js.map +1 -0
- package/dist/primitives/accept_changes.d.ts +36 -3
- package/dist/primitives/accept_changes.d.ts.map +1 -1
- package/dist/primitives/accept_changes.js +67 -31
- package/dist/primitives/accept_changes.js.map +1 -1
- package/dist/primitives/bookmarks.d.ts +44 -0
- package/dist/primitives/bookmarks.d.ts.map +1 -1
- package/dist/primitives/bookmarks.js +149 -11
- package/dist/primitives/bookmarks.js.map +1 -1
- package/dist/primitives/document.d.ts +53 -1
- package/dist/primitives/document.d.ts.map +1 -1
- package/dist/primitives/document.js +146 -2
- package/dist/primitives/document.js.map +1 -1
- package/dist/primitives/dom-helpers.d.ts.map +1 -1
- package/dist/primitives/dom-helpers.js +7 -2
- package/dist/primitives/dom-helpers.js.map +1 -1
- package/dist/primitives/index.d.ts +2 -1
- package/dist/primitives/index.d.ts.map +1 -1
- package/dist/primitives/index.js +2 -1
- package/dist/primitives/index.js.map +1 -1
- package/dist/primitives/layout.d.ts.map +1 -1
- package/dist/primitives/layout.js +41 -1
- package/dist/primitives/layout.js.map +1 -1
- package/dist/primitives/namespaces.d.ts +2 -0
- package/dist/primitives/namespaces.d.ts.map +1 -1
- package/dist/primitives/namespaces.js +2 -0
- package/dist/primitives/namespaces.js.map +1 -1
- package/dist/primitives/reject_changes.d.ts +21 -2
- package/dist/primitives/reject_changes.d.ts.map +1 -1
- package/dist/primitives/reject_changes.js +71 -31
- package/dist/primitives/reject_changes.js.map +1 -1
- package/dist/primitives/relationships.d.ts +33 -0
- package/dist/primitives/relationships.d.ts.map +1 -1
- package/dist/primitives/relationships.js +84 -1
- package/dist/primitives/relationships.js.map +1 -1
- package/dist/primitives/sectPrAudit.d.ts +11 -2
- package/dist/primitives/sectPrAudit.d.ts.map +1 -1
- package/dist/primitives/sectPrAudit.js +148 -23
- package/dist/primitives/sectPrAudit.js.map +1 -1
- package/dist/primitives/track-changes-emitter.d.ts +4 -0
- package/dist/primitives/track-changes-emitter.d.ts.map +1 -1
- package/dist/primitives/track-changes-emitter.js +5 -2
- package/dist/primitives/track-changes-emitter.js.map +1 -1
- package/dist/primitives/validate_ai_revisions.d.ts.map +1 -1
- package/dist/primitives/validate_ai_revisions.js +13 -5
- package/dist/primitives/validate_ai_revisions.js.map +1 -1
- package/dist/primitives/zip.d.ts.map +1 -1
- package/dist/primitives/zip.js +6 -2
- package/dist/primitives/zip.js.map +1 -1
- package/dist/shared/field-structure.d.ts +6 -0
- package/dist/shared/field-structure.d.ts.map +1 -1
- package/dist/shared/field-structure.js +6 -0
- package/dist/shared/field-structure.js.map +1 -1
- package/package.json +4 -8
- package/dist/atomizer.d.ts +0 -273
- package/dist/atomizer.d.ts.map +0 -1
- package/dist/atomizer.js +0 -1002
- package/dist/atomizer.js.map +0 -1
- package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
- package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/atomLcs.js +0 -376
- package/dist/baselines/atomizer/atomLcs.js.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
- package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
- package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
- package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
- package/dist/baselines/atomizer/debug.d.ts +0 -41
- package/dist/baselines/atomizer/debug.d.ts.map +0 -1
- package/dist/baselines/atomizer/debug.js +0 -85
- package/dist/baselines/atomizer/debug.js.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
- package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
- package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
- package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.js +0 -449
- package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
- package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
- package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
- package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
- package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
- package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
- package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.js +0 -209
- package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
- package/dist/baselines/atomizer/pipeline.d.ts +0 -103
- package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
- package/dist/baselines/atomizer/pipeline.js +0 -1160
- package/dist/baselines/atomizer/pipeline.js.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
- package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.js +0 -153
- package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
- package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
- package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
- package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
- package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
- package/dist/cli/compare-two.d.ts +0 -28
- package/dist/cli/compare-two.d.ts.map +0 -1
- package/dist/cli/compare-two.js +0 -112
- package/dist/cli/compare-two.js.map +0 -1
- package/dist/cli/index.d.ts +0 -3
- package/dist/cli/index.d.ts.map +0 -1
- package/dist/cli/index.js +0 -25
- package/dist/cli/index.js.map +0 -1
- package/dist/compare-types.d.ts +0 -197
- package/dist/compare-types.d.ts.map +0 -1
- package/dist/compare-types.js +0 -2
- package/dist/compare-types.js.map +0 -1
- package/dist/format-detection.d.ts +0 -120
- package/dist/format-detection.d.ts.map +0 -1
- package/dist/format-detection.js +0 -339
- package/dist/format-detection.js.map +0 -1
- package/dist/move-detection.d.ts +0 -211
- package/dist/move-detection.d.ts.map +0 -1
- package/dist/move-detection.js +0 -390
- package/dist/move-detection.js.map +0 -1
|
@@ -1,1160 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Atomizer Pipeline
|
|
3
|
-
*
|
|
4
|
-
* Main orchestration for the atomizer-based document comparison.
|
|
5
|
-
* Integrates atomization, LCS comparison, move detection, format detection,
|
|
6
|
-
* and document reconstruction.
|
|
7
|
-
*/
|
|
8
|
-
import { XMLSerializer } from '@xmldom/xmldom';
|
|
9
|
-
import { parseXml } from '../../primitives/xml.js';
|
|
10
|
-
import { DocxArchive } from '../../shared/docx/DocxArchive.js';
|
|
11
|
-
import { DEFAULT_MOVE_DETECTION_SETTINGS, DEFAULT_FORMAT_DETECTION_SETTINGS, CorrelationStatus, } from '../../core-types.js';
|
|
12
|
-
import { atomizeTree, assignParagraphIndices } from '../../atomizer.js';
|
|
13
|
-
import { detectMovesInAtomList } from '../../move-detection.js';
|
|
14
|
-
import { detectFormatChangesInAtomList } from '../../format-detection.js';
|
|
15
|
-
import { parseDocumentXml, findBody, backfillParentReferences, } from './xmlToWmlElement.js';
|
|
16
|
-
import { findAllByTagName, getLeafText } from '../../primitives/index.js';
|
|
17
|
-
import { createMergedAtomList, assignUnifiedParagraphIndices, } from './atomLcs.js';
|
|
18
|
-
import { hierarchicalCompare, markHierarchicalCorrelationStatus, } from './hierarchicalLcs.js';
|
|
19
|
-
import { reconstructDocument, computeReconstructionStats, } from './documentReconstructor.js';
|
|
20
|
-
import { modifyRevisedDocument, ContainerResolutionError } from './inPlaceModifier.js';
|
|
21
|
-
import { acceptAllChanges, rejectAllChanges, extractTextWithParagraphs, compareTexts, } from './trackChangesAcceptorAst.js';
|
|
22
|
-
import { virtualizeNumberingLabels, DEFAULT_NUMBERING_OPTIONS, } from './numberingIntegration.js';
|
|
23
|
-
import { premergeAdjacentRuns } from './premergeRuns.js';
|
|
24
|
-
export { hasFldCharInsideDel, validateFieldStructure, } from '../../shared/field-structure.js';
|
|
25
|
-
import { hasFldCharInsideDel, validateFieldStructure, } from '../../shared/field-structure.js';
|
|
26
|
-
import { AUXILIARY_PARTS, parseEntries, renumberCollidingAuxiliaryIds, restampCollidingCommentParaIds, } from './auxiliaryIdCollision.js';
|
|
27
|
-
import { maybeCaptureEmittedDocumentXml } from '../../primitives/schema-corpus-capture.js';
|
|
28
|
-
function arraysEqual(a, b) {
|
|
29
|
-
if (a.length !== b.length)
|
|
30
|
-
return false;
|
|
31
|
-
for (let i = 0; i < a.length; i++) {
|
|
32
|
-
if (a[i] !== b[i])
|
|
33
|
-
return false;
|
|
34
|
-
}
|
|
35
|
-
return true;
|
|
36
|
-
}
|
|
37
|
-
function collectReferencedBookmarkNames(root) {
|
|
38
|
-
const refs = new Set();
|
|
39
|
-
const refRegex = /\b(?:PAGEREF|REF)\s+([^\s\\]+)/g;
|
|
40
|
-
for (const node of findAllByTagName(root, 'w:instrText')) {
|
|
41
|
-
const instr = getLeafText(node) ?? '';
|
|
42
|
-
for (const match of instr.matchAll(refRegex)) {
|
|
43
|
-
const name = match[1]?.trim();
|
|
44
|
-
if (name)
|
|
45
|
-
refs.add(name);
|
|
46
|
-
}
|
|
47
|
-
}
|
|
48
|
-
return Array.from(refs).sort();
|
|
49
|
-
}
|
|
50
|
-
function collectBookmarkDiagnostics(documentXml) {
|
|
51
|
-
const root = parseDocumentXml(documentXml);
|
|
52
|
-
const startSet = new Set();
|
|
53
|
-
const endSet = new Set();
|
|
54
|
-
const startNameSet = new Set();
|
|
55
|
-
const duplicateStartSet = new Set();
|
|
56
|
-
const duplicateEndSet = new Set();
|
|
57
|
-
const duplicateStartNameSet = new Set();
|
|
58
|
-
for (const node of findAllByTagName(root, 'w:bookmarkStart')) {
|
|
59
|
-
const id = node.getAttribute('w:id');
|
|
60
|
-
if (!id)
|
|
61
|
-
continue;
|
|
62
|
-
if (startSet.has(id))
|
|
63
|
-
duplicateStartSet.add(id);
|
|
64
|
-
else
|
|
65
|
-
startSet.add(id);
|
|
66
|
-
const name = node.getAttribute('w:name');
|
|
67
|
-
if (name) {
|
|
68
|
-
if (startNameSet.has(name))
|
|
69
|
-
duplicateStartNameSet.add(name);
|
|
70
|
-
else
|
|
71
|
-
startNameSet.add(name);
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
for (const node of findAllByTagName(root, 'w:bookmarkEnd')) {
|
|
75
|
-
const id = node.getAttribute('w:id');
|
|
76
|
-
if (!id)
|
|
77
|
-
continue;
|
|
78
|
-
if (endSet.has(id))
|
|
79
|
-
duplicateEndSet.add(id);
|
|
80
|
-
else
|
|
81
|
-
endSet.add(id);
|
|
82
|
-
}
|
|
83
|
-
const startIds = Array.from(startSet).sort();
|
|
84
|
-
const endIds = Array.from(endSet).sort();
|
|
85
|
-
const startNames = Array.from(startNameSet).sort();
|
|
86
|
-
const referencedBookmarkNames = collectReferencedBookmarkNames(root);
|
|
87
|
-
const unresolvedReferenceNames = referencedBookmarkNames
|
|
88
|
-
.filter((name) => !startNameSet.has(name))
|
|
89
|
-
.sort();
|
|
90
|
-
const unmatchedStartIds = startIds.filter((id) => !endSet.has(id));
|
|
91
|
-
const unmatchedEndIds = endIds.filter((id) => !startSet.has(id));
|
|
92
|
-
return {
|
|
93
|
-
startIds,
|
|
94
|
-
endIds,
|
|
95
|
-
startNames,
|
|
96
|
-
duplicateStartNames: Array.from(duplicateStartNameSet).sort(),
|
|
97
|
-
referencedBookmarkNames,
|
|
98
|
-
unresolvedReferenceNames,
|
|
99
|
-
duplicateStartIds: Array.from(duplicateStartSet).sort(),
|
|
100
|
-
duplicateEndIds: Array.from(duplicateEndSet).sort(),
|
|
101
|
-
unmatchedStartIds,
|
|
102
|
-
unmatchedEndIds,
|
|
103
|
-
};
|
|
104
|
-
}
|
|
105
|
-
/**
|
|
106
|
-
* Bookmark round-trip safety is semantic, not byte/ID exact:
|
|
107
|
-
* - Bookmark IDs may be renumbered by reconstruction/Word and still be valid.
|
|
108
|
-
* - Bookmark names and field-reference targets must stay intact.
|
|
109
|
-
* - Structural integrity (balanced, no duplicates) must remain intact.
|
|
110
|
-
*/
|
|
111
|
-
function bookmarkDiagnosticsSemanticallyEqual(expected, actual) {
|
|
112
|
-
return (arraysEqual(expected.startNames, actual.startNames) &&
|
|
113
|
-
arraysEqual(expected.duplicateStartNames, actual.duplicateStartNames) &&
|
|
114
|
-
arraysEqual(expected.referencedBookmarkNames, actual.referencedBookmarkNames) &&
|
|
115
|
-
arraysEqual(expected.unresolvedReferenceNames, actual.unresolvedReferenceNames) &&
|
|
116
|
-
arraysEqual(expected.duplicateStartIds, actual.duplicateStartIds) &&
|
|
117
|
-
arraysEqual(expected.duplicateEndIds, actual.duplicateEndIds) &&
|
|
118
|
-
arraysEqual(expected.unmatchedStartIds, actual.unmatchedStartIds) &&
|
|
119
|
-
arraysEqual(expected.unmatchedEndIds, actual.unmatchedEndIds));
|
|
120
|
-
}
|
|
121
|
-
function diffIds(expected, actual) {
|
|
122
|
-
const expectedSet = new Set(expected);
|
|
123
|
-
const actualSet = new Set(actual);
|
|
124
|
-
const missing = expected.filter((id) => !actualSet.has(id));
|
|
125
|
-
const unexpected = actual.filter((id) => !expectedSet.has(id));
|
|
126
|
-
return { missing, unexpected };
|
|
127
|
-
}
|
|
128
|
-
function buildTextMismatchDetails(expectedText, actualText) {
|
|
129
|
-
const comparison = compareTexts(expectedText, actualText);
|
|
130
|
-
const expectedParas = expectedText.split('\n');
|
|
131
|
-
const actualParas = actualText.split('\n');
|
|
132
|
-
const maxLen = Math.max(expectedParas.length, actualParas.length);
|
|
133
|
-
let firstDifferingParagraphIndex = -1;
|
|
134
|
-
for (let i = 0; i < maxLen; i++) {
|
|
135
|
-
if ((expectedParas[i] ?? '') !== (actualParas[i] ?? '')) {
|
|
136
|
-
firstDifferingParagraphIndex = i;
|
|
137
|
-
break;
|
|
138
|
-
}
|
|
139
|
-
}
|
|
140
|
-
return {
|
|
141
|
-
expectedLength: comparison.expectedLength,
|
|
142
|
-
actualLength: comparison.actualLength,
|
|
143
|
-
firstDifferingParagraphIndex,
|
|
144
|
-
expectedParagraph: firstDifferingParagraphIndex >= 0 ? (expectedParas[firstDifferingParagraphIndex] ?? '') : '',
|
|
145
|
-
actualParagraph: firstDifferingParagraphIndex >= 0 ? (actualParas[firstDifferingParagraphIndex] ?? '') : '',
|
|
146
|
-
differenceSample: comparison.differences.slice(0, 3),
|
|
147
|
-
};
|
|
148
|
-
}
|
|
149
|
-
function buildBookmarkMismatchDetails(expected, actual) {
|
|
150
|
-
return {
|
|
151
|
-
startNames: diffIds(expected.startNames, actual.startNames),
|
|
152
|
-
referencedBookmarkNames: diffIds(expected.referencedBookmarkNames, actual.referencedBookmarkNames),
|
|
153
|
-
unresolvedReferenceNames: diffIds(expected.unresolvedReferenceNames, actual.unresolvedReferenceNames),
|
|
154
|
-
startIds: diffIds(expected.startIds, actual.startIds),
|
|
155
|
-
endIds: diffIds(expected.endIds, actual.endIds),
|
|
156
|
-
expectedDuplicateStartNames: expected.duplicateStartNames,
|
|
157
|
-
actualDuplicateStartNames: actual.duplicateStartNames,
|
|
158
|
-
expectedDuplicateStartIds: expected.duplicateStartIds,
|
|
159
|
-
actualDuplicateStartIds: actual.duplicateStartIds,
|
|
160
|
-
expectedDuplicateEndIds: expected.duplicateEndIds,
|
|
161
|
-
actualDuplicateEndIds: actual.duplicateEndIds,
|
|
162
|
-
expectedUnmatchedStartIds: expected.unmatchedStartIds,
|
|
163
|
-
actualUnmatchedStartIds: actual.unmatchedStartIds,
|
|
164
|
-
expectedUnmatchedEndIds: expected.unmatchedEndIds,
|
|
165
|
-
actualUnmatchedEndIds: actual.unmatchedEndIds,
|
|
166
|
-
};
|
|
167
|
-
}
|
|
168
|
-
function summarizeIdDelta(delta) {
|
|
169
|
-
return {
|
|
170
|
-
missingCount: delta.missing.length,
|
|
171
|
-
unexpectedCount: delta.unexpected.length,
|
|
172
|
-
firstMissing: delta.missing[0],
|
|
173
|
-
firstUnexpected: delta.unexpected[0],
|
|
174
|
-
};
|
|
175
|
-
}
|
|
176
|
-
function truncateForSummary(value, maxLength = 160) {
|
|
177
|
-
if (value.length <= maxLength) {
|
|
178
|
-
return value;
|
|
179
|
-
}
|
|
180
|
-
return `${value.slice(0, maxLength)}...`;
|
|
181
|
-
}
|
|
182
|
-
function summarizeTextMismatch(details) {
|
|
183
|
-
return {
|
|
184
|
-
firstDifferingParagraphIndex: details.firstDifferingParagraphIndex,
|
|
185
|
-
expectedParagraph: truncateForSummary(details.expectedParagraph),
|
|
186
|
-
actualParagraph: truncateForSummary(details.actualParagraph),
|
|
187
|
-
firstDifference: details.differenceSample[0] ?? 'No diff sample',
|
|
188
|
-
};
|
|
189
|
-
}
|
|
190
|
-
function summarizeBookmarkMismatch(details) {
|
|
191
|
-
return {
|
|
192
|
-
startNames: summarizeIdDelta(details.startNames),
|
|
193
|
-
referencedBookmarkNames: summarizeIdDelta(details.referencedBookmarkNames),
|
|
194
|
-
unresolvedReferenceNames: summarizeIdDelta(details.unresolvedReferenceNames),
|
|
195
|
-
startIds: summarizeIdDelta(details.startIds),
|
|
196
|
-
endIds: summarizeIdDelta(details.endIds),
|
|
197
|
-
unmatchedStartCount: details.actualUnmatchedStartIds.length,
|
|
198
|
-
unmatchedEndCount: details.actualUnmatchedEndIds.length,
|
|
199
|
-
firstUnmatchedStartId: details.actualUnmatchedStartIds[0],
|
|
200
|
-
firstUnmatchedEndId: details.actualUnmatchedEndIds[0],
|
|
201
|
-
};
|
|
202
|
-
}
|
|
203
|
-
function buildFailureSummary(failureDetails) {
|
|
204
|
-
if (!failureDetails) {
|
|
205
|
-
return undefined;
|
|
206
|
-
}
|
|
207
|
-
const summary = {};
|
|
208
|
-
if (failureDetails.acceptText) {
|
|
209
|
-
summary.acceptText = summarizeTextMismatch(failureDetails.acceptText);
|
|
210
|
-
}
|
|
211
|
-
if (failureDetails.rejectText) {
|
|
212
|
-
summary.rejectText = summarizeTextMismatch(failureDetails.rejectText);
|
|
213
|
-
}
|
|
214
|
-
if (failureDetails.acceptBookmarks) {
|
|
215
|
-
summary.acceptBookmarks = summarizeBookmarkMismatch(failureDetails.acceptBookmarks);
|
|
216
|
-
}
|
|
217
|
-
if (failureDetails.rejectBookmarks) {
|
|
218
|
-
summary.rejectBookmarks = summarizeBookmarkMismatch(failureDetails.rejectBookmarks);
|
|
219
|
-
}
|
|
220
|
-
return Object.keys(summary).length > 0 ? summary : undefined;
|
|
221
|
-
}
|
|
222
|
-
// Declared above splitStories so the function body never observes an
|
|
223
|
-
// uninitialized binding under circular imports.
|
|
224
|
-
const serializer = new XMLSerializer();
|
|
225
|
-
/**
|
|
226
|
-
* Split a docx into per-story XML fragments for field-closure validation.
|
|
227
|
-
*
|
|
228
|
-
* Each footnote/endnote entry is treated as an isolated story: a complex
|
|
229
|
-
* field whose `begin` and `end` markers straddle stories breaks Word's
|
|
230
|
-
* field state machine. We therefore validate each `<w:footnote>` and
|
|
231
|
-
* `<w:endnote>` entry independently rather than treating the whole
|
|
232
|
-
* `footnotes.xml`/`endnotes.xml` as one stream.
|
|
233
|
-
*
|
|
234
|
-
* Accepts arrays of sidecar XMLs (one per source archive) so callers can
|
|
235
|
-
* validate the union of entries from every archive that may contribute to the
|
|
236
|
-
* final result. Step 12 of `compareDocumentsAtomizer` merges entries from a
|
|
237
|
-
* mode-dependent source archive into the base archive; passing both archives'
|
|
238
|
-
* sidecars guarantees that whichever path the merge takes, the entries it
|
|
239
|
-
* could publish have already been screened. Duplicates (same `w:id` in both
|
|
240
|
-
* archives) yield redundant but harmless validation work.
|
|
241
|
-
*
|
|
242
|
-
* Header/footer stories are not yet covered — they require relationship
|
|
243
|
-
* walking to enumerate `headerN.xml`/`footerN.xml`.
|
|
244
|
-
*
|
|
245
|
-
* @conformance ECMA-376 edition 5, Part 4 § 17.16.5
|
|
246
|
-
* @see https://github.com/UseJunior/safe-docx/issues/212
|
|
247
|
-
*/
|
|
248
|
-
export function splitStories(documentXml, footnotesXmls, endnotesXmls) {
|
|
249
|
-
const stories = [{ label: 'document', xml: documentXml }];
|
|
250
|
-
const collectEntries = (sidecars, entryTag, labelPrefix) => {
|
|
251
|
-
for (let s = 0; s < sidecars.length; s++) {
|
|
252
|
-
const sidecarXml = sidecars[s];
|
|
253
|
-
if (!sidecarXml)
|
|
254
|
-
continue;
|
|
255
|
-
const doc = parseXml(sidecarXml);
|
|
256
|
-
const entries = doc.getElementsByTagName(entryTag);
|
|
257
|
-
for (let i = 0; i < entries.length; i++) {
|
|
258
|
-
const entry = entries[i];
|
|
259
|
-
const id = entry.getAttribute('w:id') ?? String(i);
|
|
260
|
-
stories.push({
|
|
261
|
-
label: `${labelPrefix}[${s}]:${id}`,
|
|
262
|
-
xml: serializer.serializeToString(entry),
|
|
263
|
-
});
|
|
264
|
-
}
|
|
265
|
-
}
|
|
266
|
-
};
|
|
267
|
-
collectEntries(footnotesXmls, 'w:footnote', 'footnote');
|
|
268
|
-
collectEntries(endnotesXmls, 'w:endnote', 'endnote');
|
|
269
|
-
return stories;
|
|
270
|
-
}
|
|
271
|
-
function evaluateSafetyChecks(originalTextForRoundTrip, revisedTextForRoundTrip, originalBookmarkDiagnostics, revisedBookmarkDiagnostics, candidateXml, auxiliarySidecars) {
|
|
272
|
-
const acceptedXml = acceptAllChanges(candidateXml);
|
|
273
|
-
const rejectedXml = rejectAllChanges(candidateXml);
|
|
274
|
-
const acceptedText = extractTextWithParagraphs(acceptedXml);
|
|
275
|
-
const rejectedText = extractTextWithParagraphs(rejectedXml);
|
|
276
|
-
const acceptedBookmarkDiagnostics = collectBookmarkDiagnostics(acceptedXml);
|
|
277
|
-
const rejectedBookmarkDiagnostics = collectBookmarkDiagnostics(rejectedXml);
|
|
278
|
-
const acceptTextComparison = compareTexts(revisedTextForRoundTrip, acceptedText);
|
|
279
|
-
const rejectTextComparison = compareTexts(originalTextForRoundTrip, rejectedText);
|
|
280
|
-
const acceptBookmarksOk = bookmarkDiagnosticsSemanticallyEqual(revisedBookmarkDiagnostics, acceptedBookmarkDiagnostics);
|
|
281
|
-
const rejectBookmarksOk = bookmarkDiagnosticsSemanticallyEqual(originalBookmarkDiagnostics, rejectedBookmarkDiagnostics);
|
|
282
|
-
// Validate field structure per-story. Each footnote/endnote entry is its own
|
|
283
|
-
// ECMA-376 story; a complex field that crosses a story boundary breaks
|
|
284
|
-
// Word's field state machine even when global begin/end counts balance.
|
|
285
|
-
// Sidecars from BOTH archives are validated because Step 12's auxiliary-part
|
|
286
|
-
// merge picks its base and source archives by reconstruction mode (inplace
|
|
287
|
-
// base = revised; rebuild base = original) and validating only one side
|
|
288
|
-
// would miss field issues that would still ship in the merged result.
|
|
289
|
-
// `acceptAllChanges` / `rejectAllChanges` only transform document.xml, so
|
|
290
|
-
// the sidecar set is identical for both transforms.
|
|
291
|
-
const acceptedStories = splitStories(acceptedXml, auxiliarySidecars.footnotesXmls, auxiliarySidecars.endnotesXmls);
|
|
292
|
-
const rejectedStories = splitStories(rejectedXml, auxiliarySidecars.footnotesXmls, auxiliarySidecars.endnotesXmls);
|
|
293
|
-
// Issue #217 conformance gate on the COMBINED output: w:fldChar MUST NOT
|
|
294
|
-
// appear inside <w:del>. ECMA-376 Part 4 § 17.16.5 makes this fatal for
|
|
295
|
-
// Word's field state machine. The full validateFieldStructure check is run
|
|
296
|
-
// on the accept/reject projections (per-story); on the combined view we
|
|
297
|
-
// only gate the strict no-fldChar-in-del rule because some legacy emit
|
|
298
|
-
// paths (e.g. delInstrText inside <w:moveFrom>) are non-conformant in shape
|
|
299
|
-
// but out of scope for #217.
|
|
300
|
-
const combinedNoFldCharInDel = !hasFldCharInsideDel(candidateXml);
|
|
301
|
-
const fieldStructureOk = combinedNoFldCharInDel &&
|
|
302
|
-
validateFieldStructure(acceptedStories) &&
|
|
303
|
-
validateFieldStructure(rejectedStories);
|
|
304
|
-
const checks = {
|
|
305
|
-
acceptText: acceptTextComparison.normalizedIdentical,
|
|
306
|
-
rejectText: rejectTextComparison.normalizedIdentical,
|
|
307
|
-
// Bookmark checks are soft: consumer compatibility pass legitimately alters
|
|
308
|
-
// bookmarks (deduplication, orphan repair, hoisting out of revision wrappers).
|
|
309
|
-
// Log mismatches in diagnostics but don't trigger fallback to rebuild.
|
|
310
|
-
acceptBookmarks: true,
|
|
311
|
-
rejectBookmarks: true,
|
|
312
|
-
fieldStructure: fieldStructureOk,
|
|
313
|
-
};
|
|
314
|
-
const failedChecks = Object.entries(checks)
|
|
315
|
-
.filter(([, ok]) => !ok)
|
|
316
|
-
.map(([name]) => name);
|
|
317
|
-
const failureDetails = {};
|
|
318
|
-
if (!checks.acceptText) {
|
|
319
|
-
failureDetails.acceptText = buildTextMismatchDetails(revisedTextForRoundTrip, acceptedText);
|
|
320
|
-
}
|
|
321
|
-
if (!checks.rejectText) {
|
|
322
|
-
failureDetails.rejectText = buildTextMismatchDetails(originalTextForRoundTrip, rejectedText);
|
|
323
|
-
}
|
|
324
|
-
// Bookmark mismatches are always collected for diagnostics even though the
|
|
325
|
-
// check itself is soft (doesn't trigger fallback).
|
|
326
|
-
if (!acceptBookmarksOk) {
|
|
327
|
-
failureDetails.acceptBookmarks = buildBookmarkMismatchDetails(revisedBookmarkDiagnostics, acceptedBookmarkDiagnostics);
|
|
328
|
-
}
|
|
329
|
-
if (!rejectBookmarksOk) {
|
|
330
|
-
failureDetails.rejectBookmarks = buildBookmarkMismatchDetails(originalBookmarkDiagnostics, rejectedBookmarkDiagnostics);
|
|
331
|
-
}
|
|
332
|
-
return {
|
|
333
|
-
safe: failedChecks.length === 0,
|
|
334
|
-
checks,
|
|
335
|
-
failedChecks,
|
|
336
|
-
failureDetails: failedChecks.length > 0 ? failureDetails : undefined,
|
|
337
|
-
failureSummary: failedChecks.length > 0 ? buildFailureSummary(failureDetails) : undefined,
|
|
338
|
-
};
|
|
339
|
-
}
|
|
340
|
-
/**
|
|
341
|
-
* Compare two DOCX documents using the atomizer-based approach.
|
|
342
|
-
*
|
|
343
|
-
* Pipeline steps:
|
|
344
|
-
* 1. Load DOCX archives
|
|
345
|
-
* 2. Extract document.xml
|
|
346
|
-
* 3. Parse to WmlElement trees
|
|
347
|
-
* 4. Atomize both documents
|
|
348
|
-
* 5. (Optional) Apply numbering virtualization
|
|
349
|
-
* 6. Run LCS on atom hashes
|
|
350
|
-
* 7. Mark correlation status
|
|
351
|
-
* 8. Run move detection
|
|
352
|
-
* 9. Run format detection
|
|
353
|
-
* 10. Reconstruct document with track changes
|
|
354
|
-
* 11. Save and return result
|
|
355
|
-
*
|
|
356
|
-
* @param original - Original document as Buffer
|
|
357
|
-
* @param revised - Revised document as Buffer
|
|
358
|
-
* @param options - Pipeline options
|
|
359
|
-
* @returns Comparison result with track changes document
|
|
360
|
-
*/
|
|
361
|
-
export async function compareDocumentsAtomizer(original, revised, options = {}) {
|
|
362
|
-
const { author = 'Comparison', date = new Date(), moveDetection = {}, formatDetection = {}, numbering = {}, premergeRuns = true, reconstructionMode = 'rebuild', } = options;
|
|
363
|
-
// Merge settings with defaults
|
|
364
|
-
const moveSettings = {
|
|
365
|
-
...DEFAULT_MOVE_DETECTION_SETTINGS,
|
|
366
|
-
...moveDetection,
|
|
367
|
-
};
|
|
368
|
-
const formatSettings = {
|
|
369
|
-
...DEFAULT_FORMAT_DETECTION_SETTINGS,
|
|
370
|
-
...formatDetection,
|
|
371
|
-
};
|
|
372
|
-
const numberingSettings = {
|
|
373
|
-
...DEFAULT_NUMBERING_OPTIONS,
|
|
374
|
-
...numbering,
|
|
375
|
-
};
|
|
376
|
-
// Step 1: Load DOCX archives
|
|
377
|
-
const originalArchive = await DocxArchive.load(original);
|
|
378
|
-
const revisedArchive = await DocxArchive.load(revised);
|
|
379
|
-
// Step 1b: Resolve auxiliary ID collisions. When both sides define
|
|
380
|
-
// different content under the same comment/footnote/endnote w:id or the
|
|
381
|
-
// same comment paraId, rewrite the revised side so no anchor or ancillary
|
|
382
|
-
// row in the merged output can bind to the other document's definition.
|
|
383
|
-
// Must run before any document.xml extraction so every downstream step sees
|
|
384
|
-
// the rewritten archive.
|
|
385
|
-
await renumberCollidingAuxiliaryIds(originalArchive, revisedArchive);
|
|
386
|
-
await restampCollidingCommentParaIds(originalArchive, revisedArchive);
|
|
387
|
-
// Step 2: Extract document.xml
|
|
388
|
-
const originalXml = await originalArchive.getDocumentXml();
|
|
389
|
-
const revisedXml = await revisedArchive.getDocumentXml();
|
|
390
|
-
// Extract numbering.xml if available
|
|
391
|
-
const originalNumberingXml = await originalArchive.getNumberingXml() ?? undefined;
|
|
392
|
-
const revisedNumberingXml = await revisedArchive.getNumberingXml() ?? undefined;
|
|
393
|
-
// Extract footnote/endnote sidecars from BOTH archives for per-story
|
|
394
|
-
// field-closure validation (issue #212). Step 12 picks the base archive by
|
|
395
|
-
// reconstruction mode (inplace = revised, rebuild = original) and merges
|
|
396
|
-
// missing referenced entries from the opposite archive. Validating both
|
|
397
|
-
// archives' sidecars covers the union of entries that could ship without
|
|
398
|
-
// having to duplicate the merge logic at safety-check time.
|
|
399
|
-
const [originalFootnotesXml, originalEndnotesXml, revisedFootnotesXml, revisedEndnotesXml,] = await Promise.all([
|
|
400
|
-
originalArchive.getFile('word/footnotes.xml'),
|
|
401
|
-
originalArchive.getFile('word/endnotes.xml'),
|
|
402
|
-
revisedArchive.getFile('word/footnotes.xml'),
|
|
403
|
-
revisedArchive.getFile('word/endnotes.xml'),
|
|
404
|
-
]);
|
|
405
|
-
const auxiliarySidecars = {
|
|
406
|
-
footnotesXmls: [originalFootnotesXml, revisedFootnotesXml],
|
|
407
|
-
endnotesXmls: [originalEndnotesXml, revisedEndnotesXml],
|
|
408
|
-
};
|
|
409
|
-
const originalPart = {
|
|
410
|
-
uri: 'word/document.xml',
|
|
411
|
-
contentType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml',
|
|
412
|
-
};
|
|
413
|
-
const revisedPart = {
|
|
414
|
-
uri: 'word/document.xml',
|
|
415
|
-
contentType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml',
|
|
416
|
-
};
|
|
417
|
-
// Project each input through the SAME accept/reject operation the candidate is
|
|
418
|
-
// checked under, so the round-trip comparison is like-for-like even when an
|
|
419
|
-
// input already carries its own tracked changes (pre-tracked w:ins / w:del,
|
|
420
|
-
// comment anchors, multi-author stacks). For a clean input these equal the raw
|
|
421
|
-
// extraction, so behavior on the common case is unchanged. (#347)
|
|
422
|
-
const originalTextForRoundTrip = extractTextWithParagraphs(rejectAllChanges(originalXml));
|
|
423
|
-
const revisedTextForRoundTrip = extractTextWithParagraphs(acceptAllChanges(revisedXml));
|
|
424
|
-
const originalBookmarkDiagnostics = collectBookmarkDiagnostics(originalXml);
|
|
425
|
-
const revisedBookmarkDiagnostics = collectBookmarkDiagnostics(revisedXml);
|
|
426
|
-
const runComparisonPass = (atomizeOptions, outputMode) => {
|
|
427
|
-
// Parse fresh trees for each pass because inplace reconstruction mutates revised AST.
|
|
428
|
-
const originalTree = parseDocumentXml(originalXml);
|
|
429
|
-
const revisedTree = parseDocumentXml(revisedXml);
|
|
430
|
-
backfillParentReferences(originalTree);
|
|
431
|
-
backfillParentReferences(revisedTree);
|
|
432
|
-
const originalBody = findBody(originalTree);
|
|
433
|
-
const revisedBody = findBody(revisedTree);
|
|
434
|
-
if (!originalBody || !revisedBody) {
|
|
435
|
-
throw new Error('Could not find w:body in one or both documents');
|
|
436
|
-
}
|
|
437
|
-
if (premergeRuns) {
|
|
438
|
-
premergeAdjacentRuns(originalBody);
|
|
439
|
-
premergeAdjacentRuns(revisedBody);
|
|
440
|
-
}
|
|
441
|
-
const { atoms: originalAtoms } = atomizeTree(originalBody, [], originalPart, atomizeOptions);
|
|
442
|
-
const { atoms: revisedAtoms } = atomizeTree(revisedBody, [], revisedPart, atomizeOptions);
|
|
443
|
-
// Assign paragraph indices for proper grouping during reconstruction
|
|
444
|
-
assignParagraphIndices(originalAtoms);
|
|
445
|
-
assignParagraphIndices(revisedAtoms);
|
|
446
|
-
// Step 5: Apply numbering virtualization (optional)
|
|
447
|
-
if (numberingSettings.enabled) {
|
|
448
|
-
virtualizeNumberingLabels(originalAtoms, originalNumberingXml, numberingSettings);
|
|
449
|
-
virtualizeNumberingLabels(revisedAtoms, revisedNumberingXml, numberingSettings);
|
|
450
|
-
}
|
|
451
|
-
// Step 6: Run hierarchical LCS (paragraph-level first, then atom-level within)
|
|
452
|
-
const lcsResult = hierarchicalCompare(originalAtoms, revisedAtoms);
|
|
453
|
-
// Step 7: Mark correlation status using hierarchical result
|
|
454
|
-
markHierarchicalCorrelationStatus(originalAtoms, revisedAtoms, lcsResult);
|
|
455
|
-
// Step 8: Run move detection
|
|
456
|
-
if (moveSettings.detectMoves) {
|
|
457
|
-
// Create a combined list for move detection
|
|
458
|
-
// Move detection looks at the revised atoms with Inserted status
|
|
459
|
-
// and original atoms with Deleted status
|
|
460
|
-
const allAtoms = [...originalAtoms, ...revisedAtoms];
|
|
461
|
-
detectMovesInAtomList(allAtoms, moveSettings);
|
|
462
|
-
}
|
|
463
|
-
// Step 9: Run format detection
|
|
464
|
-
if (formatSettings.detectFormatChanges) {
|
|
465
|
-
// Format detection operates on the revised atoms that are Equal
|
|
466
|
-
detectFormatChangesInAtomList(revisedAtoms, formatSettings);
|
|
467
|
-
}
|
|
468
|
-
// Step 10: Create merged atom list for reconstruction
|
|
469
|
-
const mergedAtoms = createMergedAtomList(originalAtoms, revisedAtoms, lcsResult);
|
|
470
|
-
// Step 10b: Assign unified paragraph indices to handle atoms from different trees
|
|
471
|
-
assignUnifiedParagraphIndices(originalAtoms, revisedAtoms, mergedAtoms, lcsResult);
|
|
472
|
-
// Step 11: Reconstruct document with track changes
|
|
473
|
-
let newDocumentXml;
|
|
474
|
-
if (outputMode === 'inplace') {
|
|
475
|
-
// In-place mode: modify the revised AST directly, producing revised-based output.
|
|
476
|
-
newDocumentXml = modifyRevisedDocument(revisedTree, originalAtoms, revisedAtoms, mergedAtoms, { author, date });
|
|
477
|
-
}
|
|
478
|
-
else {
|
|
479
|
-
// Rebuild mode: reconstruct from atoms using original as the structural base.
|
|
480
|
-
newDocumentXml = reconstructDocument(mergedAtoms, originalXml, { author, date });
|
|
481
|
-
}
|
|
482
|
-
return { mergedAtoms, newDocumentXml, outputMode };
|
|
483
|
-
};
|
|
484
|
-
const evaluateRoundTripSafety = (candidateXml) => evaluateSafetyChecks(originalTextForRoundTrip, revisedTextForRoundTrip, originalBookmarkDiagnostics, revisedBookmarkDiagnostics, candidateXml, auxiliarySidecars);
|
|
485
|
-
let comparisonResult;
|
|
486
|
-
let fallbackReason;
|
|
487
|
-
let fallbackDiagnostics;
|
|
488
|
-
if (reconstructionMode === 'inplace') {
|
|
489
|
-
// Adaptive strategy:
|
|
490
|
-
// 1) Try no-cross-run passes first (higher run anchoring fidelity).
|
|
491
|
-
// 2) If safety fails, retry with cross-run merging to handle run-fragmented docs.
|
|
492
|
-
// 3) If still unsafe, reuse rebuild reconstruction as a hard safety fallback.
|
|
493
|
-
const inplacePasses = [
|
|
494
|
-
{
|
|
495
|
-
pass: 'inplace_word_split',
|
|
496
|
-
atomizeOptions: {
|
|
497
|
-
cloneLeafNodes: true,
|
|
498
|
-
mergeAcrossRuns: false,
|
|
499
|
-
mergePunctuationAcrossRuns: false,
|
|
500
|
-
splitTextIntoWords: true,
|
|
501
|
-
},
|
|
502
|
-
},
|
|
503
|
-
{
|
|
504
|
-
pass: 'inplace_run_level',
|
|
505
|
-
atomizeOptions: {
|
|
506
|
-
cloneLeafNodes: true,
|
|
507
|
-
mergeAcrossRuns: false,
|
|
508
|
-
mergePunctuationAcrossRuns: false,
|
|
509
|
-
splitTextIntoWords: false,
|
|
510
|
-
},
|
|
511
|
-
},
|
|
512
|
-
{
|
|
513
|
-
pass: 'inplace_word_split_cross_run',
|
|
514
|
-
atomizeOptions: {
|
|
515
|
-
cloneLeafNodes: true,
|
|
516
|
-
mergeAcrossRuns: true,
|
|
517
|
-
mergePunctuationAcrossRuns: true,
|
|
518
|
-
splitTextIntoWords: true,
|
|
519
|
-
},
|
|
520
|
-
},
|
|
521
|
-
{
|
|
522
|
-
pass: 'inplace_run_level_cross_run',
|
|
523
|
-
atomizeOptions: {
|
|
524
|
-
cloneLeafNodes: true,
|
|
525
|
-
mergeAcrossRuns: true,
|
|
526
|
-
mergePunctuationAcrossRuns: true,
|
|
527
|
-
splitTextIntoWords: false,
|
|
528
|
-
},
|
|
529
|
-
},
|
|
530
|
-
];
|
|
531
|
-
const failedAttempts = [];
|
|
532
|
-
let selected;
|
|
533
|
-
for (const { pass, atomizeOptions } of inplacePasses) {
|
|
534
|
-
let candidate;
|
|
535
|
-
try {
|
|
536
|
-
candidate = runComparisonPass(atomizeOptions, 'inplace');
|
|
537
|
-
}
|
|
538
|
-
catch (e) {
|
|
539
|
-
if (e instanceof ContainerResolutionError) {
|
|
540
|
-
// Container topology mismatch — treat as failed pass (issue #65)
|
|
541
|
-
failedAttempts.push({
|
|
542
|
-
pass,
|
|
543
|
-
checks: { acceptText: false, rejectText: false, acceptBookmarks: true, rejectBookmarks: true, fieldStructure: false },
|
|
544
|
-
failedChecks: ['rejectText'],
|
|
545
|
-
failureDetails: undefined,
|
|
546
|
-
firstDiffSummary: undefined,
|
|
547
|
-
});
|
|
548
|
-
continue;
|
|
549
|
-
}
|
|
550
|
-
throw e;
|
|
551
|
-
}
|
|
552
|
-
const safety = evaluateRoundTripSafety(candidate.newDocumentXml);
|
|
553
|
-
if (safety.safe) {
|
|
554
|
-
selected = candidate;
|
|
555
|
-
break;
|
|
556
|
-
}
|
|
557
|
-
failedAttempts.push({
|
|
558
|
-
pass,
|
|
559
|
-
checks: safety.checks,
|
|
560
|
-
failedChecks: safety.failedChecks,
|
|
561
|
-
failureDetails: safety.failureDetails,
|
|
562
|
-
firstDiffSummary: safety.failureSummary,
|
|
563
|
-
});
|
|
564
|
-
}
|
|
565
|
-
if (selected) {
|
|
566
|
-
comparisonResult = selected;
|
|
567
|
-
}
|
|
568
|
-
else {
|
|
569
|
-
comparisonResult = runComparisonPass({ atomizeParagraphLevelMarkers: true }, 'rebuild');
|
|
570
|
-
fallbackReason = 'round_trip_safety_check_failed';
|
|
571
|
-
fallbackDiagnostics = {
|
|
572
|
-
attempts: failedAttempts,
|
|
573
|
-
};
|
|
574
|
-
}
|
|
575
|
-
}
|
|
576
|
-
else {
|
|
577
|
-
comparisonResult = runComparisonPass({ atomizeParagraphLevelMarkers: true }, 'rebuild');
|
|
578
|
-
}
|
|
579
|
-
// Rebuild output gets the same safety screening as inplace attempts, whether
|
|
580
|
-
// rebuild was requested directly or reached via inplace fallback. Rebuild is
|
|
581
|
-
// the terminal strategy, so failures are surfaced in diagnostics rather than
|
|
582
|
-
// blocking the output.
|
|
583
|
-
// @see https://github.com/UseJunior/safe-docx/issues/226
|
|
584
|
-
let rebuildSafetyDiagnostics;
|
|
585
|
-
if (comparisonResult.outputMode === 'rebuild') {
|
|
586
|
-
const safety = evaluateRoundTripSafety(comparisonResult.newDocumentXml);
|
|
587
|
-
if (!safety.safe) {
|
|
588
|
-
rebuildSafetyDiagnostics = {
|
|
589
|
-
checks: safety.checks,
|
|
590
|
-
failedChecks: safety.failedChecks,
|
|
591
|
-
failureDetails: safety.failureDetails,
|
|
592
|
-
firstDiffSummary: safety.failureSummary,
|
|
593
|
-
};
|
|
594
|
-
}
|
|
595
|
-
}
|
|
596
|
-
const { mergedAtoms, newDocumentXml } = comparisonResult;
|
|
597
|
-
// Step 12: Clone appropriate archive and update document.xml.
|
|
598
|
-
// Use the revised archive only for true inplace output.
|
|
599
|
-
const baseArchive = comparisonResult.outputMode === 'inplace' ? revisedArchive : originalArchive;
|
|
600
|
-
// The merge source is the *opposite* archive from the base: inplace pulls
|
|
601
|
-
// deleted-but-still-referenced definitions from the original, rebuild pulls
|
|
602
|
-
// added-but-still-referenced definitions from the revised. Without this,
|
|
603
|
-
// rebuild output ships dangling references when the original lacks an
|
|
604
|
-
// auxiliary part that the revised side introduced (issue #94).
|
|
605
|
-
const mergeSourceArchive = comparisonResult.outputMode === 'inplace' ? originalArchive : revisedArchive;
|
|
606
|
-
const resultArchive = await baseArchive.clone();
|
|
607
|
-
maybeCaptureEmittedDocumentXml(newDocumentXml);
|
|
608
|
-
resultArchive.setDocumentXml(newDocumentXml);
|
|
609
|
-
// Step 12b: Merge auxiliary part definitions (footnotes, endnotes, comments).
|
|
610
|
-
// Reconstruction may insert content (deleted in inplace, added in rebuild)
|
|
611
|
-
// whose definitions are missing from the base archive.
|
|
612
|
-
for (const descriptor of AUXILIARY_PARTS) {
|
|
613
|
-
await mergeAuxiliaryPartDefinitions(mergeSourceArchive, resultArchive, newDocumentXml, descriptor);
|
|
614
|
-
}
|
|
615
|
-
// Comment-specific post-pass: walk reply threads via commentsExtended.xml.
|
|
616
|
-
// Gated on root comment IDs in the *result* document (not on what the
|
|
617
|
-
// generic merge appended), so the pass runs even when the original already
|
|
618
|
-
// contains the root and revised only adds replies under it (issue #108).
|
|
619
|
-
// Comments anchored on footnote/endnote text count as roots too.
|
|
620
|
-
const rootCommentIds = await collectStoryReferenceIds(resultArchive, newDocumentXml, 'w:commentReference', null);
|
|
621
|
-
if (rootCommentIds.size > 0) {
|
|
622
|
-
await mergeCommentAncillaryParts(mergeSourceArchive, resultArchive, rootCommentIds);
|
|
623
|
-
}
|
|
624
|
-
// Step 13: Save result and compute stats
|
|
625
|
-
const resultBuffer = await resultArchive.save();
|
|
626
|
-
const stats = computeAtomizerStats(mergedAtoms);
|
|
627
|
-
return {
|
|
628
|
-
document: resultBuffer,
|
|
629
|
-
stats,
|
|
630
|
-
engine: 'atomizer',
|
|
631
|
-
reconstructionModeRequested: reconstructionMode,
|
|
632
|
-
reconstructionModeUsed: comparisonResult.outputMode,
|
|
633
|
-
fallbackReason,
|
|
634
|
-
fallbackDiagnostics,
|
|
635
|
-
rebuildSafetyDiagnostics,
|
|
636
|
-
};
|
|
637
|
-
}
|
|
638
|
-
/**
|
|
639
|
-
* Collect reference IDs across every result story that can host anchors: the
|
|
640
|
-
* merged document.xml plus the result archive's footnote/endnote parts (Word
|
|
641
|
-
* allows comments anchored on note text). `excludePartPath` skips the part
|
|
642
|
-
* whose own definitions are being merged — entries can't reference
|
|
643
|
-
* themselves.
|
|
644
|
-
*/
|
|
645
|
-
async function collectStoryReferenceIds(resultArchive, documentXml, referenceTag, excludePartPath) {
|
|
646
|
-
const ids = collectReferenceIds(documentXml, referenceTag);
|
|
647
|
-
for (const storyPath of ['word/footnotes.xml', 'word/endnotes.xml']) {
|
|
648
|
-
if (storyPath === excludePartPath)
|
|
649
|
-
continue;
|
|
650
|
-
const storyXml = await resultArchive.getFile(storyPath);
|
|
651
|
-
if (!storyXml)
|
|
652
|
-
continue;
|
|
653
|
-
for (const id of collectReferenceIds(storyXml, referenceTag))
|
|
654
|
-
ids.add(id);
|
|
655
|
-
}
|
|
656
|
-
return ids;
|
|
657
|
-
}
|
|
658
|
-
/**
|
|
659
|
-
* Collect reference IDs from document.xml using DOM parsing.
|
|
660
|
-
*/
|
|
661
|
-
function collectReferenceIds(documentXml, referenceTag) {
|
|
662
|
-
const ids = new Set();
|
|
663
|
-
const doc = parseXml(documentXml);
|
|
664
|
-
const refs = doc.getElementsByTagName(referenceTag);
|
|
665
|
-
for (let i = 0; i < refs.length; i++) {
|
|
666
|
-
const id = refs[i].getAttribute('w:id');
|
|
667
|
-
if (id)
|
|
668
|
-
ids.add(id);
|
|
669
|
-
}
|
|
670
|
-
return ids;
|
|
671
|
-
}
|
|
672
|
-
/**
|
|
673
|
-
* Merge auxiliary part definitions (footnotes, endnotes, comments) from the
|
|
674
|
-
* source archive into the result archive. The source archive is whichever
|
|
675
|
-
* side reconstruction may have introduced references to: original in inplace
|
|
676
|
-
* mode (deleted-but-referenced definitions), revised in rebuild mode
|
|
677
|
-
* (added-but-referenced definitions).
|
|
678
|
-
*/
|
|
679
|
-
async function mergeAuxiliaryPartDefinitions(sourceArchive, resultArchive, documentXml, descriptor) {
|
|
680
|
-
const result = { mergedIds: new Set(), createdPart: false };
|
|
681
|
-
// Anchors may live in the merged body or on note text in the result's
|
|
682
|
-
// footnote/endnote stories. AUXILIARY_PARTS merges notes before comments,
|
|
683
|
-
// so by the comment pass the note stories already carry any merged-in
|
|
684
|
-
// comment anchors.
|
|
685
|
-
const referencedIds = await collectStoryReferenceIds(resultArchive, documentXml, descriptor.referenceTag, descriptor.partPath);
|
|
686
|
-
if (referencedIds.size === 0)
|
|
687
|
-
return result;
|
|
688
|
-
const sourcePartXml = await sourceArchive.getFile(descriptor.partPath);
|
|
689
|
-
if (!sourcePartXml)
|
|
690
|
-
return result;
|
|
691
|
-
const resultPartXml = await resultArchive.getFile(descriptor.partPath);
|
|
692
|
-
const sourceParsed = parseEntries(sourcePartXml, descriptor.entryTag);
|
|
693
|
-
const resultParsed = resultPartXml ? parseEntries(resultPartXml, descriptor.entryTag) : null;
|
|
694
|
-
// Find missing entries: referenced in document.xml but not in result
|
|
695
|
-
const missingElements = [];
|
|
696
|
-
for (const id of referencedIds) {
|
|
697
|
-
if (!(resultParsed?.entries.has(id)) && sourceParsed.entries.has(id)) {
|
|
698
|
-
missingElements.push(sourceParsed.entries.get(id));
|
|
699
|
-
result.mergedIds.add(id);
|
|
700
|
-
}
|
|
701
|
-
}
|
|
702
|
-
if (missingElements.length === 0)
|
|
703
|
-
return result;
|
|
704
|
-
if (resultPartXml && resultParsed) {
|
|
705
|
-
// Insert missing entries into existing result part
|
|
706
|
-
const rootEl = resultParsed.doc.getElementsByTagName(descriptor.rootTag)[0];
|
|
707
|
-
if (rootEl) {
|
|
708
|
-
for (const el of missingElements) {
|
|
709
|
-
const imported = resultParsed.doc.importNode(el, true);
|
|
710
|
-
rootEl.appendChild(imported);
|
|
711
|
-
}
|
|
712
|
-
resultArchive.setFile(descriptor.partPath, serializer.serializeToString(resultParsed.doc));
|
|
713
|
-
}
|
|
714
|
-
}
|
|
715
|
-
else {
|
|
716
|
-
// Create part from scratch: clone root from merge source, drop every
|
|
717
|
-
// non-reserved entry, then append the missing referenced ones.
|
|
718
|
-
// Reserved entries are footnote/endnote separators identified by
|
|
719
|
-
// w:type="separator" / w:type="continuationSeparator" — Word expects
|
|
720
|
-
// them to exist and they don't carry user content. Filtering by w:type
|
|
721
|
-
// (not by magic w:id values) keeps this robust across authoring tools.
|
|
722
|
-
const newDoc = parseXml(sourcePartXml);
|
|
723
|
-
const rootEl = newDoc.getElementsByTagName(descriptor.rootTag)[0];
|
|
724
|
-
if (rootEl) {
|
|
725
|
-
const existingEntries = rootEl.getElementsByTagName(descriptor.entryTag);
|
|
726
|
-
const toRemove = [];
|
|
727
|
-
for (let i = 0; i < existingEntries.length; i++) {
|
|
728
|
-
const el = existingEntries[i];
|
|
729
|
-
const type = el.getAttribute('w:type');
|
|
730
|
-
if (type !== 'separator' && type !== 'continuationSeparator') {
|
|
731
|
-
toRemove.push(el);
|
|
732
|
-
}
|
|
733
|
-
}
|
|
734
|
-
for (const el of toRemove) {
|
|
735
|
-
rootEl.removeChild(el);
|
|
736
|
-
}
|
|
737
|
-
for (const el of missingElements) {
|
|
738
|
-
const imported = newDoc.importNode(el, true);
|
|
739
|
-
rootEl.appendChild(imported);
|
|
740
|
-
}
|
|
741
|
-
resultArchive.setFile(descriptor.partPath, serializer.serializeToString(newDoc));
|
|
742
|
-
result.createdPart = true;
|
|
743
|
-
await ensureOpcMetadata(resultArchive, descriptor);
|
|
744
|
-
}
|
|
745
|
-
}
|
|
746
|
-
return result;
|
|
747
|
-
}
|
|
748
|
-
// =============================================================================
|
|
749
|
-
// OPC Metadata Bootstrapping
|
|
750
|
-
// =============================================================================
|
|
751
|
-
const CT_NS = 'http://schemas.openxmlformats.org/package/2006/content-types';
|
|
752
|
-
const REL_NS = 'http://schemas.openxmlformats.org/package/2006/relationships';
|
|
753
|
-
/**
|
|
754
|
-
* Ensure [Content_Types].xml and document.xml.rels have entries for a
|
|
755
|
-
* newly-created auxiliary part.
|
|
756
|
-
*/
|
|
757
|
-
async function ensureOpcMetadata(archive, descriptor) {
|
|
758
|
-
// 1. Update [Content_Types].xml
|
|
759
|
-
const ctXml = await archive.getFile('[Content_Types].xml');
|
|
760
|
-
if (ctXml) {
|
|
761
|
-
const ctDoc = parseXml(ctXml);
|
|
762
|
-
const typesEl = ctDoc.documentElement;
|
|
763
|
-
const overrides = typesEl.getElementsByTagNameNS(CT_NS, 'Override');
|
|
764
|
-
const partName = `/${descriptor.partPath}`;
|
|
765
|
-
let found = false;
|
|
766
|
-
for (let i = 0; i < overrides.length; i++) {
|
|
767
|
-
if (overrides[i].getAttribute('PartName') === partName) {
|
|
768
|
-
found = true;
|
|
769
|
-
break;
|
|
770
|
-
}
|
|
771
|
-
}
|
|
772
|
-
if (!found) {
|
|
773
|
-
const override = ctDoc.createElementNS(CT_NS, 'Override');
|
|
774
|
-
override.setAttribute('PartName', partName);
|
|
775
|
-
override.setAttribute('ContentType', descriptor.contentType);
|
|
776
|
-
typesEl.appendChild(override);
|
|
777
|
-
archive.setFile('[Content_Types].xml', serializer.serializeToString(ctDoc));
|
|
778
|
-
}
|
|
779
|
-
}
|
|
780
|
-
// 2. Update word/_rels/document.xml.rels
|
|
781
|
-
const relsPath = 'word/_rels/document.xml.rels';
|
|
782
|
-
const relsXml = await archive.getFile(relsPath);
|
|
783
|
-
if (relsXml) {
|
|
784
|
-
const relsDoc = parseXml(relsXml);
|
|
785
|
-
const relsEl = relsDoc.documentElement;
|
|
786
|
-
const existingRels = relsEl.getElementsByTagNameNS(REL_NS, 'Relationship');
|
|
787
|
-
let found = false;
|
|
788
|
-
let maxId = 0;
|
|
789
|
-
for (let i = 0; i < existingRels.length; i++) {
|
|
790
|
-
const rel = existingRels[i];
|
|
791
|
-
if (rel.getAttribute('Type') === descriptor.relationshipType) {
|
|
792
|
-
found = true;
|
|
793
|
-
}
|
|
794
|
-
const id = rel.getAttribute('Id') ?? '';
|
|
795
|
-
const idMatch = /^rId(\d+)$/.exec(id);
|
|
796
|
-
if (idMatch)
|
|
797
|
-
maxId = Math.max(maxId, parseInt(idMatch[1], 10));
|
|
798
|
-
}
|
|
799
|
-
if (!found) {
|
|
800
|
-
maxId++;
|
|
801
|
-
const rel = relsDoc.createElementNS(REL_NS, 'Relationship');
|
|
802
|
-
rel.setAttribute('Id', `rId${maxId}`);
|
|
803
|
-
rel.setAttribute('Type', descriptor.relationshipType);
|
|
804
|
-
rel.setAttribute('Target', descriptor.partPath.replace('word/', ''));
|
|
805
|
-
relsEl.appendChild(rel);
|
|
806
|
-
archive.setFile(relsPath, serializer.serializeToString(relsDoc));
|
|
807
|
-
}
|
|
808
|
-
}
|
|
809
|
-
}
|
|
810
|
-
// =============================================================================
|
|
811
|
-
// Comment Ancillary Parts Merging
|
|
812
|
-
// =============================================================================
|
|
813
|
-
/**
|
|
814
|
-
* Walk the comment reply graph from each root referenced in the result
|
|
815
|
-
* document, merging reply <w:comment> entries, their commentsExtended.xml
|
|
816
|
-
* threading entries, and people.xml authors. Replies have no
|
|
817
|
-
* <w:commentReference> in document.xml — they're discoverable only via
|
|
818
|
-
* w15:paraIdParent in commentsExtended.xml. Without this expansion, rebuild
|
|
819
|
-
* mode silently drops reply threads (issue #108).
|
|
820
|
-
*/
|
|
821
|
-
async function mergeCommentAncillaryParts(sourceArchive, resultArchive, rootCommentIds) {
|
|
822
|
-
const sourceCommentsXml = await sourceArchive.getFile('word/comments.xml');
|
|
823
|
-
if (!sourceCommentsXml)
|
|
824
|
-
return;
|
|
825
|
-
const sourceDoc = parseXml(sourceCommentsXml);
|
|
826
|
-
// Build full source comment maps. Canonical paraId is the first <w:p>
|
|
827
|
-
// child's w14:paraId, matching getCommentElParaId() in primitives/comments.ts.
|
|
828
|
-
const commentById = new Map();
|
|
829
|
-
const paraIdByCommentId = new Map();
|
|
830
|
-
const commentIdByParaId = new Map();
|
|
831
|
-
const authorByCommentId = new Map();
|
|
832
|
-
const allCommentEls = sourceDoc.getElementsByTagName('w:comment');
|
|
833
|
-
for (let i = 0; i < allCommentEls.length; i++) {
|
|
834
|
-
const el = allCommentEls[i];
|
|
835
|
-
const id = el.getAttribute('w:id');
|
|
836
|
-
if (!id)
|
|
837
|
-
continue;
|
|
838
|
-
commentById.set(id, el);
|
|
839
|
-
const author = el.getAttribute('w:author');
|
|
840
|
-
if (author)
|
|
841
|
-
authorByCommentId.set(id, author);
|
|
842
|
-
const firstP = el.getElementsByTagName('w:p')[0];
|
|
843
|
-
const paraId = firstP?.getAttribute('w14:paraId');
|
|
844
|
-
if (paraId) {
|
|
845
|
-
paraIdByCommentId.set(id, paraId);
|
|
846
|
-
commentIdByParaId.set(paraId, id);
|
|
847
|
-
}
|
|
848
|
-
}
|
|
849
|
-
// Seed inclusion sets from the root IDs that appear in the result document.
|
|
850
|
-
const includedCommentIds = new Set();
|
|
851
|
-
const includedParaIds = new Set();
|
|
852
|
-
const includedAuthors = new Set();
|
|
853
|
-
for (const id of rootCommentIds) {
|
|
854
|
-
if (!commentById.has(id))
|
|
855
|
-
continue;
|
|
856
|
-
includedCommentIds.add(id);
|
|
857
|
-
const pid = paraIdByCommentId.get(id);
|
|
858
|
-
if (pid)
|
|
859
|
-
includedParaIds.add(pid);
|
|
860
|
-
const author = authorByCommentId.get(id);
|
|
861
|
-
if (author)
|
|
862
|
-
includedAuthors.add(author);
|
|
863
|
-
}
|
|
864
|
-
// BFS over commentsExtended.xml's paraIdParent graph from each included
|
|
865
|
-
// root paraId. Skip entries that don't resolve to a real source comment so
|
|
866
|
-
// we never pull in dangling commentEx/people without a backing definition.
|
|
867
|
-
const sourceExtendedXml = await sourceArchive.getFile('word/commentsExtended.xml');
|
|
868
|
-
if (sourceExtendedXml) {
|
|
869
|
-
const exDoc = parseXml(sourceExtendedXml);
|
|
870
|
-
const exEls = exDoc.getElementsByTagName('w15:commentEx');
|
|
871
|
-
const childrenOf = new Map();
|
|
872
|
-
for (let i = 0; i < exEls.length; i++) {
|
|
873
|
-
const ex = exEls[i];
|
|
874
|
-
const childPid = ex.getAttribute('w15:paraId');
|
|
875
|
-
const parentPid = ex.getAttribute('w15:paraIdParent');
|
|
876
|
-
if (!childPid || !parentPid)
|
|
877
|
-
continue;
|
|
878
|
-
const arr = childrenOf.get(parentPid);
|
|
879
|
-
if (arr)
|
|
880
|
-
arr.push(childPid);
|
|
881
|
-
else
|
|
882
|
-
childrenOf.set(parentPid, [childPid]);
|
|
883
|
-
}
|
|
884
|
-
const queue = [...includedParaIds];
|
|
885
|
-
while (queue.length > 0) {
|
|
886
|
-
const pid = queue.shift();
|
|
887
|
-
const children = childrenOf.get(pid);
|
|
888
|
-
if (!children)
|
|
889
|
-
continue;
|
|
890
|
-
for (const childPid of children) {
|
|
891
|
-
if (includedParaIds.has(childPid))
|
|
892
|
-
continue;
|
|
893
|
-
const childCommentId = commentIdByParaId.get(childPid);
|
|
894
|
-
if (!childCommentId)
|
|
895
|
-
continue;
|
|
896
|
-
includedParaIds.add(childPid);
|
|
897
|
-
includedCommentIds.add(childCommentId);
|
|
898
|
-
const author = authorByCommentId.get(childCommentId);
|
|
899
|
-
if (author)
|
|
900
|
-
includedAuthors.add(author);
|
|
901
|
-
queue.push(childPid);
|
|
902
|
-
}
|
|
903
|
-
}
|
|
904
|
-
}
|
|
905
|
-
// Append any reply <w:comment> definitions still missing from result.
|
|
906
|
-
// The generic merge already added roots when needed; we add the replies
|
|
907
|
-
// (and any roots not yet present in the result, defensively).
|
|
908
|
-
await mergeMissingCommentDefinitions(resultArchive, commentById, includedCommentIds);
|
|
909
|
-
// Merge commentsExtended and people for the expanded set.
|
|
910
|
-
await mergeCommentsExtended(sourceArchive, resultArchive, includedParaIds);
|
|
911
|
-
await mergePeople(sourceArchive, resultArchive, includedAuthors);
|
|
912
|
-
}
|
|
913
|
-
/**
|
|
914
|
-
* Append any source <w:comment> definitions in `includedCommentIds` that
|
|
915
|
-
* aren't already in result/word/comments.xml. Mirrors the append-with-importNode
|
|
916
|
-
* pattern used by mergeCommentsExtended below.
|
|
917
|
-
*/
|
|
918
|
-
async function mergeMissingCommentDefinitions(resultArchive, commentById, includedCommentIds) {
|
|
919
|
-
if (includedCommentIds.size === 0)
|
|
920
|
-
return;
|
|
921
|
-
const resultXml = await resultArchive.getFile('word/comments.xml');
|
|
922
|
-
if (!resultXml) {
|
|
923
|
-
// If result has no comments.xml at all, the generic merge would have
|
|
924
|
-
// bootstrapped it for any included root. Nothing to do here.
|
|
925
|
-
return;
|
|
926
|
-
}
|
|
927
|
-
const resultDoc = parseXml(resultXml);
|
|
928
|
-
const rootEl = resultDoc.documentElement;
|
|
929
|
-
const existingIds = new Set();
|
|
930
|
-
const existing = rootEl.getElementsByTagName('w:comment');
|
|
931
|
-
for (let i = 0; i < existing.length; i++) {
|
|
932
|
-
const id = existing[i].getAttribute('w:id');
|
|
933
|
-
if (id)
|
|
934
|
-
existingIds.add(id);
|
|
935
|
-
}
|
|
936
|
-
let appended = false;
|
|
937
|
-
for (const id of includedCommentIds) {
|
|
938
|
-
if (existingIds.has(id))
|
|
939
|
-
continue;
|
|
940
|
-
const sourceEl = commentById.get(id);
|
|
941
|
-
if (!sourceEl)
|
|
942
|
-
continue;
|
|
943
|
-
rootEl.appendChild(resultDoc.importNode(sourceEl, true));
|
|
944
|
-
appended = true;
|
|
945
|
-
}
|
|
946
|
-
if (appended) {
|
|
947
|
-
resultArchive.setFile('word/comments.xml', serializer.serializeToString(resultDoc));
|
|
948
|
-
}
|
|
949
|
-
}
|
|
950
|
-
async function mergeCommentsExtended(sourceArchive, resultArchive, mergedParaIds) {
|
|
951
|
-
if (mergedParaIds.size === 0)
|
|
952
|
-
return;
|
|
953
|
-
const sourceXml = await sourceArchive.getFile('word/commentsExtended.xml');
|
|
954
|
-
if (!sourceXml)
|
|
955
|
-
return;
|
|
956
|
-
const sourceDoc = parseXml(sourceXml);
|
|
957
|
-
const sourceEntries = sourceDoc.getElementsByTagName('w15:commentEx');
|
|
958
|
-
// Collect entries whose paraId matches a merged comment's paragraph
|
|
959
|
-
const entriesToMerge = [];
|
|
960
|
-
for (let i = 0; i < sourceEntries.length; i++) {
|
|
961
|
-
const el = sourceEntries[i];
|
|
962
|
-
const paraId = el.getAttribute('w15:paraId');
|
|
963
|
-
if (paraId && mergedParaIds.has(paraId)) {
|
|
964
|
-
entriesToMerge.push(el);
|
|
965
|
-
}
|
|
966
|
-
}
|
|
967
|
-
if (entriesToMerge.length === 0)
|
|
968
|
-
return;
|
|
969
|
-
const resultXml = await resultArchive.getFile('word/commentsExtended.xml');
|
|
970
|
-
if (resultXml) {
|
|
971
|
-
const resultDoc = parseXml(resultXml);
|
|
972
|
-
const rootEl = resultDoc.documentElement;
|
|
973
|
-
const existingParaIds = new Set();
|
|
974
|
-
const existing = rootEl.getElementsByTagName('w15:commentEx');
|
|
975
|
-
for (let i = 0; i < existing.length; i++) {
|
|
976
|
-
const pid = existing[i].getAttribute('w15:paraId');
|
|
977
|
-
if (pid)
|
|
978
|
-
existingParaIds.add(pid);
|
|
979
|
-
}
|
|
980
|
-
for (const el of entriesToMerge) {
|
|
981
|
-
const pid = el.getAttribute('w15:paraId');
|
|
982
|
-
if (pid && !existingParaIds.has(pid)) {
|
|
983
|
-
rootEl.appendChild(resultDoc.importNode(el, true));
|
|
984
|
-
}
|
|
985
|
-
}
|
|
986
|
-
resultArchive.setFile('word/commentsExtended.xml', serializer.serializeToString(resultDoc));
|
|
987
|
-
return;
|
|
988
|
-
}
|
|
989
|
-
// Bootstrap: result lacks commentsExtended.xml but the merged comments
|
|
990
|
-
// depend on it for reply threading / done state. Clone the source's root
|
|
991
|
-
// (preserves namespaces), drop non-matching entries, then add OPC metadata.
|
|
992
|
-
const newDoc = parseXml(sourceXml);
|
|
993
|
-
const newRoot = newDoc.documentElement;
|
|
994
|
-
const allEntries = newRoot.getElementsByTagName('w15:commentEx');
|
|
995
|
-
const toRemove = [];
|
|
996
|
-
for (let i = 0; i < allEntries.length; i++) {
|
|
997
|
-
const el = allEntries[i];
|
|
998
|
-
const paraId = el.getAttribute('w15:paraId');
|
|
999
|
-
if (!paraId || !mergedParaIds.has(paraId))
|
|
1000
|
-
toRemove.push(el);
|
|
1001
|
-
}
|
|
1002
|
-
for (const el of toRemove)
|
|
1003
|
-
newRoot.removeChild(el);
|
|
1004
|
-
resultArchive.setFile('word/commentsExtended.xml', serializer.serializeToString(newDoc));
|
|
1005
|
-
await ensureOpcMetadata(resultArchive, COMMENTS_EXTENDED_DESCRIPTOR);
|
|
1006
|
-
}
|
|
1007
|
-
const COMMENTS_EXTENDED_DESCRIPTOR = {
|
|
1008
|
-
label: 'commentsExtended',
|
|
1009
|
-
partPath: 'word/commentsExtended.xml',
|
|
1010
|
-
referenceTag: '',
|
|
1011
|
-
entryTag: 'w15:commentEx',
|
|
1012
|
-
rootTag: 'w15:commentsEx',
|
|
1013
|
-
contentType: 'application/vnd.ms-word.commentsExtended+xml',
|
|
1014
|
-
relationshipType: 'http://schemas.microsoft.com/office/2011/relationships/commentsExtended',
|
|
1015
|
-
idBearingTags: [], // keyed by w15:paraId, not w:id
|
|
1016
|
-
};
|
|
1017
|
-
const PEOPLE_DESCRIPTOR = {
|
|
1018
|
-
label: 'people',
|
|
1019
|
-
partPath: 'word/people.xml',
|
|
1020
|
-
referenceTag: '',
|
|
1021
|
-
entryTag: 'w15:person',
|
|
1022
|
-
rootTag: 'w15:people',
|
|
1023
|
-
contentType: 'application/vnd.ms-word.people+xml',
|
|
1024
|
-
relationshipType: 'http://schemas.microsoft.com/office/2011/relationships/people',
|
|
1025
|
-
idBearingTags: [], // keyed by w15:author, not w:id
|
|
1026
|
-
};
|
|
1027
|
-
async function mergePeople(sourceArchive, resultArchive, mergedAuthors) {
|
|
1028
|
-
if (mergedAuthors.size === 0)
|
|
1029
|
-
return;
|
|
1030
|
-
const sourceXml = await sourceArchive.getFile('word/people.xml');
|
|
1031
|
-
if (!sourceXml)
|
|
1032
|
-
return;
|
|
1033
|
-
const sourceDoc = parseXml(sourceXml);
|
|
1034
|
-
const sourcePersons = sourceDoc.getElementsByTagName('w15:person');
|
|
1035
|
-
const personsToMerge = [];
|
|
1036
|
-
for (let i = 0; i < sourcePersons.length; i++) {
|
|
1037
|
-
const el = sourcePersons[i];
|
|
1038
|
-
const author = el.getAttribute('w15:author');
|
|
1039
|
-
if (author && mergedAuthors.has(author)) {
|
|
1040
|
-
personsToMerge.push(el);
|
|
1041
|
-
}
|
|
1042
|
-
}
|
|
1043
|
-
if (personsToMerge.length === 0)
|
|
1044
|
-
return;
|
|
1045
|
-
const resultXml = await resultArchive.getFile('word/people.xml');
|
|
1046
|
-
if (resultXml) {
|
|
1047
|
-
const resultDoc = parseXml(resultXml);
|
|
1048
|
-
const rootEl = resultDoc.documentElement;
|
|
1049
|
-
const existingAuthors = new Set();
|
|
1050
|
-
const existing = rootEl.getElementsByTagName('w15:person');
|
|
1051
|
-
for (let i = 0; i < existing.length; i++) {
|
|
1052
|
-
const a = existing[i].getAttribute('w15:author');
|
|
1053
|
-
if (a)
|
|
1054
|
-
existingAuthors.add(a);
|
|
1055
|
-
}
|
|
1056
|
-
for (const el of personsToMerge) {
|
|
1057
|
-
const a = el.getAttribute('w15:author');
|
|
1058
|
-
if (a && !existingAuthors.has(a)) {
|
|
1059
|
-
rootEl.appendChild(resultDoc.importNode(el, true));
|
|
1060
|
-
}
|
|
1061
|
-
}
|
|
1062
|
-
resultArchive.setFile('word/people.xml', serializer.serializeToString(resultDoc));
|
|
1063
|
-
return;
|
|
1064
|
-
}
|
|
1065
|
-
// Bootstrap: result lacks people.xml. Clone source root (preserves
|
|
1066
|
-
// namespaces), remove non-matching authors, then add OPC metadata.
|
|
1067
|
-
const newDoc = parseXml(sourceXml);
|
|
1068
|
-
const newRoot = newDoc.documentElement;
|
|
1069
|
-
const allPersons = newRoot.getElementsByTagName('w15:person');
|
|
1070
|
-
const toRemove = [];
|
|
1071
|
-
for (let i = 0; i < allPersons.length; i++) {
|
|
1072
|
-
const el = allPersons[i];
|
|
1073
|
-
const author = el.getAttribute('w15:author');
|
|
1074
|
-
if (!author || !mergedAuthors.has(author))
|
|
1075
|
-
toRemove.push(el);
|
|
1076
|
-
}
|
|
1077
|
-
for (const el of toRemove)
|
|
1078
|
-
newRoot.removeChild(el);
|
|
1079
|
-
resultArchive.setFile('word/people.xml', serializer.serializeToString(newDoc));
|
|
1080
|
-
await ensureOpcMetadata(resultArchive, PEOPLE_DESCRIPTOR);
|
|
1081
|
-
}
|
|
1082
|
-
const fallbackParagraphStatsKeys = new WeakMap();
|
|
1083
|
-
let nextFallbackParagraphStatsKey = 0;
|
|
1084
|
-
function paragraphStatsKey(atom) {
|
|
1085
|
-
if (atom.paragraphIndex !== undefined) {
|
|
1086
|
-
return `${atom.part.uri}:${atom.paragraphIndex}`;
|
|
1087
|
-
}
|
|
1088
|
-
const pAncestor = atom.ancestorElements.find((a) => a.tagName === 'w:p');
|
|
1089
|
-
if (!pAncestor)
|
|
1090
|
-
return undefined;
|
|
1091
|
-
let key = fallbackParagraphStatsKeys.get(pAncestor);
|
|
1092
|
-
if (!key) {
|
|
1093
|
-
key = `${atom.part.uri}:paragraph-ref:${nextFallbackParagraphStatsKey++}`;
|
|
1094
|
-
fallbackParagraphStatsKeys.set(pAncestor, key);
|
|
1095
|
-
}
|
|
1096
|
-
return key;
|
|
1097
|
-
}
|
|
1098
|
-
/**
|
|
1099
|
-
* Compute comparison statistics from merged atoms.
|
|
1100
|
-
*
|
|
1101
|
-
* Range counts are contiguous same-status runs in the merged atom stream, scoped
|
|
1102
|
-
* to a paragraph. Atom counts remain available under explicit names for callers
|
|
1103
|
-
* that need the old granular benchmark signal.
|
|
1104
|
-
*/
|
|
1105
|
-
export function computeAtomizerStats(mergedAtoms) {
|
|
1106
|
-
const reconstructionStats = computeReconstructionStats(mergedAtoms);
|
|
1107
|
-
let insertedRanges = 0;
|
|
1108
|
-
let deletedRanges = 0;
|
|
1109
|
-
let formatChanges = 0;
|
|
1110
|
-
let previousRangeStatus = null;
|
|
1111
|
-
let previousRangeParagraph;
|
|
1112
|
-
const paragraphs = new Map();
|
|
1113
|
-
for (const atom of mergedAtoms) {
|
|
1114
|
-
const paragraphKey = paragraphStatsKey(atom);
|
|
1115
|
-
const status = atom.correlationStatus;
|
|
1116
|
-
const rangeStatus = status === CorrelationStatus.Inserted ||
|
|
1117
|
-
status === CorrelationStatus.Deleted ||
|
|
1118
|
-
status === CorrelationStatus.FormatChanged
|
|
1119
|
-
? status
|
|
1120
|
-
: null;
|
|
1121
|
-
if (rangeStatus) {
|
|
1122
|
-
if (rangeStatus !== previousRangeStatus || paragraphKey !== previousRangeParagraph) {
|
|
1123
|
-
if (rangeStatus === CorrelationStatus.Inserted)
|
|
1124
|
-
insertedRanges++;
|
|
1125
|
-
if (rangeStatus === CorrelationStatus.Deleted)
|
|
1126
|
-
deletedRanges++;
|
|
1127
|
-
if (rangeStatus === CorrelationStatus.FormatChanged)
|
|
1128
|
-
formatChanges++;
|
|
1129
|
-
}
|
|
1130
|
-
previousRangeStatus = rangeStatus;
|
|
1131
|
-
previousRangeParagraph = paragraphKey;
|
|
1132
|
-
}
|
|
1133
|
-
else {
|
|
1134
|
-
previousRangeStatus = null;
|
|
1135
|
-
previousRangeParagraph = undefined;
|
|
1136
|
-
}
|
|
1137
|
-
if (paragraphKey && (status === CorrelationStatus.Deleted || status === CorrelationStatus.Inserted)) {
|
|
1138
|
-
const flags = paragraphs.get(paragraphKey) ?? { hasDeleted: false, hasInserted: false };
|
|
1139
|
-
if (status === CorrelationStatus.Deleted)
|
|
1140
|
-
flags.hasDeleted = true;
|
|
1141
|
-
if (status === CorrelationStatus.Inserted)
|
|
1142
|
-
flags.hasInserted = true;
|
|
1143
|
-
paragraphs.set(paragraphKey, flags);
|
|
1144
|
-
}
|
|
1145
|
-
}
|
|
1146
|
-
const modifiedParagraphs = Array.from(paragraphs.values()).filter((flags) => flags.hasDeleted && flags.hasInserted).length;
|
|
1147
|
-
return {
|
|
1148
|
-
insertions: insertedRanges,
|
|
1149
|
-
deletions: deletedRanges,
|
|
1150
|
-
modifications: modifiedParagraphs,
|
|
1151
|
-
insertedRanges,
|
|
1152
|
-
deletedRanges,
|
|
1153
|
-
insertedAtoms: reconstructionStats.insertions,
|
|
1154
|
-
deletedAtoms: reconstructionStats.deletions,
|
|
1155
|
-
modifiedParagraphs,
|
|
1156
|
-
formatChanges,
|
|
1157
|
-
formatChangeAtoms: reconstructionStats.formatChanges,
|
|
1158
|
-
};
|
|
1159
|
-
}
|
|
1160
|
-
//# sourceMappingURL=pipeline.js.map
|