@usejunior/docx-core 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -83
- package/dist/.tsbuildinfo +1 -1
- package/dist/cli/conformance-adapter.d.ts +14 -0
- package/dist/cli/conformance-adapter.d.ts.map +1 -1
- package/dist/cli/conformance-adapter.js +104 -12
- package/dist/cli/conformance-adapter.js.map +1 -1
- package/dist/footnotes.d.ts +7 -5
- package/dist/footnotes.d.ts.map +1 -1
- package/dist/footnotes.js +7 -5
- package/dist/footnotes.js.map +1 -1
- package/dist/generated/ecma-376-vocabulary.d.ts +93 -0
- package/dist/generated/ecma-376-vocabulary.d.ts.map +1 -0
- package/dist/generated/ecma-376-vocabulary.js +87 -0
- package/dist/generated/ecma-376-vocabulary.js.map +1 -0
- package/dist/generation/compile.js +2 -2
- package/dist/generation/compile.js.map +1 -1
- package/dist/generation/emit/comments-part.d.ts +2 -0
- package/dist/generation/emit/comments-part.d.ts.map +1 -1
- package/dist/generation/emit/comments-part.js +2 -0
- package/dist/generation/emit/comments-part.js.map +1 -1
- package/dist/generation/emit/paragraph.d.ts +3 -0
- package/dist/generation/emit/paragraph.d.ts.map +1 -1
- package/dist/generation/emit/paragraph.js +3 -0
- package/dist/generation/emit/paragraph.js.map +1 -1
- package/dist/generation/emit/run.d.ts.map +1 -1
- package/dist/generation/emit/run.js +6 -1
- package/dist/generation/emit/run.js.map +1 -1
- package/dist/generation/emit/settings-part.d.ts +12 -3
- package/dist/generation/emit/settings-part.d.ts.map +1 -1
- package/dist/generation/emit/settings-part.js +23 -5
- package/dist/generation/emit/settings-part.js.map +1 -1
- package/dist/generation/ordering.d.ts +3 -1
- package/dist/generation/ordering.d.ts.map +1 -1
- package/dist/generation/ordering.js +3 -1
- package/dist/generation/ordering.js.map +1 -1
- package/dist/generation/schema-enum-domains.d.ts +27 -0
- package/dist/generation/schema-enum-domains.d.ts.map +1 -0
- package/dist/generation/schema-enum-domains.js +69 -0
- package/dist/generation/schema-enum-domains.js.map +1 -0
- package/dist/generation/structural-checks.js +7 -1
- package/dist/generation/structural-checks.js.map +1 -1
- package/dist/generation/validate-spec.d.ts.map +1 -1
- package/dist/generation/validate-spec.js +149 -31
- package/dist/generation/validate-spec.js.map +1 -1
- package/dist/index.d.ts +7 -24
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -46
- package/dist/index.js.map +1 -1
- package/dist/primitives/accept_ai_edits.d.ts +87 -0
- package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
- package/dist/primitives/accept_ai_edits.js +253 -0
- package/dist/primitives/accept_ai_edits.js.map +1 -0
- package/dist/primitives/accept_changes.d.ts +36 -3
- package/dist/primitives/accept_changes.d.ts.map +1 -1
- package/dist/primitives/accept_changes.js +67 -31
- package/dist/primitives/accept_changes.js.map +1 -1
- package/dist/primitives/bookmarks.d.ts +44 -0
- package/dist/primitives/bookmarks.d.ts.map +1 -1
- package/dist/primitives/bookmarks.js +149 -11
- package/dist/primitives/bookmarks.js.map +1 -1
- package/dist/primitives/document.d.ts +53 -1
- package/dist/primitives/document.d.ts.map +1 -1
- package/dist/primitives/document.js +146 -2
- package/dist/primitives/document.js.map +1 -1
- package/dist/primitives/dom-helpers.d.ts.map +1 -1
- package/dist/primitives/dom-helpers.js +7 -2
- package/dist/primitives/dom-helpers.js.map +1 -1
- package/dist/primitives/index.d.ts +2 -1
- package/dist/primitives/index.d.ts.map +1 -1
- package/dist/primitives/index.js +2 -1
- package/dist/primitives/index.js.map +1 -1
- package/dist/primitives/layout.d.ts.map +1 -1
- package/dist/primitives/layout.js +41 -1
- package/dist/primitives/layout.js.map +1 -1
- package/dist/primitives/namespaces.d.ts +2 -0
- package/dist/primitives/namespaces.d.ts.map +1 -1
- package/dist/primitives/namespaces.js +2 -0
- package/dist/primitives/namespaces.js.map +1 -1
- package/dist/primitives/reject_changes.d.ts +21 -2
- package/dist/primitives/reject_changes.d.ts.map +1 -1
- package/dist/primitives/reject_changes.js +71 -31
- package/dist/primitives/reject_changes.js.map +1 -1
- package/dist/primitives/relationships.d.ts +33 -0
- package/dist/primitives/relationships.d.ts.map +1 -1
- package/dist/primitives/relationships.js +84 -1
- package/dist/primitives/relationships.js.map +1 -1
- package/dist/primitives/sectPrAudit.d.ts +11 -2
- package/dist/primitives/sectPrAudit.d.ts.map +1 -1
- package/dist/primitives/sectPrAudit.js +148 -23
- package/dist/primitives/sectPrAudit.js.map +1 -1
- package/dist/primitives/track-changes-emitter.d.ts +4 -0
- package/dist/primitives/track-changes-emitter.d.ts.map +1 -1
- package/dist/primitives/track-changes-emitter.js +5 -2
- package/dist/primitives/track-changes-emitter.js.map +1 -1
- package/dist/primitives/validate_ai_revisions.d.ts.map +1 -1
- package/dist/primitives/validate_ai_revisions.js +13 -5
- package/dist/primitives/validate_ai_revisions.js.map +1 -1
- package/dist/primitives/zip.d.ts.map +1 -1
- package/dist/primitives/zip.js +6 -2
- package/dist/primitives/zip.js.map +1 -1
- package/dist/shared/field-structure.d.ts +6 -0
- package/dist/shared/field-structure.d.ts.map +1 -1
- package/dist/shared/field-structure.js +6 -0
- package/dist/shared/field-structure.js.map +1 -1
- package/package.json +4 -8
- package/dist/atomizer.d.ts +0 -273
- package/dist/atomizer.d.ts.map +0 -1
- package/dist/atomizer.js +0 -1002
- package/dist/atomizer.js.map +0 -1
- package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
- package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/atomLcs.js +0 -376
- package/dist/baselines/atomizer/atomLcs.js.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
- package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
- package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
- package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
- package/dist/baselines/atomizer/debug.d.ts +0 -41
- package/dist/baselines/atomizer/debug.d.ts.map +0 -1
- package/dist/baselines/atomizer/debug.js +0 -85
- package/dist/baselines/atomizer/debug.js.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
- package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
- package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
- package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.js +0 -449
- package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
- package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
- package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
- package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
- package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
- package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
- package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.js +0 -209
- package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
- package/dist/baselines/atomizer/pipeline.d.ts +0 -103
- package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
- package/dist/baselines/atomizer/pipeline.js +0 -1160
- package/dist/baselines/atomizer/pipeline.js.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
- package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.js +0 -153
- package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
- package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
- package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
- package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
- package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
- package/dist/cli/compare-two.d.ts +0 -28
- package/dist/cli/compare-two.d.ts.map +0 -1
- package/dist/cli/compare-two.js +0 -112
- package/dist/cli/compare-two.js.map +0 -1
- package/dist/cli/index.d.ts +0 -3
- package/dist/cli/index.d.ts.map +0 -1
- package/dist/cli/index.js +0 -25
- package/dist/cli/index.js.map +0 -1
- package/dist/compare-types.d.ts +0 -197
- package/dist/compare-types.d.ts.map +0 -1
- package/dist/compare-types.js +0 -2
- package/dist/compare-types.js.map +0 -1
- package/dist/format-detection.d.ts +0 -120
- package/dist/format-detection.d.ts.map +0 -1
- package/dist/format-detection.js +0 -339
- package/dist/format-detection.js.map +0 -1
- package/dist/move-detection.d.ts +0 -211
- package/dist/move-detection.d.ts.map +0 -1
- package/dist/move-detection.js +0 -390
- package/dist/move-detection.js.map +0 -1
|
@@ -1,1449 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Document Reconstructor
|
|
3
|
-
*
|
|
4
|
-
* Rebuilds document.xml from marked atoms with track changes.
|
|
5
|
-
* Generates w:ins, w:del, w:moveFrom, w:moveTo elements as appropriate.
|
|
6
|
-
*/
|
|
7
|
-
import { XMLSerializer } from '@xmldom/xmldom';
|
|
8
|
-
import { parseXml } from '../../primitives/xml.js';
|
|
9
|
-
import { CorrelationStatus } from '../../core-types.js';
|
|
10
|
-
import { getLeafText, childElements, findChildByTagName } from '../../primitives/index.js';
|
|
11
|
-
import { allocateRevisionId, buildPPrChangeElement, convertSerializedDeletionContent, createRevisionContext, createRevisionIdState, escapeXmlAttr, formatDate, wrapSerializedContentWithDel, wrapSerializedContentWithIns, } from '../../primitives/track-changes-emitter.js';
|
|
12
|
-
import { serializeToXml, cloneElement } from './xmlToWmlElement.js';
|
|
13
|
-
import { EMPTY_PARAGRAPH_TAG, isParagraphLevelLeaf, nearestHyperlinkAncestor } from '../../atomizer.js';
|
|
14
|
-
import { enforceConsumerCompatibility } from './consumerCompatibility.js';
|
|
15
|
-
import { placeParagraphMarkRevisionMarker } from './inPlaceModifier-wrappers.js';
|
|
16
|
-
import { areRunPropertiesEqual } from '../../format-detection.js';
|
|
17
|
-
import { debug } from './debug.js';
|
|
18
|
-
const SYNTHETIC_DOC = parseXml('<root xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"/>');
|
|
19
|
-
const W_NS = 'http://schemas.openxmlformats.org/wordprocessingml/2006/main';
|
|
20
|
-
function createEl(tag, attrs) {
|
|
21
|
-
const el = SYNTHETIC_DOC.createElementNS(W_NS, tag);
|
|
22
|
-
if (attrs)
|
|
23
|
-
for (const [k, v] of Object.entries(attrs))
|
|
24
|
-
el.setAttribute(k, v);
|
|
25
|
-
return el;
|
|
26
|
-
}
|
|
27
|
-
/**
|
|
28
|
-
* Get or allocate move range IDs for a move name.
|
|
29
|
-
*/
|
|
30
|
-
function getMoveRangeIds(state, moveName) {
|
|
31
|
-
let ids = state.moveRangeIds.get(moveName);
|
|
32
|
-
if (!ids) {
|
|
33
|
-
ids = {
|
|
34
|
-
sourceRangeId: allocateRevisionId(state),
|
|
35
|
-
destRangeId: allocateRevisionId(state),
|
|
36
|
-
};
|
|
37
|
-
state.moveRangeIds.set(moveName, ids);
|
|
38
|
-
}
|
|
39
|
-
return ids;
|
|
40
|
-
}
|
|
41
|
-
/**
|
|
42
|
-
* Reconstruct document.xml from merged atoms with track changes.
|
|
43
|
-
*
|
|
44
|
-
* @param mergedAtoms - Atoms with correlation status set
|
|
45
|
-
* @param originalXml - Original document.xml for structure preservation
|
|
46
|
-
* @param options - Reconstruction options
|
|
47
|
-
* @returns New document.xml with track changes
|
|
48
|
-
*/
|
|
49
|
-
export function reconstructDocument(mergedAtoms, originalXml, options) {
|
|
50
|
-
const { author, date } = options;
|
|
51
|
-
const dateStr = formatDate(date);
|
|
52
|
-
const revState = createRevisionIdState();
|
|
53
|
-
// Group atoms by paragraph
|
|
54
|
-
const rawParagraphGroups = groupAtomsByParagraph(mergedAtoms);
|
|
55
|
-
// Consolidate adjacent same-status changes for better readability
|
|
56
|
-
const paragraphGroups = consolidateAdjacentChanges(rawParagraphGroups);
|
|
57
|
-
// Reset debug counters
|
|
58
|
-
resetDebugCounters();
|
|
59
|
-
resetEmptyParagraphCounters();
|
|
60
|
-
debug('reconstructor', `${mergedAtoms.length} atoms -> ${paragraphGroups.length} paragraphs`);
|
|
61
|
-
// Build track changes XML for each paragraph
|
|
62
|
-
const paragraphXmls = [];
|
|
63
|
-
for (const group of paragraphGroups) {
|
|
64
|
-
const paragraphXml = buildParagraphXml(group, author, dateStr, revState);
|
|
65
|
-
paragraphXmls.push(paragraphXml);
|
|
66
|
-
}
|
|
67
|
-
const counters = getDebugCounters();
|
|
68
|
-
debug('reconstructor', `buildRunContent processed: ${counters.atoms} atoms, ${counters.wt} w:t elements`);
|
|
69
|
-
const emptyCounters = getEmptyParagraphCounters();
|
|
70
|
-
debug('reconstructor', `Empty paragraphs: inserted=${emptyCounters.inserted}, deleted=${emptyCounters.deleted}, equal=${emptyCounters.equal}, other=${emptyCounters.other}`);
|
|
71
|
-
// Reconstruct the document, preserving original body structure (tables, SDTs, etc.)
|
|
72
|
-
return buildDocumentPreservingStructure(originalXml, paragraphXmls, paragraphGroups, () => allocateRevisionId(revState));
|
|
73
|
-
}
|
|
74
|
-
/**
|
|
75
|
-
* Group atoms by paragraph based on their ancestor chain.
|
|
76
|
-
*
|
|
77
|
-
* First sorts atoms by paragraphIndex to ensure all atoms belonging to the same
|
|
78
|
-
* paragraph are contiguous, then groups them sequentially.
|
|
79
|
-
*/
|
|
80
|
-
function groupAtomsByParagraph(atoms) {
|
|
81
|
-
const groups = [];
|
|
82
|
-
let currentGroup = null;
|
|
83
|
-
let currentRunGroup = null;
|
|
84
|
-
const uniqueIndices = new Set(atoms.map(a => a.paragraphIndex));
|
|
85
|
-
debug('reconstructor', `groupAtomsByParagraph: ${atoms.length} atoms, ${uniqueIndices.size} unique paragraphIndices`);
|
|
86
|
-
// Sort atoms by paragraphIndex to ensure all atoms with the same index are contiguous.
|
|
87
|
-
// Use stable sort to preserve relative order within the same paragraph (deleted before inserted).
|
|
88
|
-
const sortedAtoms = [...atoms].sort((a, b) => {
|
|
89
|
-
const aIdx = a.paragraphIndex ?? Number.MAX_SAFE_INTEGER;
|
|
90
|
-
const bIdx = b.paragraphIndex ?? Number.MAX_SAFE_INTEGER;
|
|
91
|
-
return aIdx - bIdx;
|
|
92
|
-
});
|
|
93
|
-
for (const atom of sortedAtoms) {
|
|
94
|
-
// Find paragraph ancestor
|
|
95
|
-
const pAncestor = findAncestorByTag(atom, 'w:p');
|
|
96
|
-
const rAncestor = findAncestorByTag(atom, 'w:r');
|
|
97
|
-
// Check if we need a new paragraph
|
|
98
|
-
const pPr = pAncestor ? findChildByTag(pAncestor, 'w:pPr') : null;
|
|
99
|
-
// Pass currentRunGroup and current atom to check if we should start a new paragraph
|
|
100
|
-
// Uses paragraphIndex for comparison instead of object references
|
|
101
|
-
if (!currentGroup || shouldStartNewParagraph(currentGroup, currentRunGroup, atom)) {
|
|
102
|
-
if (currentRunGroup && currentGroup) {
|
|
103
|
-
currentGroup.runGroups.push(currentRunGroup);
|
|
104
|
-
}
|
|
105
|
-
currentRunGroup = null;
|
|
106
|
-
currentGroup = {
|
|
107
|
-
pPr: pPr ? cloneElement(pPr) : null,
|
|
108
|
-
runGroups: [],
|
|
109
|
-
};
|
|
110
|
-
groups.push(currentGroup);
|
|
111
|
-
}
|
|
112
|
-
// Check if we need a new run group
|
|
113
|
-
// Use the first-class rPr field from the atom when available,
|
|
114
|
-
// falling back to ancestor walk for atoms created before rPr was populated.
|
|
115
|
-
const atomRPr = getEffectiveAtomRPr(atom);
|
|
116
|
-
const rPr = atomRPr ?? (rAncestor ? findChildByTag(rAncestor, 'w:rPr') : null);
|
|
117
|
-
if (!currentRunGroup || shouldStartNewRunGroup(currentRunGroup, atom)) {
|
|
118
|
-
if (currentRunGroup) {
|
|
119
|
-
currentGroup.runGroups.push(currentRunGroup);
|
|
120
|
-
}
|
|
121
|
-
currentRunGroup = {
|
|
122
|
-
status: atom.correlationStatus,
|
|
123
|
-
atoms: [atom],
|
|
124
|
-
rPr: rPr ? cloneElement(rPr) : null,
|
|
125
|
-
moveName: atom.moveName,
|
|
126
|
-
};
|
|
127
|
-
}
|
|
128
|
-
else {
|
|
129
|
-
currentRunGroup.atoms.push(atom);
|
|
130
|
-
}
|
|
131
|
-
}
|
|
132
|
-
// Don't forget the last groups
|
|
133
|
-
if (currentRunGroup && currentGroup) {
|
|
134
|
-
currentGroup.runGroups.push(currentRunGroup);
|
|
135
|
-
}
|
|
136
|
-
return groups;
|
|
137
|
-
}
|
|
138
|
-
/**
|
|
139
|
-
* Check if a RunGroup contains only whitespace.
|
|
140
|
-
*/
|
|
141
|
-
function isWhitespaceOnlyGroup(group) {
|
|
142
|
-
return group.atoms.every(atom => {
|
|
143
|
-
const text = getLeafText(atom.contentElement) ?? '';
|
|
144
|
-
return text.trim() === '';
|
|
145
|
-
});
|
|
146
|
-
}
|
|
147
|
-
/**
|
|
148
|
-
* Reorder atoms within change blocks.
|
|
149
|
-
*
|
|
150
|
-
* Identifies "change blocks" (contiguous regions with Del/Ins) and reorders
|
|
151
|
-
* to put all deletions first, then all insertions.
|
|
152
|
-
* Whitespace between changes is duplicated into both groups to preserve it
|
|
153
|
-
* regardless of accept/reject.
|
|
154
|
-
*/
|
|
155
|
-
function reorderChangeBlocks(groups) {
|
|
156
|
-
for (const paraGroup of groups) {
|
|
157
|
-
const runGroups = paraGroup.runGroups;
|
|
158
|
-
const result = [];
|
|
159
|
-
let i = 0;
|
|
160
|
-
while (i < runGroups.length) {
|
|
161
|
-
const current = runGroups[i];
|
|
162
|
-
// Check if we're entering a change block
|
|
163
|
-
const isChange = current.status === CorrelationStatus.Deleted ||
|
|
164
|
-
current.status === CorrelationStatus.Inserted;
|
|
165
|
-
if (!isChange) {
|
|
166
|
-
result.push(current);
|
|
167
|
-
i++;
|
|
168
|
-
continue;
|
|
169
|
-
}
|
|
170
|
-
// Collect the entire change block
|
|
171
|
-
const deletions = [];
|
|
172
|
-
const insertions = [];
|
|
173
|
-
while (i < runGroups.length) {
|
|
174
|
-
const group = runGroups[i];
|
|
175
|
-
if (group.status === CorrelationStatus.Deleted) {
|
|
176
|
-
deletions.push(...group.atoms);
|
|
177
|
-
i++;
|
|
178
|
-
}
|
|
179
|
-
else if (group.status === CorrelationStatus.Inserted) {
|
|
180
|
-
insertions.push(...group.atoms);
|
|
181
|
-
i++;
|
|
182
|
-
}
|
|
183
|
-
else if (group.status === CorrelationStatus.Equal && isWhitespaceOnlyGroup(group)) {
|
|
184
|
-
// Duplicate whitespace into both deletions and insertions
|
|
185
|
-
// so it's preserved regardless of accept/reject
|
|
186
|
-
for (const atom of group.atoms) {
|
|
187
|
-
// Clone for deletions (mark as deleted)
|
|
188
|
-
const delAtom = {
|
|
189
|
-
...atom,
|
|
190
|
-
correlationStatus: CorrelationStatus.Deleted,
|
|
191
|
-
};
|
|
192
|
-
deletions.push(delAtom);
|
|
193
|
-
// Clone for insertions (mark as inserted)
|
|
194
|
-
const insAtom = {
|
|
195
|
-
...atom,
|
|
196
|
-
correlationStatus: CorrelationStatus.Inserted,
|
|
197
|
-
};
|
|
198
|
-
insertions.push(insAtom);
|
|
199
|
-
}
|
|
200
|
-
i++;
|
|
201
|
-
}
|
|
202
|
-
else {
|
|
203
|
-
// Non-whitespace Equal or other status - end of block
|
|
204
|
-
break;
|
|
205
|
-
}
|
|
206
|
-
}
|
|
207
|
-
// Output reordered: all deletions first, then all insertions
|
|
208
|
-
// rPr is set to null — buildRunContent will sub-group atoms by rPr
|
|
209
|
-
if (deletions.length > 0) {
|
|
210
|
-
result.push({
|
|
211
|
-
status: CorrelationStatus.Deleted,
|
|
212
|
-
atoms: deletions,
|
|
213
|
-
rPr: null,
|
|
214
|
-
});
|
|
215
|
-
}
|
|
216
|
-
if (insertions.length > 0) {
|
|
217
|
-
result.push({
|
|
218
|
-
status: CorrelationStatus.Inserted,
|
|
219
|
-
atoms: insertions,
|
|
220
|
-
rPr: null,
|
|
221
|
-
});
|
|
222
|
-
}
|
|
223
|
-
}
|
|
224
|
-
paraGroup.runGroups = result;
|
|
225
|
-
}
|
|
226
|
-
return groups;
|
|
227
|
-
}
|
|
228
|
-
/**
|
|
229
|
-
* Consolidate adjacent RunGroups with the same status within each paragraph.
|
|
230
|
-
*
|
|
231
|
-
* This makes change tracking more readable by grouping consecutive deletions
|
|
232
|
-
* together and consecutive insertions together, rather than interleaving them
|
|
233
|
-
* at the word level.
|
|
234
|
-
*
|
|
235
|
-
* For example, instead of:
|
|
236
|
-
* <del>word1</del><ins>word2</ins> <del>word3</del><ins>word4</ins>
|
|
237
|
-
*
|
|
238
|
-
* We get:
|
|
239
|
-
* <del>word1 word3</del><ins>word2 word4</ins>
|
|
240
|
-
*/
|
|
241
|
-
function consolidateAdjacentChanges(groups) {
|
|
242
|
-
return reorderChangeBlocks(groups);
|
|
243
|
-
}
|
|
244
|
-
/**
|
|
245
|
-
* Find an ancestor element by tag name.
|
|
246
|
-
*/
|
|
247
|
-
function findAncestorByTag(atom, tagName) {
|
|
248
|
-
for (let i = atom.ancestorElements.length - 1; i >= 0; i--) {
|
|
249
|
-
if (atom.ancestorElements[i].tagName === tagName) {
|
|
250
|
-
return atom.ancestorElements[i];
|
|
251
|
-
}
|
|
252
|
-
}
|
|
253
|
-
return null;
|
|
254
|
-
}
|
|
255
|
-
/**
|
|
256
|
-
* Find a child element by tag name.
|
|
257
|
-
*/
|
|
258
|
-
function findChildByTag(element, tagName) {
|
|
259
|
-
for (let i = 0; i < element.childNodes.length; i++) {
|
|
260
|
-
const child = element.childNodes[i];
|
|
261
|
-
if (child.nodeType === 1 && child.tagName === tagName) {
|
|
262
|
-
return child;
|
|
263
|
-
}
|
|
264
|
-
}
|
|
265
|
-
return null;
|
|
266
|
-
}
|
|
267
|
-
/**
|
|
268
|
-
* Determine if we should start a new paragraph.
|
|
269
|
-
*
|
|
270
|
-
* Uses paragraphIndex for comparison instead of object references, because
|
|
271
|
-
* atoms from original and revised documents have different tree objects.
|
|
272
|
-
*
|
|
273
|
-
* @param currentGroup - The current paragraph group being built
|
|
274
|
-
* @param currentRunGroup - The current run group (may not be pushed to currentGroup yet)
|
|
275
|
-
* @param currentAtom - The current atom being processed
|
|
276
|
-
*/
|
|
277
|
-
function shouldStartNewParagraph(currentGroup, currentRunGroup, currentAtom) {
|
|
278
|
-
const currentParagraphIndex = currentAtom.paragraphIndex;
|
|
279
|
-
// If no paragraph index, fall back to false (stay in current paragraph)
|
|
280
|
-
if (currentParagraphIndex === undefined)
|
|
281
|
-
return false;
|
|
282
|
-
// First check currentRunGroup (which may not be pushed to runGroups yet)
|
|
283
|
-
if (currentRunGroup && currentRunGroup.atoms.length > 0) {
|
|
284
|
-
const lastAtom = currentRunGroup.atoms[currentRunGroup.atoms.length - 1];
|
|
285
|
-
const lastParagraphIndex = lastAtom.paragraphIndex;
|
|
286
|
-
// Same paragraph index means same paragraph, even if from different trees
|
|
287
|
-
if (lastParagraphIndex !== undefined) {
|
|
288
|
-
return currentParagraphIndex !== lastParagraphIndex;
|
|
289
|
-
}
|
|
290
|
-
}
|
|
291
|
-
// Fall back to checking runGroups
|
|
292
|
-
if (currentGroup.runGroups.length === 0) {
|
|
293
|
-
return false;
|
|
294
|
-
}
|
|
295
|
-
// Check last atom's paragraph index
|
|
296
|
-
const lastRunGroup = currentGroup.runGroups[currentGroup.runGroups.length - 1];
|
|
297
|
-
if (!lastRunGroup || lastRunGroup.atoms.length === 0) {
|
|
298
|
-
return false;
|
|
299
|
-
}
|
|
300
|
-
const lastAtom = lastRunGroup.atoms[lastRunGroup.atoms.length - 1];
|
|
301
|
-
const lastParagraphIndex = lastAtom.paragraphIndex;
|
|
302
|
-
if (lastParagraphIndex !== undefined) {
|
|
303
|
-
return currentParagraphIndex !== lastParagraphIndex;
|
|
304
|
-
}
|
|
305
|
-
// No paragraph indices available, stay in current paragraph
|
|
306
|
-
return false;
|
|
307
|
-
}
|
|
308
|
-
/**
|
|
309
|
-
* Get the effective rPr for an atom — uses the first-class `rPr` field
|
|
310
|
-
* when available, otherwise returns null.
|
|
311
|
-
*/
|
|
312
|
-
function getEffectiveAtomRPr(atom) {
|
|
313
|
-
return atom.rPr ?? null;
|
|
314
|
-
}
|
|
315
|
-
/**
|
|
316
|
-
* Determine if we should start a new run group.
|
|
317
|
-
*/
|
|
318
|
-
function shouldStartNewRunGroup(currentGroup, atom) {
|
|
319
|
-
// Different status = new group
|
|
320
|
-
if (currentGroup.status !== atom.correlationStatus) {
|
|
321
|
-
return true;
|
|
322
|
-
}
|
|
323
|
-
// Different move name = new group
|
|
324
|
-
if (currentGroup.moveName !== atom.moveName) {
|
|
325
|
-
return true;
|
|
326
|
-
}
|
|
327
|
-
// Skip rPr splitting for MovedSource/MovedDestination: every moved run
|
|
328
|
-
// group is wrapped by wrapWithMoveFrom/wrapWithMoveTo, so splitting one
|
|
329
|
-
// move into several groups would emit moveFromRangeStart/End (resp.
|
|
330
|
-
// moveToRangeStart/End) once per slice with the same w:name and range ids.
|
|
331
|
-
// This stays required now that explicit move-range markers atomize: the
|
|
332
|
-
// synthetic-range suppression keyed off those markers is per paragraph, so
|
|
333
|
-
// a detected move in a marker-free paragraph still synthesizes one range
|
|
334
|
-
// pair per moved run group.
|
|
335
|
-
if (currentGroup.status === CorrelationStatus.MovedSource ||
|
|
336
|
-
currentGroup.status === CorrelationStatus.MovedDestination) {
|
|
337
|
-
return false;
|
|
338
|
-
}
|
|
339
|
-
// Different rPr = new group (prevents formatting bleed between runs)
|
|
340
|
-
const currentRPr = getEffectiveAtomRPr(currentGroup.atoms[currentGroup.atoms.length - 1]);
|
|
341
|
-
const newRPr = getEffectiveAtomRPr(atom);
|
|
342
|
-
// Fast path: reference equality or both null
|
|
343
|
-
if (currentRPr === newRPr)
|
|
344
|
-
return false;
|
|
345
|
-
if (currentRPr === null && newRPr === null)
|
|
346
|
-
return false;
|
|
347
|
-
return !areRunPropertiesEqual(currentRPr, newRPr);
|
|
348
|
-
}
|
|
349
|
-
/**
|
|
350
|
-
* Check if a paragraph group represents an empty paragraph with a specific status.
|
|
351
|
-
*
|
|
352
|
-
* @param group - The paragraph group to check
|
|
353
|
-
* @param status - The correlation status to check for
|
|
354
|
-
* @returns True if all atoms are empty paragraph markers with the given status
|
|
355
|
-
*/
|
|
356
|
-
function isEmptyParagraphWithStatus(group, status) {
|
|
357
|
-
// Check if all run groups contain only empty paragraph atoms with the given status
|
|
358
|
-
for (const runGroup of group.runGroups) {
|
|
359
|
-
// If any atom is not an empty paragraph marker, this is not an empty paragraph
|
|
360
|
-
const hasNonEmptyAtom = runGroup.atoms.some((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
|
|
361
|
-
if (hasNonEmptyAtom) {
|
|
362
|
-
return false;
|
|
363
|
-
}
|
|
364
|
-
// If any atom doesn't have the expected status, return false
|
|
365
|
-
const hasWrongStatus = runGroup.atoms.some((atom) => atom.correlationStatus !== status);
|
|
366
|
-
if (hasWrongStatus) {
|
|
367
|
-
return false;
|
|
368
|
-
}
|
|
369
|
-
}
|
|
370
|
-
// All atoms are empty paragraph markers with the expected status
|
|
371
|
-
return group.runGroups.length > 0;
|
|
372
|
-
}
|
|
373
|
-
// Debug counters for empty paragraphs
|
|
374
|
-
let debugEmptyParaInserted = 0;
|
|
375
|
-
let debugEmptyParaDeleted = 0;
|
|
376
|
-
let debugEmptyParaEqual = 0;
|
|
377
|
-
let debugEmptyParaOther = 0;
|
|
378
|
-
/**
|
|
379
|
-
* Reset empty paragraph debug counters.
|
|
380
|
-
*/
|
|
381
|
-
export function resetEmptyParagraphCounters() {
|
|
382
|
-
debugEmptyParaInserted = 0;
|
|
383
|
-
debugEmptyParaDeleted = 0;
|
|
384
|
-
debugEmptyParaEqual = 0;
|
|
385
|
-
debugEmptyParaOther = 0;
|
|
386
|
-
}
|
|
387
|
-
/**
|
|
388
|
-
* Get empty paragraph debug counters.
|
|
389
|
-
*/
|
|
390
|
-
export function getEmptyParagraphCounters() {
|
|
391
|
-
return {
|
|
392
|
-
inserted: debugEmptyParaInserted,
|
|
393
|
-
deleted: debugEmptyParaDeleted,
|
|
394
|
-
equal: debugEmptyParaEqual,
|
|
395
|
-
other: debugEmptyParaOther,
|
|
396
|
-
};
|
|
397
|
-
}
|
|
398
|
-
/**
|
|
399
|
-
* Check if a paragraph group contains only empty paragraph atoms.
|
|
400
|
-
*/
|
|
401
|
-
function isEmptyParagraphGroup(group) {
|
|
402
|
-
for (const runGroup of group.runGroups) {
|
|
403
|
-
const hasNonEmptyAtom = runGroup.atoms.some((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
|
|
404
|
-
if (hasNonEmptyAtom) {
|
|
405
|
-
return false;
|
|
406
|
-
}
|
|
407
|
-
}
|
|
408
|
-
return group.runGroups.length > 0;
|
|
409
|
-
}
|
|
410
|
-
const NO_EXPLICIT_MOVE_MARKERS = { moveFrom: false, moveTo: false };
|
|
411
|
-
function collectExplicitMoveMarkers(group) {
|
|
412
|
-
let moveFrom = false;
|
|
413
|
-
let moveTo = false;
|
|
414
|
-
for (const runGroup of group.runGroups) {
|
|
415
|
-
for (const atom of runGroup.atoms) {
|
|
416
|
-
const tag = atom.contentElement.tagName;
|
|
417
|
-
if (tag === 'w:moveFromRangeStart' || tag === 'w:moveFromRangeEnd') {
|
|
418
|
-
moveFrom = true;
|
|
419
|
-
}
|
|
420
|
-
else if (tag === 'w:moveToRangeStart' || tag === 'w:moveToRangeEnd') {
|
|
421
|
-
moveTo = true;
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
}
|
|
425
|
-
return { moveFrom, moveTo };
|
|
426
|
-
}
|
|
427
|
-
/**
|
|
428
|
-
* Build XML for a single paragraph with track changes.
|
|
429
|
-
*/
|
|
430
|
-
function buildParagraphXml(group, author, dateStr, revState) {
|
|
431
|
-
const revisionCtx = createRevisionContext({ author, date: dateStr, idState: revState });
|
|
432
|
-
// Track empty paragraph statuses for debugging
|
|
433
|
-
if (isEmptyParagraphGroup(group)) {
|
|
434
|
-
const status = group.runGroups[0]?.atoms[0]?.correlationStatus;
|
|
435
|
-
if (status === CorrelationStatus.Inserted) {
|
|
436
|
-
debugEmptyParaInserted++;
|
|
437
|
-
}
|
|
438
|
-
else if (status === CorrelationStatus.Deleted) {
|
|
439
|
-
debugEmptyParaDeleted++;
|
|
440
|
-
}
|
|
441
|
-
else if (status === CorrelationStatus.Equal) {
|
|
442
|
-
debugEmptyParaEqual++;
|
|
443
|
-
}
|
|
444
|
-
else {
|
|
445
|
-
debugEmptyParaOther++;
|
|
446
|
-
}
|
|
447
|
-
// Debug: log the first few empty paragraphs for investigation
|
|
448
|
-
const debugLimit = 5;
|
|
449
|
-
const totalEmpty = debugEmptyParaInserted + debugEmptyParaDeleted + debugEmptyParaEqual + debugEmptyParaOther;
|
|
450
|
-
if (totalEmpty <= debugLimit) {
|
|
451
|
-
const atoms = group.runGroups.flatMap(rg => rg.atoms);
|
|
452
|
-
const statuses = atoms.map(a => a.correlationStatus).join(', ');
|
|
453
|
-
debug('reconstructor', `Empty paragraph #${totalEmpty}: status=${status}, atomCount=${atoms.length}, atomStatuses=[${statuses}]`);
|
|
454
|
-
}
|
|
455
|
-
}
|
|
456
|
-
// Whole-paragraph insert/delete encoding must match Word/Aspose behavior.
|
|
457
|
-
//
|
|
458
|
-
// IMPORTANT: <w:ins> is not a container for <w:p> in WordprocessingML.
|
|
459
|
-
// Aspose encodes a paragraph insertion like:
|
|
460
|
-
// <w:p>
|
|
461
|
-
// <w:pPr><w:rPr><w:ins .../></w:rPr></w:pPr>
|
|
462
|
-
// <w:ins ...><w:r>...</w:r></w:ins>
|
|
463
|
-
// </w:p>
|
|
464
|
-
//
|
|
465
|
-
// That structure both renders in Word and allows Reject All to remove the paragraph
|
|
466
|
-
// entirely (instead of leaving behind a stub <w:p> break).
|
|
467
|
-
if (isEntireParagraphWithStatus(group, CorrelationStatus.Inserted)) {
|
|
468
|
-
const paraId = allocateRevisionId(revState);
|
|
469
|
-
const insertedRunXml = paragraphHasHyperlinkAtoms(group)
|
|
470
|
-
? buildWholeParagraphRevisionContent(group, (runs) => wrapSerializedContentWithIns(runs, revisionCtx))
|
|
471
|
-
: wrapSerializedContentWithIns(group.runGroups.map((runGroup) => buildRunContentAsPlainRun(runGroup)).join(''), revisionCtx);
|
|
472
|
-
const pPrChangeEl = buildPPrChangeElement(group.pPr, revisionCtx);
|
|
473
|
-
const parts = [];
|
|
474
|
-
parts.push('<w:p>');
|
|
475
|
-
parts.push(serializePPrWithParaRevisionMarker(group.pPr, 'w:ins', paraId, author, dateStr, pPrChangeEl));
|
|
476
|
-
parts.push(insertedRunXml);
|
|
477
|
-
parts.push('</w:p>');
|
|
478
|
-
return parts.join('');
|
|
479
|
-
}
|
|
480
|
-
if (isEntireParagraphWithStatus(group, CorrelationStatus.Deleted)) {
|
|
481
|
-
const paraId = allocateRevisionId(revState);
|
|
482
|
-
const parts = [];
|
|
483
|
-
parts.push('<w:p>');
|
|
484
|
-
parts.push(serializePPrWithParaRevisionMarker(group.pPr, 'w:del', paraId, author, dateStr));
|
|
485
|
-
parts.push(paragraphHasHyperlinkAtoms(group)
|
|
486
|
-
? buildWholeParagraphRevisionContent(group, (runs) => wrapSerializedContentWithDel(runs, revisionCtx))
|
|
487
|
-
: wrapSerializedContentWithDel(group.runGroups.map((runGroup) => buildRunContentAsPlainRun(runGroup)).join(''), revisionCtx));
|
|
488
|
-
parts.push('</w:p>');
|
|
489
|
-
return parts.join('');
|
|
490
|
-
}
|
|
491
|
-
// Empty inserted paragraphs — use paragraph-mark revision marker (same as whole-paragraph).
|
|
492
|
-
// In OOXML, <w:ins> is NOT a valid container for <w:p>. The correct encoding places the
|
|
493
|
-
// marker inside w:pPr > w:rPr.
|
|
494
|
-
if (isEmptyParagraphWithStatus(group, CorrelationStatus.Inserted)) {
|
|
495
|
-
const paraId = allocateRevisionId(revState);
|
|
496
|
-
const pPrChangeEl = buildPPrChangeElement(group.pPr, revisionCtx);
|
|
497
|
-
const pPrXml = serializePPrWithParaRevisionMarker(group.pPr, 'w:ins', paraId, author, dateStr, pPrChangeEl);
|
|
498
|
-
return `<w:p>${pPrXml}</w:p>`;
|
|
499
|
-
}
|
|
500
|
-
// Empty deleted paragraphs — use paragraph-mark revision marker.
|
|
501
|
-
if (isEmptyParagraphWithStatus(group, CorrelationStatus.Deleted)) {
|
|
502
|
-
const paraId = allocateRevisionId(revState);
|
|
503
|
-
const pPrXml = serializePPrWithParaRevisionMarker(group.pPr, 'w:del', paraId, author, dateStr);
|
|
504
|
-
return `<w:p>${pPrXml}</w:p>`;
|
|
505
|
-
}
|
|
506
|
-
const parts = [];
|
|
507
|
-
parts.push('<w:p>');
|
|
508
|
-
// Add paragraph properties
|
|
509
|
-
if (group.pPr) {
|
|
510
|
-
parts.push(serializeToXml(group.pPr));
|
|
511
|
-
}
|
|
512
|
-
// Add run groups with track changes, restoring w:hyperlink wrappers when
|
|
513
|
-
// the paragraph contains hyperlink atoms (issue #368). Hyperlink-free
|
|
514
|
-
// paragraphs keep the legacy per-group emission byte-identical.
|
|
515
|
-
const explicitMoveMarkers = collectExplicitMoveMarkers(group);
|
|
516
|
-
if (paragraphHasHyperlinkAtoms(group)) {
|
|
517
|
-
parts.push(buildRunGroupsWithHyperlinks(group.runGroups, author, dateStr, revState, explicitMoveMarkers));
|
|
518
|
-
}
|
|
519
|
-
else {
|
|
520
|
-
for (const runGroup of group.runGroups) {
|
|
521
|
-
const runXml = buildRunGroupXml(runGroup, author, dateStr, revState, explicitMoveMarkers);
|
|
522
|
-
parts.push(runXml);
|
|
523
|
-
}
|
|
524
|
-
}
|
|
525
|
-
parts.push('</w:p>');
|
|
526
|
-
return parts.join('');
|
|
527
|
-
}
|
|
528
|
-
/**
|
|
529
|
-
* Serialize paragraph properties with a paragraph-level revision marker (w:ins or w:del)
|
|
530
|
-
* placed inside w:pPr > w:rPr, per OOXML spec.
|
|
531
|
-
*
|
|
532
|
-
* DOM-based implementation — replaces the former regex-based approach.
|
|
533
|
-
*/
|
|
534
|
-
function serializePPrWithParaRevisionMarker(pPr, markerTag, id, author, dateStr, pPrChangeEl) {
|
|
535
|
-
// Clone pPr or synthesize empty one.
|
|
536
|
-
const effectivePPr = pPr ? cloneElement(pPr) : createEl('w:pPr');
|
|
537
|
-
// Find or create w:rPr at schema-correct position.
|
|
538
|
-
let rPr = findChildByTagName(effectivePPr, 'w:rPr');
|
|
539
|
-
if (!rPr) {
|
|
540
|
-
rPr = createEl('w:rPr');
|
|
541
|
-
const sectPr = findChildByTagName(effectivePPr, 'w:sectPr');
|
|
542
|
-
const existingPPrChange = findChildByTagName(effectivePPr, 'w:pPrChange');
|
|
543
|
-
const insertBefore = sectPr ?? existingPPrChange ?? null;
|
|
544
|
-
if (insertBefore) {
|
|
545
|
-
effectivePPr.insertBefore(rPr, insertBefore);
|
|
546
|
-
}
|
|
547
|
-
else {
|
|
548
|
-
effectivePPr.appendChild(rPr);
|
|
549
|
-
}
|
|
550
|
-
}
|
|
551
|
-
// Reuse a pre-existing paragraph-mark marker of the same kind cloned from the
|
|
552
|
-
// source pPr (issue #452): CT_ParaRPr allows at most one of each tracked-change
|
|
553
|
-
// child, and the source revision's metadata (author/date/id) outranks a
|
|
554
|
-
// synthetic duplicate. Either way the marker is placed in its schema-correct
|
|
555
|
-
// slot ahead of formatting children.
|
|
556
|
-
const existingMarker = findChildByTagName(rPr, markerTag);
|
|
557
|
-
const marker = existingMarker ??
|
|
558
|
-
createEl(markerTag, {
|
|
559
|
-
'w:id': String(id),
|
|
560
|
-
'w:author': author,
|
|
561
|
-
'w:date': dateStr,
|
|
562
|
-
});
|
|
563
|
-
placeParagraphMarkRevisionMarker(rPr, marker, markerTag);
|
|
564
|
-
// Append pPrChange at end if provided.
|
|
565
|
-
if (pPrChangeEl) {
|
|
566
|
-
effectivePPr.appendChild(pPrChangeEl);
|
|
567
|
-
}
|
|
568
|
-
return serializeToXml(effectivePPr);
|
|
569
|
-
}
|
|
570
|
-
/**
|
|
571
|
-
* Returns true if every atom in the paragraph is of the specified status
|
|
572
|
-
* (ignoring EMPTY_PARAGRAPH_TAG markers).
|
|
573
|
-
*/
|
|
574
|
-
function isEntireParagraphWithStatus(group, status) {
|
|
575
|
-
let sawAnyContent = false;
|
|
576
|
-
let sawTargetStatus = false;
|
|
577
|
-
for (const runGroup of group.runGroups) {
|
|
578
|
-
for (const atom of runGroup.atoms) {
|
|
579
|
-
const el = atom.contentElement;
|
|
580
|
-
if (el.tagName === EMPTY_PARAGRAPH_TAG)
|
|
581
|
-
continue;
|
|
582
|
-
sawAnyContent = true;
|
|
583
|
-
// A whole-paragraph wrap should still apply even if there are "noise" atoms
|
|
584
|
-
// (pure whitespace runs, tabs, breaks) marked Equal due to normalization or
|
|
585
|
-
// LCS alignment. Those atoms would otherwise prevent wrapping and Word would
|
|
586
|
-
// leave an empty <w:p> stub on Reject All.
|
|
587
|
-
const isWhitespaceOnlyText = el.tagName === 'w:t' && ((getLeafText(el) ?? '').trim() === '');
|
|
588
|
-
const isWhitespaceAtom = isWhitespaceOnlyText || el.tagName === 'w:tab' || el.tagName === 'w:br' || el.tagName === 'w:cr';
|
|
589
|
-
if (atom.correlationStatus === status) {
|
|
590
|
-
sawTargetStatus = true;
|
|
591
|
-
continue;
|
|
592
|
-
}
|
|
593
|
-
if (isWhitespaceAtom) {
|
|
594
|
-
continue; // ignore for whole-paragraph classification
|
|
595
|
-
}
|
|
596
|
-
return false;
|
|
597
|
-
}
|
|
598
|
-
}
|
|
599
|
-
// If there's no content at all, let the empty-paragraph handlers deal with it.
|
|
600
|
-
// Also require at least one atom with the target status so we don't wrap equal-only paragraphs.
|
|
601
|
-
return sawAnyContent && sawTargetStatus;
|
|
602
|
-
}
|
|
603
|
-
/**
|
|
604
|
-
* Build a <w:r> without track-change wrappers. Used when the whole paragraph is already
|
|
605
|
-
* wrapped (paragraph-level <w:ins>/<w:del>).
|
|
606
|
-
*
|
|
607
|
-
* When group.rPr is null, sub-groups atoms by per-atom rPr to prevent formatting bleed.
|
|
608
|
-
*/
|
|
609
|
-
function buildRunContentAsPlainRun(group) {
|
|
610
|
-
const contentAtoms = group.atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
|
|
611
|
-
if (contentAtoms.length === 0)
|
|
612
|
-
return '';
|
|
613
|
-
// Paragraph-level markers must sit outside <w:r>; route through the
|
|
614
|
-
// marker-aware helper which buffers run atoms and flushes on each marker.
|
|
615
|
-
if (groupHasParagraphLevelAtoms(group)) {
|
|
616
|
-
return buildRunContentWithParagraphMarkers(group);
|
|
617
|
-
}
|
|
618
|
-
// If group has explicit rPr, emit a single run
|
|
619
|
-
if (group.rPr !== null) {
|
|
620
|
-
return buildSingleRun(group.atoms, group.rPr);
|
|
621
|
-
}
|
|
622
|
-
// No group-level rPr — sub-group by per-atom rPr
|
|
623
|
-
const subGroups = subGroupByRPr(contentAtoms);
|
|
624
|
-
return subGroups.map(sg => buildSingleRun(sg.atoms, sg.rPr)).join('');
|
|
625
|
-
}
|
|
626
|
-
/**
|
|
627
|
-
* Build XML for a run group with appropriate track changes wrapper.
|
|
628
|
-
*
|
|
629
|
-
* `explicitMoveMarkers` reports whether the surrounding paragraph's atom
|
|
630
|
-
* stream already carries explicit moveFromRange / moveToRange markers; moved
|
|
631
|
-
* groups then skip synthetic range emission (see ExplicitMoveMarkers).
|
|
632
|
-
*/
|
|
633
|
-
function buildRunGroupXml(group, author, dateStr, revState, explicitMoveMarkers = NO_EXPLICIT_MOVE_MARKERS) {
|
|
634
|
-
const runContent = buildRunContent(group);
|
|
635
|
-
// If run content is empty (e.g., only empty paragraph atoms), return empty string
|
|
636
|
-
// This avoids generating empty track changes wrappers
|
|
637
|
-
if (!runContent) {
|
|
638
|
-
return '';
|
|
639
|
-
}
|
|
640
|
-
switch (group.status) {
|
|
641
|
-
case CorrelationStatus.Equal:
|
|
642
|
-
case CorrelationStatus.Unknown:
|
|
643
|
-
return runContent;
|
|
644
|
-
case CorrelationStatus.Inserted:
|
|
645
|
-
return wrapWithIns(runContent, author, dateStr, revState);
|
|
646
|
-
case CorrelationStatus.Deleted:
|
|
647
|
-
return wrapWithDel(runContent, author, dateStr, revState);
|
|
648
|
-
case CorrelationStatus.MovedSource:
|
|
649
|
-
return wrapWithMoveFrom(runContent, author, dateStr, group.moveName || 'move1', revState, explicitMoveMarkers.moveFrom);
|
|
650
|
-
case CorrelationStatus.MovedDestination:
|
|
651
|
-
return wrapWithMoveTo(runContent, author, dateStr, group.moveName || 'move1', revState, explicitMoveMarkers.moveTo);
|
|
652
|
-
case CorrelationStatus.FormatChanged:
|
|
653
|
-
// For format changes, we include the run with rPrChange
|
|
654
|
-
return buildFormatChangeRun(group, author, dateStr, revState);
|
|
655
|
-
default:
|
|
656
|
-
return runContent;
|
|
657
|
-
}
|
|
658
|
-
}
|
|
659
|
-
// Debug counter for atoms processed
|
|
660
|
-
let debugAtomCounter = 0;
|
|
661
|
-
let debugWtCounter = 0;
|
|
662
|
-
/**
|
|
663
|
-
* Reset debug counters (for testing).
|
|
664
|
-
*/
|
|
665
|
-
export function resetDebugCounters() {
|
|
666
|
-
debugAtomCounter = 0;
|
|
667
|
-
debugWtCounter = 0;
|
|
668
|
-
}
|
|
669
|
-
/**
|
|
670
|
-
* Get debug counters (for testing).
|
|
671
|
-
*/
|
|
672
|
-
export function getDebugCounters() {
|
|
673
|
-
return { atoms: debugAtomCounter, wt: debugWtCounter };
|
|
674
|
-
}
|
|
675
|
-
/**
|
|
676
|
-
* Sub-group atoms by contiguous rPr — atoms with the same effective rPr
|
|
677
|
-
* stay in one sub-group, a change in rPr starts a new sub-group.
|
|
678
|
-
*/
|
|
679
|
-
function subGroupByRPr(atoms) {
|
|
680
|
-
if (atoms.length === 0)
|
|
681
|
-
return [];
|
|
682
|
-
const result = [];
|
|
683
|
-
let currentRPr = getEffectiveAtomRPr(atoms[0]);
|
|
684
|
-
let currentAtoms = [atoms[0]];
|
|
685
|
-
for (let i = 1; i < atoms.length; i++) {
|
|
686
|
-
const atomRPr = getEffectiveAtomRPr(atoms[i]);
|
|
687
|
-
// Fast path: reference equality or both null
|
|
688
|
-
let same = currentRPr === atomRPr;
|
|
689
|
-
if (!same && currentRPr === null && atomRPr === null) {
|
|
690
|
-
same = true;
|
|
691
|
-
}
|
|
692
|
-
if (!same) {
|
|
693
|
-
same = areRunPropertiesEqual(currentRPr, atomRPr);
|
|
694
|
-
}
|
|
695
|
-
if (same) {
|
|
696
|
-
currentAtoms.push(atoms[i]);
|
|
697
|
-
}
|
|
698
|
-
else {
|
|
699
|
-
result.push({ rPr: currentRPr, atoms: currentAtoms });
|
|
700
|
-
currentRPr = atomRPr;
|
|
701
|
-
currentAtoms = [atoms[i]];
|
|
702
|
-
}
|
|
703
|
-
}
|
|
704
|
-
result.push({ rPr: currentRPr, atoms: currentAtoms });
|
|
705
|
-
return result;
|
|
706
|
-
}
|
|
707
|
-
/**
|
|
708
|
-
* Attribute fingerprint of a w:hyperlink element, used to recognize "the
|
|
709
|
-
* same" hyperlink across the original and revised trees (equal/deleted atoms
|
|
710
|
-
* reference the original tree's element, inserted atoms the revised tree's).
|
|
711
|
-
*/
|
|
712
|
-
function hyperlinkKey(el) {
|
|
713
|
-
const parts = [];
|
|
714
|
-
for (let i = 0; i < el.attributes.length; i++) {
|
|
715
|
-
const attr = el.attributes.item(i);
|
|
716
|
-
if (attr.name.startsWith('xmlns'))
|
|
717
|
-
continue;
|
|
718
|
-
parts.push(`${attr.name}=${attr.value}`);
|
|
719
|
-
}
|
|
720
|
-
return parts.sort().join('\u0000');
|
|
721
|
-
}
|
|
722
|
-
/**
|
|
723
|
-
* Resolve the hyperlink wrapper an atom belongs to, preferring the
|
|
724
|
-
* original-tree element so the re-emitted r:id resolves against the
|
|
725
|
-
* original-based rebuild package: deleted atoms carry original ancestry
|
|
726
|
-
* directly; equal atoms (revised tree) reach it via comparisonUnitAtomBefore.
|
|
727
|
-
*
|
|
728
|
-
* @conformance ECMA-376 edition 5, Part 1 § 17.16.22
|
|
729
|
-
* @see https://github.com/UseJunior/safe-docx/issues/368
|
|
730
|
-
*/
|
|
731
|
-
function resolveHyperlinkForAtom(atom) {
|
|
732
|
-
const own = nearestHyperlinkAncestor(atom);
|
|
733
|
-
if (!own)
|
|
734
|
-
return null;
|
|
735
|
-
if (atom.sourceDocument === 'original') {
|
|
736
|
-
return { element: own, key: hyperlinkKey(own), fromOriginal: true };
|
|
737
|
-
}
|
|
738
|
-
const before = atom.comparisonUnitAtomBefore;
|
|
739
|
-
const beforeHyperlink = before ? nearestHyperlinkAncestor(before) : null;
|
|
740
|
-
// Attribute to the original wrapper only when both trees agree on the
|
|
741
|
-
// hyperlink's attributes. When they differ (e.g. the revision retargeted
|
|
742
|
-
// the link to a new r:id), emitting the original wrapper would pin the
|
|
743
|
-
// still-equal link text to the STALE target in the accepted document —
|
|
744
|
-
// worse than dropping the wrapper. Such atoms fall through to the
|
|
745
|
-
// revised-only policy below instead.
|
|
746
|
-
// TODO(#376): the faithful tracked representation of a retargeted link
|
|
747
|
-
// is delete-old-link + insert-new-link (what Word emits), which needs the
|
|
748
|
-
// hyperlink fingerprint in atom identity so the LCS stops matching text
|
|
749
|
-
// across different link targets.
|
|
750
|
-
if (beforeHyperlink && hyperlinkKey(beforeHyperlink) === hyperlinkKey(own)) {
|
|
751
|
-
return { element: beforeHyperlink, key: hyperlinkKey(beforeHyperlink), fromOriginal: true };
|
|
752
|
-
}
|
|
753
|
-
// Revised-only attribution (purely inserted hyperlink). Emitting its r:id
|
|
754
|
-
// would dangle against the original-based package, so the caller only
|
|
755
|
-
// wraps when the hyperlink carries no relationship reference (anchor-only).
|
|
756
|
-
return { element: own, key: hyperlinkKey(own), fromOriginal: false };
|
|
757
|
-
}
|
|
758
|
-
/**
|
|
759
|
-
* Whether a resolved hyperlink is safe to re-emit. Original-attributed
|
|
760
|
-
* wrappers always are; revised-only wrappers are safe only without an r:id
|
|
761
|
-
* (internal anchor links), because the rebuild package ships the ORIGINAL
|
|
762
|
-
* document.xml.rels and a revised-only r:id would be a dangling reference
|
|
763
|
-
* (Word treats those as a corrupt package). Revised-only r:id hyperlinks
|
|
764
|
-
* keep today's behavior — content emitted unwrapped.
|
|
765
|
-
*/
|
|
766
|
-
function isEmittableHyperlink(resolved) {
|
|
767
|
-
return resolved.fromOriginal || resolved.element.getAttribute('r:id') === null;
|
|
768
|
-
}
|
|
769
|
-
/**
|
|
770
|
-
* True when any atom in the paragraph sits inside a w:hyperlink. Gates the
|
|
771
|
-
* hyperlink-aware emission paths so hyperlink-free paragraphs keep the
|
|
772
|
-
* byte-identical legacy output.
|
|
773
|
-
*/
|
|
774
|
-
function paragraphHasHyperlinkAtoms(group) {
|
|
775
|
-
return group.runGroups.some((rg) => rg.atoms.some((atom) => nearestHyperlinkAncestor(atom) !== null));
|
|
776
|
-
}
|
|
777
|
-
/**
|
|
778
|
-
* Split a RunGroup into contiguous hyperlink-pure sub-groups.
|
|
779
|
-
*
|
|
780
|
-
* Moved groups are returned whole: splitting them would emit
|
|
781
|
-
* moveFromRangeStart/End once per slice, corrupting the move ranges. A move
|
|
782
|
-
* spanning a hyperlink keeps today's unwrapped emission.
|
|
783
|
-
*/
|
|
784
|
-
function splitRunGroupByHyperlink(group) {
|
|
785
|
-
if (group.status === CorrelationStatus.MovedSource ||
|
|
786
|
-
group.status === CorrelationStatus.MovedDestination) {
|
|
787
|
-
return [{ group, hyperlink: null }];
|
|
788
|
-
}
|
|
789
|
-
const segments = [];
|
|
790
|
-
let current = null;
|
|
791
|
-
for (const atom of group.atoms) {
|
|
792
|
-
// Emit-ability is decided per merged bucket, not per atom: an inserted
|
|
793
|
-
// atom inside an otherwise-original hyperlink folds into the adjacent
|
|
794
|
-
// original-attributed bucket via the shared key.
|
|
795
|
-
const resolved = resolveHyperlinkForAtom(atom);
|
|
796
|
-
const key = resolved?.key ?? null;
|
|
797
|
-
if (current && (current.hyperlink?.key ?? null) === key) {
|
|
798
|
-
current.group.atoms.push(atom);
|
|
799
|
-
// Prefer an original-attributed representative within the segment.
|
|
800
|
-
if (resolved?.fromOriginal && current.hyperlink && !current.hyperlink.fromOriginal) {
|
|
801
|
-
current.hyperlink = resolved;
|
|
802
|
-
}
|
|
803
|
-
}
|
|
804
|
-
else {
|
|
805
|
-
current = {
|
|
806
|
-
group: { ...group, atoms: [atom] },
|
|
807
|
-
hyperlink: resolved,
|
|
808
|
-
};
|
|
809
|
-
segments.push(current);
|
|
810
|
-
}
|
|
811
|
-
}
|
|
812
|
-
return segments;
|
|
813
|
-
}
|
|
814
|
-
/**
|
|
815
|
-
* Serialize the opening tag of a re-emitted w:hyperlink wrapper, copying the
|
|
816
|
-
* source element's attributes verbatim (r:id, w:anchor, w:history, ...).
|
|
817
|
-
*/
|
|
818
|
-
function serializeHyperlinkOpenTag(el) {
|
|
819
|
-
const attrs = [];
|
|
820
|
-
for (let i = 0; i < el.attributes.length; i++) {
|
|
821
|
-
const attr = el.attributes.item(i);
|
|
822
|
-
if (attr.name.startsWith('xmlns'))
|
|
823
|
-
continue;
|
|
824
|
-
attrs.push(` ${attr.name}="${escapeXmlAttr(attr.value)}"`);
|
|
825
|
-
}
|
|
826
|
-
return `<w:hyperlink${attrs.join('')}>`;
|
|
827
|
-
}
|
|
828
|
-
/**
|
|
829
|
-
* Merge adjacent segments that resolve to the same hyperlink fingerprint, so
|
|
830
|
-
* an equal/deleted/inserted sequence inside one link shares one wrapper.
|
|
831
|
-
*/
|
|
832
|
-
function mergeAdjacentHyperlinkSegments(segments) {
|
|
833
|
-
const buckets = [];
|
|
834
|
-
for (const segment of segments) {
|
|
835
|
-
const last = buckets[buckets.length - 1];
|
|
836
|
-
if (last && (last.hyperlink?.key ?? null) === (segment.hyperlink?.key ?? null)) {
|
|
837
|
-
last.groups.push(segment.group);
|
|
838
|
-
if (segment.hyperlink?.fromOriginal && last.hyperlink && !last.hyperlink.fromOriginal) {
|
|
839
|
-
last.hyperlink = segment.hyperlink;
|
|
840
|
-
}
|
|
841
|
-
}
|
|
842
|
-
else {
|
|
843
|
-
buckets.push({ hyperlink: segment.hyperlink, groups: [segment.group] });
|
|
844
|
-
}
|
|
845
|
-
}
|
|
846
|
-
return buckets;
|
|
847
|
-
}
|
|
848
|
-
/**
|
|
849
|
-
* Emit a paragraph's run groups with w:hyperlink wrappers restored around the
|
|
850
|
-
* runs whose atoms came from inside a hyperlink. Track-change wrappers nest
|
|
851
|
-
* INSIDE the hyperlink (`<w:hyperlink><w:ins>…`): CT_Hyperlink admits
|
|
852
|
-
* EG_RunLevelElts (w:ins / w:del / range markers), while CT_RunTrackChange
|
|
853
|
-
* does not admit w:hyperlink.
|
|
854
|
-
*
|
|
855
|
-
* @conformance ECMA-376 edition 5, Part 1 § 17.16.22
|
|
856
|
-
* @see https://github.com/UseJunior/safe-docx/issues/368
|
|
857
|
-
*/
|
|
858
|
-
function buildRunGroupsWithHyperlinks(runGroups, author, dateStr, revState, explicitMoveMarkers = NO_EXPLICIT_MOVE_MARKERS) {
|
|
859
|
-
const buckets = mergeAdjacentHyperlinkSegments(runGroups.flatMap(splitRunGroupByHyperlink));
|
|
860
|
-
const parts = [];
|
|
861
|
-
for (const bucket of buckets) {
|
|
862
|
-
const content = bucket.groups
|
|
863
|
-
.map((g) => buildRunGroupXml(g, author, dateStr, revState, explicitMoveMarkers))
|
|
864
|
-
.join('');
|
|
865
|
-
if (!content)
|
|
866
|
-
continue;
|
|
867
|
-
parts.push(bucket.hyperlink && isEmittableHyperlink(bucket.hyperlink)
|
|
868
|
-
? `${serializeHyperlinkOpenTag(bucket.hyperlink.element)}${content}</w:hyperlink>`
|
|
869
|
-
: content);
|
|
870
|
-
}
|
|
871
|
-
return parts.join('');
|
|
872
|
-
}
|
|
873
|
-
/**
|
|
874
|
-
* Whole-paragraph insert/delete emission with hyperlink wrappers restored.
|
|
875
|
-
* Each bucket gets its own revision wrapper so the hyperlink can stay
|
|
876
|
-
* OUTSIDE the w:ins / w:del (see buildRunGroupsWithHyperlinks).
|
|
877
|
-
*/
|
|
878
|
-
function buildWholeParagraphRevisionContent(group, wrap) {
|
|
879
|
-
const buckets = mergeAdjacentHyperlinkSegments(group.runGroups.flatMap(splitRunGroupByHyperlink));
|
|
880
|
-
const parts = [];
|
|
881
|
-
for (const bucket of buckets) {
|
|
882
|
-
const runs = bucket.groups.map((g) => buildRunContentAsPlainRun(g)).join('');
|
|
883
|
-
if (!runs)
|
|
884
|
-
continue;
|
|
885
|
-
const wrapped = wrap(runs);
|
|
886
|
-
parts.push(bucket.hyperlink && isEmittableHyperlink(bucket.hyperlink)
|
|
887
|
-
? `${serializeHyperlinkOpenTag(bucket.hyperlink.element)}${wrapped}</w:hyperlink>`
|
|
888
|
-
: wrapped);
|
|
889
|
-
}
|
|
890
|
-
return parts.join('');
|
|
891
|
-
}
|
|
892
|
-
/**
|
|
893
|
-
* Returns true when any atom in the group is a paragraph-level marker
|
|
894
|
-
* (commentRange / bookmark / moveFromRange / moveToRange / perm) that must
|
|
895
|
-
* be emitted outside <w:r>.
|
|
896
|
-
*/
|
|
897
|
-
function groupHasParagraphLevelAtoms(group) {
|
|
898
|
-
for (const atom of group.atoms) {
|
|
899
|
-
if (isParagraphLevelLeaf(atom.contentElement))
|
|
900
|
-
return true;
|
|
901
|
-
}
|
|
902
|
-
return false;
|
|
903
|
-
}
|
|
904
|
-
/**
|
|
905
|
-
* Marker-aware emission for run groups containing paragraph-level atoms.
|
|
906
|
-
*
|
|
907
|
-
* Walks atoms left-to-right. Run-level atoms accumulate in a buffer; on
|
|
908
|
-
* encountering a paragraph-level atom (or end of group) the buffer is flushed
|
|
909
|
-
* via subGroupByRPr + buildSingleRun (one <w:r> per contiguous rPr) and the
|
|
910
|
-
* marker is emitted as a bare element.
|
|
911
|
-
*
|
|
912
|
-
* group.rPr is intentionally ignored here — RunGroup.rPr is captured from the
|
|
913
|
-
* first atom in groupAtomsByParagraph(), and for moved groups
|
|
914
|
-
* shouldStartNewRunGroup() suppresses rPr-based splitting. Always re-deriving
|
|
915
|
-
* rPr per atom prevents formatting bleed and bogus rPr inheritance from
|
|
916
|
-
* illegally nested markers.
|
|
917
|
-
*/
|
|
918
|
-
function buildRunContentWithParagraphMarkers(group) {
|
|
919
|
-
const parts = [];
|
|
920
|
-
let runBuffer = [];
|
|
921
|
-
const flush = () => {
|
|
922
|
-
if (runBuffer.length === 0)
|
|
923
|
-
return;
|
|
924
|
-
for (const sg of subGroupByRPr(runBuffer)) {
|
|
925
|
-
const run = buildSingleRun(sg.atoms, sg.rPr);
|
|
926
|
-
if (run)
|
|
927
|
-
parts.push(run);
|
|
928
|
-
}
|
|
929
|
-
runBuffer = [];
|
|
930
|
-
};
|
|
931
|
-
for (const atom of group.atoms) {
|
|
932
|
-
if (atom.contentElement.tagName === EMPTY_PARAGRAPH_TAG)
|
|
933
|
-
continue;
|
|
934
|
-
if (isParagraphLevelLeaf(atom.contentElement)) {
|
|
935
|
-
flush();
|
|
936
|
-
parts.push(serializeAtomElement(atom.contentElement));
|
|
937
|
-
}
|
|
938
|
-
else {
|
|
939
|
-
runBuffer.push(atom);
|
|
940
|
-
}
|
|
941
|
-
}
|
|
942
|
-
flush();
|
|
943
|
-
return parts.join('');
|
|
944
|
-
}
|
|
945
|
-
/**
|
|
946
|
-
* Build a single <w:r> element from a set of atoms with the given rPr.
|
|
947
|
-
* Preserves pendingText coalescing, collapsedFieldAtoms expansion,
|
|
948
|
-
* and debug counter increments.
|
|
949
|
-
*/
|
|
950
|
-
function buildSingleRun(atoms, rPr) {
|
|
951
|
-
const contentAtoms = atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
|
|
952
|
-
if (contentAtoms.length === 0)
|
|
953
|
-
return '';
|
|
954
|
-
const parts = [];
|
|
955
|
-
parts.push('<w:r>');
|
|
956
|
-
if (rPr)
|
|
957
|
-
parts.push(serializeToXml(rPr));
|
|
958
|
-
let pendingText = '';
|
|
959
|
-
const flushPendingText = () => {
|
|
960
|
-
if (!pendingText)
|
|
961
|
-
return;
|
|
962
|
-
const escaped = escapeXmlText(pendingText);
|
|
963
|
-
const needsPreserve = pendingText.startsWith(' ') ||
|
|
964
|
-
pendingText.endsWith(' ') ||
|
|
965
|
-
pendingText.includes(' ');
|
|
966
|
-
parts.push(needsPreserve
|
|
967
|
-
? `<w:t xml:space="preserve">${escaped}</w:t>`
|
|
968
|
-
: `<w:t>${escaped}</w:t>`);
|
|
969
|
-
pendingText = '';
|
|
970
|
-
};
|
|
971
|
-
for (const atom of contentAtoms) {
|
|
972
|
-
debugAtomCounter++;
|
|
973
|
-
if (atom.collapsedFieldAtoms && atom.collapsedFieldAtoms.length > 0) {
|
|
974
|
-
flushPendingText();
|
|
975
|
-
for (const fieldAtom of atom.collapsedFieldAtoms) {
|
|
976
|
-
parts.push(serializeAtomElement(fieldAtom.contentElement));
|
|
977
|
-
}
|
|
978
|
-
continue;
|
|
979
|
-
}
|
|
980
|
-
const el = atom.contentElement;
|
|
981
|
-
if (el.tagName === 'w:t') {
|
|
982
|
-
pendingText += getLeafText(el) ?? '';
|
|
983
|
-
continue;
|
|
984
|
-
}
|
|
985
|
-
flushPendingText();
|
|
986
|
-
parts.push(serializeAtomElement(el));
|
|
987
|
-
}
|
|
988
|
-
flushPendingText();
|
|
989
|
-
parts.push('</w:r>');
|
|
990
|
-
return parts.join('');
|
|
991
|
-
}
|
|
992
|
-
/**
|
|
993
|
-
* Serialize an atom's content element to XML string.
|
|
994
|
-
*/
|
|
995
|
-
function serializeAtomElement(element) {
|
|
996
|
-
if (element.tagName === 'w:t') {
|
|
997
|
-
debugWtCounter++;
|
|
998
|
-
// Text element - preserve xml:space if needed
|
|
999
|
-
const text = escapeXmlText(getLeafText(element) ?? '');
|
|
1000
|
-
if (text.startsWith(' ') || text.endsWith(' ') || text.includes(' ')) {
|
|
1001
|
-
return `<w:t xml:space="preserve">${text}</w:t>`;
|
|
1002
|
-
}
|
|
1003
|
-
else {
|
|
1004
|
-
return `<w:t>${text}</w:t>`;
|
|
1005
|
-
}
|
|
1006
|
-
}
|
|
1007
|
-
else if (element.tagName === 'w:br') {
|
|
1008
|
-
return '<w:br/>';
|
|
1009
|
-
}
|
|
1010
|
-
else if (element.tagName === 'w:tab') {
|
|
1011
|
-
return '<w:tab/>';
|
|
1012
|
-
}
|
|
1013
|
-
else if (element.tagName === 'w:cr') {
|
|
1014
|
-
return '<w:cr/>';
|
|
1015
|
-
}
|
|
1016
|
-
else {
|
|
1017
|
-
// Other elements (including field chars, instrText) - serialize as-is
|
|
1018
|
-
return serializeToXml(element);
|
|
1019
|
-
}
|
|
1020
|
-
}
|
|
1021
|
-
/**
|
|
1022
|
-
* Build the content of a run from atoms.
|
|
1023
|
-
*
|
|
1024
|
-
* Returns empty string if all atoms are empty paragraph markers,
|
|
1025
|
-
* which ensures no empty <w:r> elements are generated.
|
|
1026
|
-
*
|
|
1027
|
-
* When group.rPr is non-null, emits a single <w:r> with that rPr.
|
|
1028
|
-
* When group.rPr is null (e.g., after reorderChangeBlocks merges atoms
|
|
1029
|
-
* from multiple original RunGroups), sub-groups atoms by their per-atom
|
|
1030
|
-
* rPr and emits one <w:r> per sub-group to prevent formatting bleed.
|
|
1031
|
-
*/
|
|
1032
|
-
function buildRunContent(group) {
|
|
1033
|
-
// Check if this run group contains only empty paragraph atoms
|
|
1034
|
-
const contentAtoms = group.atoms.filter((atom) => atom.contentElement.tagName !== EMPTY_PARAGRAPH_TAG);
|
|
1035
|
-
// If no content atoms, return empty string (don't generate empty run)
|
|
1036
|
-
if (contentAtoms.length === 0) {
|
|
1037
|
-
return '';
|
|
1038
|
-
}
|
|
1039
|
-
// Paragraph-level markers must sit outside <w:r>; route through the
|
|
1040
|
-
// marker-aware helper which buffers run atoms and flushes on each marker.
|
|
1041
|
-
if (groupHasParagraphLevelAtoms(group)) {
|
|
1042
|
-
return buildRunContentWithParagraphMarkers(group);
|
|
1043
|
-
}
|
|
1044
|
-
// If group has explicit rPr, emit a single run
|
|
1045
|
-
if (group.rPr !== null) {
|
|
1046
|
-
return buildSingleRun(group.atoms, group.rPr);
|
|
1047
|
-
}
|
|
1048
|
-
// No group-level rPr — sub-group by per-atom rPr
|
|
1049
|
-
const subGroups = subGroupByRPr(contentAtoms);
|
|
1050
|
-
return subGroups.map(sg => buildSingleRun(sg.atoms, sg.rPr)).join('');
|
|
1051
|
-
}
|
|
1052
|
-
/**
|
|
1053
|
-
* Wrap content with w:ins element.
|
|
1054
|
-
*/
|
|
1055
|
-
function wrapWithIns(content, author, dateStr, revState) {
|
|
1056
|
-
return wrapSerializedContentWithIns(content, createRevisionContext({ author, date: dateStr, idState: revState }));
|
|
1057
|
-
}
|
|
1058
|
-
/**
|
|
1059
|
-
* Wrap content with w:del element.
|
|
1060
|
-
*/
|
|
1061
|
-
function wrapWithDel(content, author, dateStr, revState) {
|
|
1062
|
-
return wrapSerializedContentWithDel(content, createRevisionContext({ author, date: dateStr, idState: revState }));
|
|
1063
|
-
}
|
|
1064
|
-
/**
|
|
1065
|
-
* Wrap content with w:moveFrom elements.
|
|
1066
|
-
*
|
|
1067
|
-
* When `suppressRangeMarkers` is true the paragraph's atom stream already
|
|
1068
|
-
* carries explicit w:moveFromRangeStart/End markers (re-emitted by
|
|
1069
|
-
* buildRunContentWithParagraphMarkers), so only the w:moveFrom wrapper is
|
|
1070
|
-
* synthesized — emitting a second range pair would corrupt the move ranges.
|
|
1071
|
-
*
|
|
1072
|
-
* @see https://github.com/UseJunior/safe-docx/issues/110
|
|
1073
|
-
*/
|
|
1074
|
-
function wrapWithMoveFrom(content, author, dateStr, moveName, revState, suppressRangeMarkers = false) {
|
|
1075
|
-
if (suppressRangeMarkers) {
|
|
1076
|
-
const moveId = allocateRevisionId(revState);
|
|
1077
|
-
const delContent = convertSerializedDeletionContent(content);
|
|
1078
|
-
return `<w:moveFrom w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${delContent}</w:moveFrom>`;
|
|
1079
|
-
}
|
|
1080
|
-
const ids = getMoveRangeIds(revState, moveName);
|
|
1081
|
-
const moveId = allocateRevisionId(revState);
|
|
1082
|
-
const delContent = convertSerializedDeletionContent(content);
|
|
1083
|
-
return (`<w:moveFromRangeStart w:id="${ids.sourceRangeId}" w:name="${moveName}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}"/>` +
|
|
1084
|
-
`<w:moveFrom w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${delContent}</w:moveFrom>` +
|
|
1085
|
-
`<w:moveFromRangeEnd w:id="${ids.sourceRangeId}"/>`);
|
|
1086
|
-
}
|
|
1087
|
-
/**
|
|
1088
|
-
* Wrap content with w:moveTo elements.
|
|
1089
|
-
*
|
|
1090
|
-
* When `suppressRangeMarkers` is true the paragraph's atom stream already
|
|
1091
|
-
* carries explicit w:moveToRangeStart/End markers, so only the w:moveTo
|
|
1092
|
-
* wrapper is synthesized (see wrapWithMoveFrom).
|
|
1093
|
-
*/
|
|
1094
|
-
function wrapWithMoveTo(content, author, dateStr, moveName, revState, suppressRangeMarkers = false) {
|
|
1095
|
-
if (suppressRangeMarkers) {
|
|
1096
|
-
const moveId = allocateRevisionId(revState);
|
|
1097
|
-
return `<w:moveTo w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${content}</w:moveTo>`;
|
|
1098
|
-
}
|
|
1099
|
-
const ids = getMoveRangeIds(revState, moveName);
|
|
1100
|
-
const moveId = allocateRevisionId(revState);
|
|
1101
|
-
return (`<w:moveToRangeStart w:id="${ids.destRangeId}" w:name="${moveName}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}"/>` +
|
|
1102
|
-
`<w:moveTo w:id="${moveId}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">${content}</w:moveTo>` +
|
|
1103
|
-
`<w:moveToRangeEnd w:id="${ids.destRangeId}"/>`);
|
|
1104
|
-
}
|
|
1105
|
-
/**
|
|
1106
|
-
* Build run with format change tracking (w:rPrChange).
|
|
1107
|
-
*/
|
|
1108
|
-
function buildFormatChangeRun(group, author, dateStr, revState) {
|
|
1109
|
-
const parts = [];
|
|
1110
|
-
parts.push('<w:r>');
|
|
1111
|
-
// Build rPr with rPrChange
|
|
1112
|
-
const effectiveRPr = group.rPr ?? group.atoms[0]?.rPr ?? null;
|
|
1113
|
-
if (effectiveRPr || group.atoms[0]?.formatChange) {
|
|
1114
|
-
parts.push('<w:rPr>');
|
|
1115
|
-
// Current properties
|
|
1116
|
-
if (effectiveRPr) {
|
|
1117
|
-
for (const child of childElements(effectiveRPr)) {
|
|
1118
|
-
if (child.tagName !== 'w:rPrChange') {
|
|
1119
|
-
parts.push(serializeToXml(child));
|
|
1120
|
-
}
|
|
1121
|
-
}
|
|
1122
|
-
}
|
|
1123
|
-
// Add rPrChange with old properties (wrapped in w:rPr per OOXML spec).
|
|
1124
|
-
// Kept as the original per-child serialization (NOT delegated to
|
|
1125
|
-
// buildRPrChangeElement) to preserve byte-identical output: xmldom emits
|
|
1126
|
-
// inline `xmlns:w="..."` declarations when serializing detached children,
|
|
1127
|
-
// and downstream consumers may pin on that exact serialized form. The
|
|
1128
|
-
// DOM-aware buildRPrChangeElement helper exists for new primitive code
|
|
1129
|
-
// paths (#136 onward).
|
|
1130
|
-
const formatChange = group.atoms[0]?.formatChange;
|
|
1131
|
-
if (formatChange?.oldRunProperties) {
|
|
1132
|
-
const id = allocateRevisionId(revState);
|
|
1133
|
-
parts.push(`<w:rPrChange w:id="${id}" w:author="${escapeXmlAttr(author)}" w:date="${dateStr}">`);
|
|
1134
|
-
parts.push('<w:rPr>');
|
|
1135
|
-
for (const child of childElements(formatChange.oldRunProperties)) {
|
|
1136
|
-
parts.push(serializeToXml(child));
|
|
1137
|
-
}
|
|
1138
|
-
parts.push('</w:rPr>');
|
|
1139
|
-
parts.push('</w:rPrChange>');
|
|
1140
|
-
}
|
|
1141
|
-
parts.push('</w:rPr>');
|
|
1142
|
-
}
|
|
1143
|
-
// Add atom content
|
|
1144
|
-
for (const atom of group.atoms) {
|
|
1145
|
-
const element = atom.contentElement;
|
|
1146
|
-
if (element.tagName === 'w:t') {
|
|
1147
|
-
const text = escapeXmlText(getLeafText(element) ?? '');
|
|
1148
|
-
if (text.startsWith(' ') || text.endsWith(' ') || text.includes(' ')) {
|
|
1149
|
-
parts.push(`<w:t xml:space="preserve">${text}</w:t>`);
|
|
1150
|
-
}
|
|
1151
|
-
else {
|
|
1152
|
-
parts.push(`<w:t>${text}</w:t>`);
|
|
1153
|
-
}
|
|
1154
|
-
}
|
|
1155
|
-
else {
|
|
1156
|
-
parts.push(serializeToXml(element));
|
|
1157
|
-
}
|
|
1158
|
-
}
|
|
1159
|
-
parts.push('</w:r>');
|
|
1160
|
-
return parts.join('');
|
|
1161
|
-
}
|
|
1162
|
-
/**
|
|
1163
|
-
* Parse the original document body into a structural map.
|
|
1164
|
-
*
|
|
1165
|
-
* Recursively finds ALL <w:p> elements in document order, regardless of
|
|
1166
|
-
* wrapper (tables, SDTs, customXml, nested tables, etc.). This matches
|
|
1167
|
-
* the atomizer's recursive tree walk in atomizer.ts.
|
|
1168
|
-
*/
|
|
1169
|
-
function parseOriginalBodyStructure(originalXml) {
|
|
1170
|
-
const doc = parseXml(originalXml);
|
|
1171
|
-
const bodies = doc.getElementsByTagName('w:body');
|
|
1172
|
-
if (!bodies.length) {
|
|
1173
|
-
throw new Error('Could not find w:body in document');
|
|
1174
|
-
}
|
|
1175
|
-
const body = bodies[0];
|
|
1176
|
-
// getElementsByTagName returns ALL descendants in document order —
|
|
1177
|
-
// this naturally recurses through tables, SDTs, customXml, nested tables, etc.
|
|
1178
|
-
const paragraphs = body.getElementsByTagName('w:p');
|
|
1179
|
-
const slots = [];
|
|
1180
|
-
for (let i = 0; i < paragraphs.length; i++) {
|
|
1181
|
-
const el = paragraphs[i];
|
|
1182
|
-
slots.push({ index: i, element: el, parent: el.parentNode });
|
|
1183
|
-
}
|
|
1184
|
-
return { doc, body, slots };
|
|
1185
|
-
}
|
|
1186
|
-
/**
|
|
1187
|
-
* Determine if a ParagraphGroup is "rooted" (maps to an original paragraph slot)
|
|
1188
|
-
* or "purely inserted" (new content with no original counterpart).
|
|
1189
|
-
*
|
|
1190
|
-
* A group is rooted if ANY run group has a status other than Inserted or
|
|
1191
|
-
* MovedDestination — i.e., it contains content from the original document.
|
|
1192
|
-
* This correctly handles Equal, Deleted, MovedSource, and FormatChanged.
|
|
1193
|
-
*/
|
|
1194
|
-
function isRootedGroup(group) {
|
|
1195
|
-
return group.runGroups.some((rg) => rg.status !== CorrelationStatus.Inserted &&
|
|
1196
|
-
rg.status !== CorrelationStatus.MovedDestination);
|
|
1197
|
-
}
|
|
1198
|
-
/**
|
|
1199
|
-
* Build the final document preserving original body structure.
|
|
1200
|
-
*
|
|
1201
|
-
* Instead of replacing <w:body> content with flat paragraphs, this uses the
|
|
1202
|
-
* original body DOM as a scaffold: rooted paragraphs replace their corresponding
|
|
1203
|
-
* <w:p> slots, inserted paragraphs are placed adjacent to their context, and
|
|
1204
|
-
* all structural wrappers (tables, SDTs, etc.) are preserved.
|
|
1205
|
-
*/
|
|
1206
|
-
function buildDocumentPreservingStructure(originalXml, paragraphXmls, paragraphGroups, allocateRevisionId) {
|
|
1207
|
-
const { doc, body, slots } = parseOriginalBodyStructure(originalXml);
|
|
1208
|
-
let slotCursor = 0;
|
|
1209
|
-
let lastEmittedNode = null;
|
|
1210
|
-
// Find body-level <w:sectPr> (must stay as last child of body)
|
|
1211
|
-
const bodyChildren = childElements(body);
|
|
1212
|
-
const finalSectPr = bodyChildren.length > 0 &&
|
|
1213
|
-
bodyChildren[bodyChildren.length - 1].tagName === 'w:sectPr'
|
|
1214
|
-
? bodyChildren[bodyChildren.length - 1]
|
|
1215
|
-
: null;
|
|
1216
|
-
for (let i = 0; i < paragraphGroups.length; i++) {
|
|
1217
|
-
const group = paragraphGroups[i];
|
|
1218
|
-
const paraXml = paragraphXmls[i];
|
|
1219
|
-
// Parse the reconstructed paragraph XML into a DOM node
|
|
1220
|
-
const fragDoc = parseXml(`<__wrap xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml" xmlns:w15="http://schemas.microsoft.com/office/word/2012/wordml" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships" xmlns:mc="http://schemas.openxmlformats.org/markup-compatibility/2006">${paraXml}</__wrap>`);
|
|
1221
|
-
const newNode = doc.importNode(fragDoc.documentElement.firstChild, true);
|
|
1222
|
-
if (isRootedGroup(group)) {
|
|
1223
|
-
// Replace the corresponding original <w:p> slot
|
|
1224
|
-
if (slotCursor < slots.length) {
|
|
1225
|
-
const slot = slots[slotCursor];
|
|
1226
|
-
slot.parent.replaceChild(newNode, slot.element);
|
|
1227
|
-
lastEmittedNode = newNode;
|
|
1228
|
-
slotCursor++;
|
|
1229
|
-
}
|
|
1230
|
-
else {
|
|
1231
|
-
// More rooted paragraphs than slots — append to body (before sectPr)
|
|
1232
|
-
if (finalSectPr) {
|
|
1233
|
-
body.insertBefore(newNode, finalSectPr);
|
|
1234
|
-
}
|
|
1235
|
-
else {
|
|
1236
|
-
body.appendChild(newNode);
|
|
1237
|
-
}
|
|
1238
|
-
lastEmittedNode = newNode;
|
|
1239
|
-
}
|
|
1240
|
-
}
|
|
1241
|
-
else {
|
|
1242
|
-
// Inserted paragraph — place in context
|
|
1243
|
-
if (lastEmittedNode) {
|
|
1244
|
-
// Insert after the previous paragraph in the same parent
|
|
1245
|
-
const parent = lastEmittedNode.parentNode;
|
|
1246
|
-
const nextSibling = lastEmittedNode.nextSibling;
|
|
1247
|
-
// Guard: never insert after body-level <w:sectPr>
|
|
1248
|
-
if (nextSibling === finalSectPr && parent === body) {
|
|
1249
|
-
parent.insertBefore(newNode, finalSectPr);
|
|
1250
|
-
}
|
|
1251
|
-
else {
|
|
1252
|
-
parent.insertBefore(newNode, nextSibling);
|
|
1253
|
-
}
|
|
1254
|
-
lastEmittedNode = newNode;
|
|
1255
|
-
}
|
|
1256
|
-
else if (slotCursor < slots.length) {
|
|
1257
|
-
// No previous node — insert before the next rooted slot
|
|
1258
|
-
const nextSlot = slots[slotCursor];
|
|
1259
|
-
nextSlot.parent.insertBefore(newNode, nextSlot.element);
|
|
1260
|
-
lastEmittedNode = newNode;
|
|
1261
|
-
}
|
|
1262
|
-
else {
|
|
1263
|
-
// No context — append to body (before sectPr)
|
|
1264
|
-
if (finalSectPr) {
|
|
1265
|
-
body.insertBefore(newNode, finalSectPr);
|
|
1266
|
-
}
|
|
1267
|
-
else {
|
|
1268
|
-
body.appendChild(newNode);
|
|
1269
|
-
}
|
|
1270
|
-
lastEmittedNode = newNode;
|
|
1271
|
-
}
|
|
1272
|
-
}
|
|
1273
|
-
}
|
|
1274
|
-
// Remove any leftover original <w:p> slots that weren't consumed
|
|
1275
|
-
// (this happens when the original has more paragraphs than the merged result)
|
|
1276
|
-
for (let i = slotCursor; i < slots.length; i++) {
|
|
1277
|
-
const slot = slots[i];
|
|
1278
|
-
slot.parent.removeChild(slot.element);
|
|
1279
|
-
}
|
|
1280
|
-
// Strip inter-paragraph bookmark/comment/move-range/permission markers from
|
|
1281
|
-
// the scaffold. These are bookmarkStart/End, commentRangeStart/End,
|
|
1282
|
-
// moveFromRange*/moveToRange*, and permStart/End elements that were siblings
|
|
1283
|
-
// of <w:p> in the original body. The paragraph rebuilder handles its own
|
|
1284
|
-
// bookmark logic, so keeping these orphaned markers causes unmatched
|
|
1285
|
-
// bookmark IDs. Body-level move-range markers are likewise scaffold
|
|
1286
|
-
// remnants: in-paragraph markers travel through the atom stream, and
|
|
1287
|
-
// detected moves synthesize fresh range pairs inside the reconstructed
|
|
1288
|
-
// paragraphs, so a leftover body-level pair would either dangle or double an
|
|
1289
|
-
// emitted range.
|
|
1290
|
-
//
|
|
1291
|
-
// Comment range markers are treated differently: a sibling-level
|
|
1292
|
-
// commentRangeStart/End is the legitimate shape for a comment range that
|
|
1293
|
-
// spans whole paragraphs, and such markers never enter the atom stream
|
|
1294
|
-
// (see isParagraphLevelLeaf in atomizer.ts), so nothing re-emits them.
|
|
1295
|
-
// Stripping them unconditionally destroys multi-paragraph comment ranges
|
|
1296
|
-
// (issue #103). Instead, strip a sibling-level comment range marker only
|
|
1297
|
-
// when its counterpart (same w:id) is absent from the rebuilt body —
|
|
1298
|
-
// i.e., it is a genuinely orphaned scaffold remnant.
|
|
1299
|
-
const SCAFFOLD_STRIP_TAGS = new Set([
|
|
1300
|
-
'w:bookmarkStart', 'w:bookmarkEnd',
|
|
1301
|
-
'w:commentRangeStart', 'w:commentRangeEnd',
|
|
1302
|
-
'w:moveFromRangeStart', 'w:moveFromRangeEnd',
|
|
1303
|
-
'w:moveToRangeStart', 'w:moveToRangeEnd',
|
|
1304
|
-
'w:permStart', 'w:permEnd',
|
|
1305
|
-
]);
|
|
1306
|
-
const COMMENT_RANGE_TAGS = new Set(['w:commentRangeStart', 'w:commentRangeEnd']);
|
|
1307
|
-
const commentRangeStartIds = new Set();
|
|
1308
|
-
const commentRangeEndIds = new Set();
|
|
1309
|
-
for (const el of Array.from(body.getElementsByTagName('*'))) {
|
|
1310
|
-
const id = el.getAttribute('w:id');
|
|
1311
|
-
if (id == null)
|
|
1312
|
-
continue;
|
|
1313
|
-
if (el.tagName === 'w:commentRangeStart')
|
|
1314
|
-
commentRangeStartIds.add(id);
|
|
1315
|
-
else if (el.tagName === 'w:commentRangeEnd')
|
|
1316
|
-
commentRangeEndIds.add(id);
|
|
1317
|
-
}
|
|
1318
|
-
const toRemove = [];
|
|
1319
|
-
for (const el of Array.from(body.getElementsByTagName('*'))) {
|
|
1320
|
-
if (SCAFFOLD_STRIP_TAGS.has(el.tagName) && el.parentNode) {
|
|
1321
|
-
// Only strip if NOT inside a reconstructed <w:p> (i.e., it's a scaffold remnant)
|
|
1322
|
-
let insideParagraph = false;
|
|
1323
|
-
let ancestor = el.parentNode;
|
|
1324
|
-
while (ancestor && ancestor !== body) {
|
|
1325
|
-
if (ancestor.tagName === 'w:p') {
|
|
1326
|
-
insideParagraph = true;
|
|
1327
|
-
break;
|
|
1328
|
-
}
|
|
1329
|
-
ancestor = ancestor.parentNode;
|
|
1330
|
-
}
|
|
1331
|
-
if (insideParagraph)
|
|
1332
|
-
continue;
|
|
1333
|
-
if (COMMENT_RANGE_TAGS.has(el.tagName)) {
|
|
1334
|
-
const id = el.getAttribute('w:id');
|
|
1335
|
-
const counterpartIds = el.tagName === 'w:commentRangeStart'
|
|
1336
|
-
? commentRangeEndIds
|
|
1337
|
-
: commentRangeStartIds;
|
|
1338
|
-
if (id != null && counterpartIds.has(id))
|
|
1339
|
-
continue;
|
|
1340
|
-
}
|
|
1341
|
-
toRemove.push(el);
|
|
1342
|
-
}
|
|
1343
|
-
}
|
|
1344
|
-
for (const el of toRemove) {
|
|
1345
|
-
el.parentNode.removeChild(el);
|
|
1346
|
-
}
|
|
1347
|
-
// Balance bookmarks and enforce consumer-compatibility invariants on the
|
|
1348
|
-
// rebuilt body. This dedupes bookmark Names/IDs, hoists bookmarkStart/End
|
|
1349
|
-
// out of <w:ins>/<w:del> wrappers (so they survive accept/reject), and
|
|
1350
|
-
// synthesizes recovery markers for orphaned starts/ends. Mirrors the
|
|
1351
|
-
// post-processing applied in inplace mode (inPlaceModifier.ts).
|
|
1352
|
-
enforceConsumerCompatibility(body, allocateRevisionId);
|
|
1353
|
-
// Serialize modified body and splice back into original envelope
|
|
1354
|
-
const serializer = new XMLSerializer();
|
|
1355
|
-
let newBodyXml = serializer.serializeToString(body);
|
|
1356
|
-
// Strip redundant xmlns:w declarations from inner elements.
|
|
1357
|
-
// XMLSerializer adds xmlns:w="..." on imported paragraph/rPr nodes because
|
|
1358
|
-
// they were parsed in a separate fragment document. These redundant
|
|
1359
|
-
// redeclarations are valid XML but confuse some OOXML consumers (Pages,
|
|
1360
|
-
// Google Docs) and prevent them from rendering tracked changes.
|
|
1361
|
-
// The w: namespace is already declared on the document root element.
|
|
1362
|
-
const W_NS_DECL = ' xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"';
|
|
1363
|
-
// Keep the first declaration (on <w:body>) but remove all others
|
|
1364
|
-
let firstFound = false;
|
|
1365
|
-
newBodyXml = newBodyXml.replace(new RegExp(W_NS_DECL.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'), 'g'), (match) => {
|
|
1366
|
-
if (!firstFound) {
|
|
1367
|
-
firstFound = true;
|
|
1368
|
-
return match;
|
|
1369
|
-
}
|
|
1370
|
-
return '';
|
|
1371
|
-
});
|
|
1372
|
-
// Replace original body in the full document string
|
|
1373
|
-
const bodyRegex = /<w:body[^>]*>[\s\S]*?<\/w:body>/;
|
|
1374
|
-
return originalXml.replace(bodyRegex, newBodyXml);
|
|
1375
|
-
}
|
|
1376
|
-
/**
|
|
1377
|
-
* Build the final document by replacing body content (legacy flat mode).
|
|
1378
|
-
*
|
|
1379
|
-
* Note: sectPr elements are NOT extracted and appended separately because:
|
|
1380
|
-
* 1. Section properties inside pPr elements are already preserved in the reconstructed paragraphs
|
|
1381
|
-
* 2. The regex to extract "final sectPr" was incorrectly matching sectPr inside pPr elements
|
|
1382
|
-
* and capturing large amounts of body content, causing duplicate text.
|
|
1383
|
-
*
|
|
1384
|
-
* @deprecated Use buildDocumentPreservingStructure instead. Retained as fallback.
|
|
1385
|
-
*/
|
|
1386
|
-
export function buildDocument(originalXml, paragraphXmls) {
|
|
1387
|
-
// Extract document structure
|
|
1388
|
-
const bodyMatch = originalXml.match(/(<w:body[^>]*>)([\s\S]*?)(<\/w:body>)/);
|
|
1389
|
-
if (!bodyMatch) {
|
|
1390
|
-
throw new Error('Could not find w:body in document');
|
|
1391
|
-
}
|
|
1392
|
-
const beforeBody = originalXml.slice(0, originalXml.indexOf(bodyMatch[0]));
|
|
1393
|
-
const bodyOpenTag = bodyMatch[1];
|
|
1394
|
-
const bodyCloseTag = bodyMatch[3];
|
|
1395
|
-
const afterBody = originalXml.slice(originalXml.indexOf(bodyMatch[0]) + bodyMatch[0].length);
|
|
1396
|
-
// Build new body (no separate sectPr extraction - it's in the paragraphs' pPr)
|
|
1397
|
-
const newBodyContent = paragraphXmls.join('\n');
|
|
1398
|
-
return beforeBody + bodyOpenTag + '\n' + newBodyContent + '\n' + bodyCloseTag + afterBody;
|
|
1399
|
-
}
|
|
1400
|
-
/**
|
|
1401
|
-
* Escape XML text content.
|
|
1402
|
-
*/
|
|
1403
|
-
function escapeXmlText(text) {
|
|
1404
|
-
return text
|
|
1405
|
-
.replace(/&/g, '&')
|
|
1406
|
-
.replace(/</g, '<')
|
|
1407
|
-
.replace(/>/g, '>');
|
|
1408
|
-
}
|
|
1409
|
-
/**
|
|
1410
|
-
* Count statistics from merged atoms.
|
|
1411
|
-
*/
|
|
1412
|
-
export function computeReconstructionStats(mergedAtoms) {
|
|
1413
|
-
let insertions = 0;
|
|
1414
|
-
let deletions = 0;
|
|
1415
|
-
let moves = 0;
|
|
1416
|
-
let formatChanges = 0;
|
|
1417
|
-
const paragraphs = new Set();
|
|
1418
|
-
for (const atom of mergedAtoms) {
|
|
1419
|
-
// Count paragraph
|
|
1420
|
-
const pAncestor = findAncestorByTag(atom, 'w:p');
|
|
1421
|
-
if (pAncestor) {
|
|
1422
|
-
paragraphs.add(pAncestor);
|
|
1423
|
-
}
|
|
1424
|
-
// Count by status
|
|
1425
|
-
switch (atom.correlationStatus) {
|
|
1426
|
-
case CorrelationStatus.Inserted:
|
|
1427
|
-
insertions++;
|
|
1428
|
-
break;
|
|
1429
|
-
case CorrelationStatus.Deleted:
|
|
1430
|
-
deletions++;
|
|
1431
|
-
break;
|
|
1432
|
-
case CorrelationStatus.MovedSource:
|
|
1433
|
-
case CorrelationStatus.MovedDestination:
|
|
1434
|
-
moves++;
|
|
1435
|
-
break;
|
|
1436
|
-
case CorrelationStatus.FormatChanged:
|
|
1437
|
-
formatChanges++;
|
|
1438
|
-
break;
|
|
1439
|
-
}
|
|
1440
|
-
}
|
|
1441
|
-
return {
|
|
1442
|
-
paragraphs: paragraphs.size,
|
|
1443
|
-
insertions,
|
|
1444
|
-
deletions,
|
|
1445
|
-
moves: Math.floor(moves / 2), // Source and destination counted separately
|
|
1446
|
-
formatChanges,
|
|
1447
|
-
};
|
|
1448
|
-
}
|
|
1449
|
-
//# sourceMappingURL=documentReconstructor.js.map
|