@usejunior/docx-core 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -83
- package/dist/.tsbuildinfo +1 -1
- package/dist/cli/conformance-adapter.d.ts +14 -0
- package/dist/cli/conformance-adapter.d.ts.map +1 -1
- package/dist/cli/conformance-adapter.js +104 -12
- package/dist/cli/conformance-adapter.js.map +1 -1
- package/dist/footnotes.d.ts +7 -5
- package/dist/footnotes.d.ts.map +1 -1
- package/dist/footnotes.js +7 -5
- package/dist/footnotes.js.map +1 -1
- package/dist/generated/ecma-376-vocabulary.d.ts +93 -0
- package/dist/generated/ecma-376-vocabulary.d.ts.map +1 -0
- package/dist/generated/ecma-376-vocabulary.js +87 -0
- package/dist/generated/ecma-376-vocabulary.js.map +1 -0
- package/dist/generation/compile.js +2 -2
- package/dist/generation/compile.js.map +1 -1
- package/dist/generation/emit/comments-part.d.ts +2 -0
- package/dist/generation/emit/comments-part.d.ts.map +1 -1
- package/dist/generation/emit/comments-part.js +2 -0
- package/dist/generation/emit/comments-part.js.map +1 -1
- package/dist/generation/emit/paragraph.d.ts +3 -0
- package/dist/generation/emit/paragraph.d.ts.map +1 -1
- package/dist/generation/emit/paragraph.js +3 -0
- package/dist/generation/emit/paragraph.js.map +1 -1
- package/dist/generation/emit/run.d.ts.map +1 -1
- package/dist/generation/emit/run.js +6 -1
- package/dist/generation/emit/run.js.map +1 -1
- package/dist/generation/emit/settings-part.d.ts +12 -3
- package/dist/generation/emit/settings-part.d.ts.map +1 -1
- package/dist/generation/emit/settings-part.js +23 -5
- package/dist/generation/emit/settings-part.js.map +1 -1
- package/dist/generation/ordering.d.ts +3 -1
- package/dist/generation/ordering.d.ts.map +1 -1
- package/dist/generation/ordering.js +3 -1
- package/dist/generation/ordering.js.map +1 -1
- package/dist/generation/schema-enum-domains.d.ts +27 -0
- package/dist/generation/schema-enum-domains.d.ts.map +1 -0
- package/dist/generation/schema-enum-domains.js +69 -0
- package/dist/generation/schema-enum-domains.js.map +1 -0
- package/dist/generation/structural-checks.js +7 -1
- package/dist/generation/structural-checks.js.map +1 -1
- package/dist/generation/validate-spec.d.ts.map +1 -1
- package/dist/generation/validate-spec.js +149 -31
- package/dist/generation/validate-spec.js.map +1 -1
- package/dist/index.d.ts +7 -24
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -46
- package/dist/index.js.map +1 -1
- package/dist/primitives/accept_ai_edits.d.ts +87 -0
- package/dist/primitives/accept_ai_edits.d.ts.map +1 -0
- package/dist/primitives/accept_ai_edits.js +253 -0
- package/dist/primitives/accept_ai_edits.js.map +1 -0
- package/dist/primitives/accept_changes.d.ts +36 -3
- package/dist/primitives/accept_changes.d.ts.map +1 -1
- package/dist/primitives/accept_changes.js +67 -31
- package/dist/primitives/accept_changes.js.map +1 -1
- package/dist/primitives/bookmarks.d.ts +44 -0
- package/dist/primitives/bookmarks.d.ts.map +1 -1
- package/dist/primitives/bookmarks.js +149 -11
- package/dist/primitives/bookmarks.js.map +1 -1
- package/dist/primitives/document.d.ts +53 -1
- package/dist/primitives/document.d.ts.map +1 -1
- package/dist/primitives/document.js +146 -2
- package/dist/primitives/document.js.map +1 -1
- package/dist/primitives/dom-helpers.d.ts.map +1 -1
- package/dist/primitives/dom-helpers.js +7 -2
- package/dist/primitives/dom-helpers.js.map +1 -1
- package/dist/primitives/index.d.ts +2 -1
- package/dist/primitives/index.d.ts.map +1 -1
- package/dist/primitives/index.js +2 -1
- package/dist/primitives/index.js.map +1 -1
- package/dist/primitives/layout.d.ts.map +1 -1
- package/dist/primitives/layout.js +41 -1
- package/dist/primitives/layout.js.map +1 -1
- package/dist/primitives/namespaces.d.ts +2 -0
- package/dist/primitives/namespaces.d.ts.map +1 -1
- package/dist/primitives/namespaces.js +2 -0
- package/dist/primitives/namespaces.js.map +1 -1
- package/dist/primitives/reject_changes.d.ts +21 -2
- package/dist/primitives/reject_changes.d.ts.map +1 -1
- package/dist/primitives/reject_changes.js +71 -31
- package/dist/primitives/reject_changes.js.map +1 -1
- package/dist/primitives/relationships.d.ts +33 -0
- package/dist/primitives/relationships.d.ts.map +1 -1
- package/dist/primitives/relationships.js +84 -1
- package/dist/primitives/relationships.js.map +1 -1
- package/dist/primitives/sectPrAudit.d.ts +11 -2
- package/dist/primitives/sectPrAudit.d.ts.map +1 -1
- package/dist/primitives/sectPrAudit.js +148 -23
- package/dist/primitives/sectPrAudit.js.map +1 -1
- package/dist/primitives/track-changes-emitter.d.ts +4 -0
- package/dist/primitives/track-changes-emitter.d.ts.map +1 -1
- package/dist/primitives/track-changes-emitter.js +5 -2
- package/dist/primitives/track-changes-emitter.js.map +1 -1
- package/dist/primitives/validate_ai_revisions.d.ts.map +1 -1
- package/dist/primitives/validate_ai_revisions.js +13 -5
- package/dist/primitives/validate_ai_revisions.js.map +1 -1
- package/dist/primitives/zip.d.ts.map +1 -1
- package/dist/primitives/zip.js +6 -2
- package/dist/primitives/zip.js.map +1 -1
- package/dist/shared/field-structure.d.ts +6 -0
- package/dist/shared/field-structure.d.ts.map +1 -1
- package/dist/shared/field-structure.js +6 -0
- package/dist/shared/field-structure.js.map +1 -1
- package/package.json +4 -8
- package/dist/atomizer.d.ts +0 -273
- package/dist/atomizer.d.ts.map +0 -1
- package/dist/atomizer.js +0 -1002
- package/dist/atomizer.js.map +0 -1
- package/dist/baselines/atomizer/atomLcs.d.ts +0 -82
- package/dist/baselines/atomizer/atomLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/atomLcs.js +0 -376
- package/dist/baselines/atomizer/atomLcs.js.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts +0 -99
- package/dist/baselines/atomizer/auxiliaryIdCollision.d.ts.map +0 -1
- package/dist/baselines/atomizer/auxiliaryIdCollision.js +0 -415
- package/dist/baselines/atomizer/auxiliaryIdCollision.js.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.d.ts +0 -2
- package/dist/baselines/atomizer/consumerCompatibility.d.ts.map +0 -1
- package/dist/baselines/atomizer/consumerCompatibility.js +0 -188
- package/dist/baselines/atomizer/consumerCompatibility.js.map +0 -1
- package/dist/baselines/atomizer/debug.d.ts +0 -41
- package/dist/baselines/atomizer/debug.d.ts.map +0 -1
- package/dist/baselines/atomizer/debug.js +0 -85
- package/dist/baselines/atomizer/debug.js.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.d.ts +0 -75
- package/dist/baselines/atomizer/documentReconstructor.d.ts.map +0 -1
- package/dist/baselines/atomizer/documentReconstructor.js +0 -1449
- package/dist/baselines/atomizer/documentReconstructor.js.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.d.ts +0 -99
- package/dist/baselines/atomizer/formattingFidelity.d.ts.map +0 -1
- package/dist/baselines/atomizer/formattingFidelity.js +0 -449
- package/dist/baselines/atomizer/formattingFidelity.js.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts +0 -121
- package/dist/baselines/atomizer/hierarchicalLcs.d.ts.map +0 -1
- package/dist/baselines/atomizer/hierarchicalLcs.js +0 -753
- package/dist/baselines/atomizer/hierarchicalLcs.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts +0 -37
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js +0 -189
- package/dist/baselines/atomizer/inPlaceModifier-bookmarks.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts +0 -74
- package/dist/baselines/atomizer/inPlaceModifier-containers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-containers.js +0 -171
- package/dist/baselines/atomizer/inPlaceModifier-containers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts +0 -88
- package/dist/baselines/atomizer/inPlaceModifier-deletion.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js +0 -326
- package/dist/baselines/atomizer/inPlaceModifier-deletion.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts +0 -85
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js +0 -402
- package/dist/baselines/atomizer/inPlaceModifier-postprocess.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts +0 -39
- package/dist/baselines/atomizer/inPlaceModifier-presplit.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js +0 -265
- package/dist/baselines/atomizer/inPlaceModifier-presplit.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts +0 -62
- package/dist/baselines/atomizer/inPlaceModifier-shared.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-shared.js +0 -139
- package/dist/baselines/atomizer/inPlaceModifier-shared.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts +0 -198
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js +0 -475
- package/dist/baselines/atomizer/inPlaceModifier-wrappers.js.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.d.ts +0 -27
- package/dist/baselines/atomizer/inPlaceModifier.d.ts.map +0 -1
- package/dist/baselines/atomizer/inPlaceModifier.js +0 -648
- package/dist/baselines/atomizer/inPlaceModifier.js.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.d.ts +0 -59
- package/dist/baselines/atomizer/numberingIntegration.d.ts.map +0 -1
- package/dist/baselines/atomizer/numberingIntegration.js +0 -209
- package/dist/baselines/atomizer/numberingIntegration.js.map +0 -1
- package/dist/baselines/atomizer/pipeline.d.ts +0 -103
- package/dist/baselines/atomizer/pipeline.d.ts.map +0 -1
- package/dist/baselines/atomizer/pipeline.js +0 -1160
- package/dist/baselines/atomizer/pipeline.js.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.d.ts +0 -26
- package/dist/baselines/atomizer/premergeRuns.d.ts.map +0 -1
- package/dist/baselines/atomizer/premergeRuns.js +0 -153
- package/dist/baselines/atomizer/premergeRuns.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts +0 -63
- package/dist/baselines/atomizer/trackChangesAcceptor.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptor.js +0 -254
- package/dist/baselines/atomizer/trackChangesAcceptor.js.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts +0 -64
- package/dist/baselines/atomizer/trackChangesAcceptorAst.d.ts.map +0 -1
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js +0 -642
- package/dist/baselines/atomizer/trackChangesAcceptorAst.js.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts +0 -65
- package/dist/baselines/atomizer/xmlToWmlElement.d.ts.map +0 -1
- package/dist/baselines/atomizer/xmlToWmlElement.js +0 -96
- package/dist/baselines/atomizer/xmlToWmlElement.js.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts +0 -51
- package/dist/baselines/wmlcomparer/DocxodusWasm.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DocxodusWasm.js +0 -83
- package/dist/baselines/wmlcomparer/DocxodusWasm.js.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts +0 -40
- package/dist/baselines/wmlcomparer/DotnetCli.d.ts.map +0 -1
- package/dist/baselines/wmlcomparer/DotnetCli.js +0 -142
- package/dist/baselines/wmlcomparer/DotnetCli.js.map +0 -1
- package/dist/cli/compare-two.d.ts +0 -28
- package/dist/cli/compare-two.d.ts.map +0 -1
- package/dist/cli/compare-two.js +0 -112
- package/dist/cli/compare-two.js.map +0 -1
- package/dist/cli/index.d.ts +0 -3
- package/dist/cli/index.d.ts.map +0 -1
- package/dist/cli/index.js +0 -25
- package/dist/cli/index.js.map +0 -1
- package/dist/compare-types.d.ts +0 -197
- package/dist/compare-types.d.ts.map +0 -1
- package/dist/compare-types.js +0 -2
- package/dist/compare-types.js.map +0 -1
- package/dist/format-detection.d.ts +0 -120
- package/dist/format-detection.d.ts.map +0 -1
- package/dist/format-detection.js +0 -339
- package/dist/format-detection.js.map +0 -1
- package/dist/move-detection.d.ts +0 -211
- package/dist/move-detection.d.ts.map +0 -1
- package/dist/move-detection.js +0 -390
- package/dist/move-detection.js.map +0 -1
|
@@ -1,753 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Hierarchical LCS Comparison
|
|
3
|
-
*
|
|
4
|
-
* Implements two-level comparison like WmlComparer:
|
|
5
|
-
* 1. First pass: LCS on paragraph GROUPS (coarse alignment)
|
|
6
|
-
* 2. Second pass: LCS on atoms WITHIN matched groups (fine alignment)
|
|
7
|
-
*
|
|
8
|
-
* This prevents atoms from one paragraph matching random fragments
|
|
9
|
-
* in other paragraphs.
|
|
10
|
-
*
|
|
11
|
-
* Additionally, large paragraphs (like definition sections) are split
|
|
12
|
-
* on soft breaks (w:br) to prevent cross-definition contamination.
|
|
13
|
-
*/
|
|
14
|
-
import { CorrelationStatus } from '../../core-types.js';
|
|
15
|
-
import { sha1, EMPTY_PARAGRAPH_TAG } from '../../atomizer.js';
|
|
16
|
-
import { computeAtomLcs } from './atomLcs.js';
|
|
17
|
-
import { debug } from './debug.js';
|
|
18
|
-
/**
|
|
19
|
-
* Maximum atoms in a group before we split on w:br boundaries.
|
|
20
|
-
* This handles mega-paragraphs like definition sections.
|
|
21
|
-
*/
|
|
22
|
-
const MAX_ATOMS_BEFORE_SPLIT = 50;
|
|
23
|
-
/**
|
|
24
|
-
* Default paragraph-level similarity threshold used for group matching.
|
|
25
|
-
*
|
|
26
|
-
* Lower values favor treating modified paragraphs as aligned pairs (so atom-level
|
|
27
|
-
* comparison can run), while higher values favor whole-paragraph replacement.
|
|
28
|
-
*/
|
|
29
|
-
export const DEFAULT_PARAGRAPH_SIMILARITY_THRESHOLD = 0.25;
|
|
30
|
-
/**
|
|
31
|
-
* Group atoms by paragraph index.
|
|
32
|
-
*
|
|
33
|
-
* @param atoms - Atoms with paragraphIndex set
|
|
34
|
-
* @returns Array of paragraph groups in document order
|
|
35
|
-
*/
|
|
36
|
-
export function groupAtomsByParagraphIndex(atoms) {
|
|
37
|
-
const groups = new Map();
|
|
38
|
-
for (const atom of atoms) {
|
|
39
|
-
const idx = atom.paragraphIndex ?? -1;
|
|
40
|
-
if (!groups.has(idx)) {
|
|
41
|
-
groups.set(idx, []);
|
|
42
|
-
}
|
|
43
|
-
groups.get(idx).push(atom);
|
|
44
|
-
}
|
|
45
|
-
// Convert to array sorted by paragraph index
|
|
46
|
-
const result = [];
|
|
47
|
-
const sortedIndices = [...groups.keys()].sort((a, b) => a - b);
|
|
48
|
-
for (const idx of sortedIndices) {
|
|
49
|
-
const atoms = groups.get(idx);
|
|
50
|
-
const textContent = extractGroupTextContent(atoms);
|
|
51
|
-
// For empty paragraph groups, use the atom's sha1Hash (which has context)
|
|
52
|
-
// instead of the empty text hash. This prevents all empty paragraphs from
|
|
53
|
-
// matching each other regardless of position.
|
|
54
|
-
const isEmptyParagraphGroup = atoms.length === 1 &&
|
|
55
|
-
atoms[0].contentElement.tagName === EMPTY_PARAGRAPH_TAG;
|
|
56
|
-
const textHash = isEmptyParagraphGroup
|
|
57
|
-
? atoms[0].sha1Hash // Use context-aware atom hash
|
|
58
|
-
: sha1(textContent);
|
|
59
|
-
const normalizedTextHash = isEmptyParagraphGroup
|
|
60
|
-
? textHash
|
|
61
|
-
: sha1(normalizeText(textContent));
|
|
62
|
-
result.push({
|
|
63
|
-
paragraphIndex: idx,
|
|
64
|
-
atoms,
|
|
65
|
-
textHash,
|
|
66
|
-
normalizedTextHash,
|
|
67
|
-
textContent,
|
|
68
|
-
});
|
|
69
|
-
}
|
|
70
|
-
return result;
|
|
71
|
-
}
|
|
72
|
-
/**
|
|
73
|
-
* Create a ComparisonUnitGroup from a list of atoms.
|
|
74
|
-
*/
|
|
75
|
-
function createGroup(atoms, groupIndex) {
|
|
76
|
-
const textContent = extractGroupTextContent(atoms);
|
|
77
|
-
// For empty paragraph groups, use the atom's sha1Hash (which has context)
|
|
78
|
-
const isEmptyGroup = atoms.length === 1 &&
|
|
79
|
-
atoms[0].contentElement.tagName === EMPTY_PARAGRAPH_TAG;
|
|
80
|
-
const textHash = isEmptyGroup
|
|
81
|
-
? atoms[0].sha1Hash
|
|
82
|
-
: sha1(textContent);
|
|
83
|
-
const normalizedTextHash = isEmptyGroup
|
|
84
|
-
? textHash
|
|
85
|
-
: sha1(normalizeText(textContent));
|
|
86
|
-
return {
|
|
87
|
-
paragraphIndex: groupIndex,
|
|
88
|
-
atoms,
|
|
89
|
-
textHash,
|
|
90
|
-
normalizedTextHash,
|
|
91
|
-
textContent,
|
|
92
|
-
};
|
|
93
|
-
}
|
|
94
|
-
/**
|
|
95
|
-
* Group atoms by paragraph, then split large paragraphs on soft breaks (w:br).
|
|
96
|
-
*
|
|
97
|
-
* This handles mega-paragraphs like definition sections where many definitions
|
|
98
|
-
* are in a single paragraph separated by soft breaks. Without this split,
|
|
99
|
-
* the atom-level LCS can match fragments across definition boundaries.
|
|
100
|
-
*
|
|
101
|
-
* @param atoms - Atoms with paragraphIndex set
|
|
102
|
-
* @returns Array of groups, potentially more than the number of paragraphs
|
|
103
|
-
*/
|
|
104
|
-
export function groupAtomsByParagraphAndBreaks(atoms) {
|
|
105
|
-
// First, group by paragraph index
|
|
106
|
-
const paragraphMap = new Map();
|
|
107
|
-
for (const atom of atoms) {
|
|
108
|
-
const idx = atom.paragraphIndex ?? -1;
|
|
109
|
-
if (!paragraphMap.has(idx)) {
|
|
110
|
-
paragraphMap.set(idx, []);
|
|
111
|
-
}
|
|
112
|
-
paragraphMap.get(idx).push(atom);
|
|
113
|
-
}
|
|
114
|
-
// Convert to array sorted by paragraph index
|
|
115
|
-
const sortedIndices = [...paragraphMap.keys()].sort((a, b) => a - b);
|
|
116
|
-
// Now process each paragraph, splitting large ones on w:br
|
|
117
|
-
const result = [];
|
|
118
|
-
let groupIndex = 0;
|
|
119
|
-
for (const paraIdx of sortedIndices) {
|
|
120
|
-
const paraAtoms = paragraphMap.get(paraIdx);
|
|
121
|
-
// Small paragraph - keep as-is
|
|
122
|
-
if (paraAtoms.length <= MAX_ATOMS_BEFORE_SPLIT) {
|
|
123
|
-
result.push(createGroup(paraAtoms, groupIndex++));
|
|
124
|
-
continue;
|
|
125
|
-
}
|
|
126
|
-
// Large paragraph - split on w:br boundaries
|
|
127
|
-
let currentAtoms = [];
|
|
128
|
-
for (const atom of paraAtoms) {
|
|
129
|
-
currentAtoms.push(atom);
|
|
130
|
-
// Split AFTER w:br (keep the break with the preceding content)
|
|
131
|
-
if (atom.contentElement.tagName === 'w:br') {
|
|
132
|
-
if (currentAtoms.length > 0) {
|
|
133
|
-
result.push(createGroup(currentAtoms, groupIndex++));
|
|
134
|
-
currentAtoms = [];
|
|
135
|
-
}
|
|
136
|
-
}
|
|
137
|
-
}
|
|
138
|
-
// Don't forget trailing atoms after last break
|
|
139
|
-
if (currentAtoms.length > 0) {
|
|
140
|
-
result.push(createGroup(currentAtoms, groupIndex++));
|
|
141
|
-
}
|
|
142
|
-
}
|
|
143
|
-
return result;
|
|
144
|
-
}
|
|
145
|
-
/**
|
|
146
|
-
* Extract concatenated text content from a group of atoms.
|
|
147
|
-
* Used for paragraph-level comparison and similarity calculation.
|
|
148
|
-
*/
|
|
149
|
-
function extractGroupTextContent(atoms) {
|
|
150
|
-
const textParts = [];
|
|
151
|
-
for (const atom of atoms) {
|
|
152
|
-
// Treat run separators as visible token boundaries for similarity purposes.
|
|
153
|
-
if (atom.contentElement.tagName === 'w:br' ||
|
|
154
|
-
atom.contentElement.tagName === 'w:cr' ||
|
|
155
|
-
atom.contentElement.tagName === 'w:tab') {
|
|
156
|
-
textParts.push(' ');
|
|
157
|
-
continue;
|
|
158
|
-
}
|
|
159
|
-
const text = atom.contentElement.textContent;
|
|
160
|
-
if (text) {
|
|
161
|
-
textParts.push(text);
|
|
162
|
-
}
|
|
163
|
-
}
|
|
164
|
-
return textParts.join('');
|
|
165
|
-
}
|
|
166
|
-
/**
|
|
167
|
-
* Normalize text for similarity comparison.
|
|
168
|
-
* - Trim whitespace
|
|
169
|
-
* - Collapse multiple spaces
|
|
170
|
-
* - Lowercase for case-insensitive comparison
|
|
171
|
-
*
|
|
172
|
-
* NOTE: Do NOT strip punctuation here — this function feeds normalizedTextHash
|
|
173
|
-
* which is used for Pass 1 anchoring. Changing it would alter which paragraphs
|
|
174
|
-
* are considered coarsely equal. Punctuation stripping is in tokenize() only.
|
|
175
|
-
*/
|
|
176
|
-
function normalizeText(text) {
|
|
177
|
-
return text
|
|
178
|
-
.trim()
|
|
179
|
-
.replace(/\s+/g, ' ')
|
|
180
|
-
.toLowerCase();
|
|
181
|
-
}
|
|
182
|
-
/**
|
|
183
|
-
* Build an IDF (inverse document frequency) map from all paragraph groups.
|
|
184
|
-
*
|
|
185
|
-
* IDF(word) = log(totalGroups / groupsContainingWord)
|
|
186
|
-
*
|
|
187
|
-
* Words appearing in many paragraphs (legal boilerplate like "holders",
|
|
188
|
-
* "Corporation", "Preferred Stock") get low weight. Distinctive words
|
|
189
|
-
* ("Liquidation", "Dividends") get high weight.
|
|
190
|
-
*/
|
|
191
|
-
function buildIdfMap(groups) {
|
|
192
|
-
const docFreq = new Map();
|
|
193
|
-
const totalGroups = groups.length;
|
|
194
|
-
for (const group of groups) {
|
|
195
|
-
if (isEmptyParagraphGroup(group))
|
|
196
|
-
continue;
|
|
197
|
-
const words = new Set(tokenize(group.textContent));
|
|
198
|
-
for (const word of words) {
|
|
199
|
-
docFreq.set(word, (docFreq.get(word) ?? 0) + 1);
|
|
200
|
-
}
|
|
201
|
-
}
|
|
202
|
-
const idf = new Map();
|
|
203
|
-
for (const [word, freq] of docFreq) {
|
|
204
|
-
idf.set(word, Math.log(totalGroups / freq));
|
|
205
|
-
}
|
|
206
|
-
return idf;
|
|
207
|
-
}
|
|
208
|
-
/**
|
|
209
|
-
* Build a precomputed TF-IDF vector for a paragraph group.
|
|
210
|
-
*
|
|
211
|
-
* TF(word) = count(word in paragraph) / totalWords
|
|
212
|
-
* TF-IDF(word) = TF(word) * IDF(word)
|
|
213
|
-
*/
|
|
214
|
-
function buildTfidfVector(group, idf) {
|
|
215
|
-
const words = tokenize(group.textContent);
|
|
216
|
-
if (words.length === 0) {
|
|
217
|
-
return { vector: new Map(), magnitude: 0 };
|
|
218
|
-
}
|
|
219
|
-
// Count term frequencies
|
|
220
|
-
const tf = new Map();
|
|
221
|
-
for (const word of words) {
|
|
222
|
-
tf.set(word, (tf.get(word) ?? 0) + 1);
|
|
223
|
-
}
|
|
224
|
-
// Build TF-IDF vector
|
|
225
|
-
const vector = new Map();
|
|
226
|
-
let sumSquares = 0;
|
|
227
|
-
for (const [word, count] of tf) {
|
|
228
|
-
const tfidf = (count / words.length) * (idf.get(word) ?? 0);
|
|
229
|
-
if (tfidf > 0) {
|
|
230
|
-
vector.set(word, tfidf);
|
|
231
|
-
sumSquares += tfidf * tfidf;
|
|
232
|
-
}
|
|
233
|
-
}
|
|
234
|
-
return { vector, magnitude: Math.sqrt(sumSquares) };
|
|
235
|
-
}
|
|
236
|
-
/**
|
|
237
|
-
* Compute cosine similarity between two precomputed TF-IDF vectors.
|
|
238
|
-
*/
|
|
239
|
-
function computeTfidfCosineSimilarity(a, b) {
|
|
240
|
-
if (a.magnitude === 0 || b.magnitude === 0)
|
|
241
|
-
return 0;
|
|
242
|
-
// Iterate over the smaller vector for efficiency
|
|
243
|
-
const [smaller, larger] = a.vector.size <= b.vector.size ? [a, b] : [b, a];
|
|
244
|
-
let dot = 0;
|
|
245
|
-
for (const [word, weight] of smaller.vector) {
|
|
246
|
-
const otherWeight = larger.vector.get(word);
|
|
247
|
-
if (otherWeight !== undefined) {
|
|
248
|
-
dot += weight * otherWeight;
|
|
249
|
-
}
|
|
250
|
-
}
|
|
251
|
-
return dot / (a.magnitude * b.magnitude);
|
|
252
|
-
}
|
|
253
|
-
/**
|
|
254
|
-
* Tokenize text into words for TF-IDF.
|
|
255
|
-
* Strips punctuation (unlike normalizeText) so that "Corporation," and
|
|
256
|
-
* "Corporation" produce the same token.
|
|
257
|
-
*/
|
|
258
|
-
function tokenize(text) {
|
|
259
|
-
return text
|
|
260
|
-
.trim()
|
|
261
|
-
.replace(/[^\w\s]/g, ' ')
|
|
262
|
-
.replace(/\s+/g, ' ')
|
|
263
|
-
.toLowerCase()
|
|
264
|
-
.trim()
|
|
265
|
-
.split(' ')
|
|
266
|
-
.filter(w => w.length > 0);
|
|
267
|
-
}
|
|
268
|
-
/**
|
|
269
|
-
* Check if a group contains only empty paragraph atoms.
|
|
270
|
-
*/
|
|
271
|
-
function isEmptyParagraphGroup(group) {
|
|
272
|
-
return group.atoms.length === 1 &&
|
|
273
|
-
group.atoms[0].contentElement.tagName === EMPTY_PARAGRAPH_TAG;
|
|
274
|
-
}
|
|
275
|
-
/**
|
|
276
|
-
* Paragraph groups are considered coarse-equal if:
|
|
277
|
-
* 1) Their raw text hash matches exactly, or
|
|
278
|
-
* 2) Their normalized-text hash matches (heuristic assist only).
|
|
279
|
-
*
|
|
280
|
-
* Empty paragraphs intentionally require strict hash equality.
|
|
281
|
-
*/
|
|
282
|
-
function groupsCoarselyEqual(a, b) {
|
|
283
|
-
if (a.textHash === b.textHash) {
|
|
284
|
-
return true;
|
|
285
|
-
}
|
|
286
|
-
if (isEmptyParagraphGroup(a) || isEmptyParagraphGroup(b)) {
|
|
287
|
-
return false;
|
|
288
|
-
}
|
|
289
|
-
return a.normalizedTextHash === b.normalizedTextHash;
|
|
290
|
-
}
|
|
291
|
-
/**
|
|
292
|
-
* Compute similarity between two groups using Jaccard index on words.
|
|
293
|
-
*
|
|
294
|
-
* @returns Value between 0 (completely different) and 1 (identical)
|
|
295
|
-
*/
|
|
296
|
-
function computeGroupSimilarity(a, b) {
|
|
297
|
-
// For empty paragraph groups, only consider them similar if their
|
|
298
|
-
// context-aware hashes match. This prevents empty paragraphs from
|
|
299
|
-
// matching each other regardless of position.
|
|
300
|
-
if (isEmptyParagraphGroup(a) || isEmptyParagraphGroup(b)) {
|
|
301
|
-
// If one is empty and the other isn't, they're not similar
|
|
302
|
-
if (isEmptyParagraphGroup(a) !== isEmptyParagraphGroup(b)) {
|
|
303
|
-
return 0;
|
|
304
|
-
}
|
|
305
|
-
// Both are empty paragraph groups - compare their context-aware hashes
|
|
306
|
-
// They must match exactly (return 1) or not at all (return 0)
|
|
307
|
-
return a.textHash === b.textHash ? 1 : 0;
|
|
308
|
-
}
|
|
309
|
-
const textA = normalizeText(a.textContent);
|
|
310
|
-
const textB = normalizeText(b.textContent);
|
|
311
|
-
const wordsA = new Set(textA.split(' ').filter(w => w.length > 0));
|
|
312
|
-
const wordsB = new Set(textB.split(' ').filter(w => w.length > 0));
|
|
313
|
-
if (wordsA.size === 0 && wordsB.size === 0) {
|
|
314
|
-
return 1; // Both empty
|
|
315
|
-
}
|
|
316
|
-
if (wordsA.size === 0 || wordsB.size === 0) {
|
|
317
|
-
return 0; // One empty
|
|
318
|
-
}
|
|
319
|
-
const intersection = new Set([...wordsA].filter(x => wordsB.has(x)));
|
|
320
|
-
const union = new Set([...wordsA, ...wordsB]);
|
|
321
|
-
return intersection.size / union.size;
|
|
322
|
-
}
|
|
323
|
-
/**
|
|
324
|
-
* Build ordered gaps between consecutive Pass 1 anchors.
|
|
325
|
-
*
|
|
326
|
-
* Anchors divide both documents into regions. Similarity matching is scoped
|
|
327
|
-
* to each gap — a source paragraph can only match a revised paragraph if both
|
|
328
|
-
* fall within the same gap.
|
|
329
|
-
*/
|
|
330
|
-
function buildGaps(anchors, unmatchedOriginal, unmatchedRevised, n, m) {
|
|
331
|
-
const gaps = [];
|
|
332
|
-
// Sentinel boundaries: before first anchor and after last anchor
|
|
333
|
-
const boundaries = [
|
|
334
|
-
{ origBound: -1, revBound: -1 },
|
|
335
|
-
...anchors.map(a => ({ origBound: a.originalIndex, revBound: a.revisedIndex })),
|
|
336
|
-
{ origBound: n, revBound: m },
|
|
337
|
-
];
|
|
338
|
-
for (let i = 0; i < boundaries.length - 1; i++) {
|
|
339
|
-
const lo = boundaries[i];
|
|
340
|
-
const hi = boundaries[i + 1];
|
|
341
|
-
const origInGap = unmatchedOriginal.filter(idx => idx > lo.origBound && idx < hi.origBound);
|
|
342
|
-
const revInGap = unmatchedRevised.filter(idx => idx > lo.revBound && idx < hi.revBound);
|
|
343
|
-
if (origInGap.length > 0 || revInGap.length > 0) {
|
|
344
|
-
gaps.push({ origIndices: origInGap, revIndices: revInGap });
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
return gaps;
|
|
348
|
-
}
|
|
349
|
-
/**
|
|
350
|
-
* Run LCS within a gap using TF-IDF cosine similarity as the equality criterion.
|
|
351
|
-
*
|
|
352
|
-
* Two groups are "equal" (matchable) if their TF-IDF cosine similarity
|
|
353
|
-
* exceeds the threshold. Standard DP LCS with backtracking.
|
|
354
|
-
*/
|
|
355
|
-
function similarityLcs(origIndices, revIndices, originalGroups, revisedGroups, tfidfVectors, threshold) {
|
|
356
|
-
const ni = origIndices.length;
|
|
357
|
-
const nj = revIndices.length;
|
|
358
|
-
// Similarity predicate for LCS equality check
|
|
359
|
-
const similar = (oi, ri) => {
|
|
360
|
-
const origGroup = originalGroups[origIndices[oi]];
|
|
361
|
-
const revGroup = revisedGroups[revIndices[ri]];
|
|
362
|
-
const vecA = tfidfVectors.get(origGroup);
|
|
363
|
-
const vecB = tfidfVectors.get(revGroup);
|
|
364
|
-
if (!vecA || !vecB)
|
|
365
|
-
return false;
|
|
366
|
-
return computeTfidfCosineSimilarity(vecA, vecB) >= threshold;
|
|
367
|
-
};
|
|
368
|
-
// Standard DP LCS
|
|
369
|
-
const dp = Array(ni + 1)
|
|
370
|
-
.fill(null)
|
|
371
|
-
.map(() => Array(nj + 1).fill(0));
|
|
372
|
-
for (let i = 1; i <= ni; i++) {
|
|
373
|
-
for (let j = 1; j <= nj; j++) {
|
|
374
|
-
if (similar(i - 1, j - 1)) {
|
|
375
|
-
dp[i][j] = dp[i - 1][j - 1] + 1;
|
|
376
|
-
}
|
|
377
|
-
else {
|
|
378
|
-
dp[i][j] = Math.max(dp[i - 1][j], dp[i][j - 1]);
|
|
379
|
-
}
|
|
380
|
-
}
|
|
381
|
-
}
|
|
382
|
-
// Backtrack
|
|
383
|
-
const matches = [];
|
|
384
|
-
let ci = ni;
|
|
385
|
-
let cj = nj;
|
|
386
|
-
while (ci > 0 && cj > 0) {
|
|
387
|
-
if (similar(ci - 1, cj - 1)) {
|
|
388
|
-
matches.unshift({
|
|
389
|
-
originalIndex: origIndices[ci - 1],
|
|
390
|
-
revisedIndex: revIndices[cj - 1],
|
|
391
|
-
});
|
|
392
|
-
ci--;
|
|
393
|
-
cj--;
|
|
394
|
-
}
|
|
395
|
-
else if (dp[ci - 1][cj] > dp[ci][cj - 1]) {
|
|
396
|
-
ci--;
|
|
397
|
-
}
|
|
398
|
-
else {
|
|
399
|
-
cj--;
|
|
400
|
-
}
|
|
401
|
-
}
|
|
402
|
-
return matches;
|
|
403
|
-
}
|
|
404
|
-
/**
|
|
405
|
-
* Compute a container key for a paragraph group based on its first atom's ancestor chain.
|
|
406
|
-
* Returns "" for body-level paragraphs, or a path like "w:tbl:0/w:tr:2/w:tc:1" for table cells.
|
|
407
|
-
*/
|
|
408
|
-
function getGroupContainerKey(group) {
|
|
409
|
-
const atom = group.atoms[0];
|
|
410
|
-
if (!atom)
|
|
411
|
-
return '';
|
|
412
|
-
const parts = [];
|
|
413
|
-
for (const el of atom.ancestorElements) {
|
|
414
|
-
if (el.tagName === 'w:tc' || el.tagName === 'w:tr' || el.tagName === 'w:tbl') {
|
|
415
|
-
let index = 0;
|
|
416
|
-
let sibling = el.previousSibling;
|
|
417
|
-
while (sibling) {
|
|
418
|
-
if (sibling.nodeType === 1 && sibling.tagName === el.tagName) {
|
|
419
|
-
index++;
|
|
420
|
-
}
|
|
421
|
-
sibling = sibling.previousSibling;
|
|
422
|
-
}
|
|
423
|
-
parts.push(`${el.tagName}:${index}`);
|
|
424
|
-
}
|
|
425
|
-
}
|
|
426
|
-
return parts.join('/');
|
|
427
|
-
}
|
|
428
|
-
/**
|
|
429
|
-
* Compute LCS on paragraph groups with order-constrained similarity fallback.
|
|
430
|
-
*
|
|
431
|
-
* Two passes:
|
|
432
|
-
* 1. LCS with exact text hash matching (fast path)
|
|
433
|
-
* 2. Order-constrained similarity matching: gap-scoped mini-LCS with TF-IDF
|
|
434
|
-
*
|
|
435
|
-
* @param originalGroups - Groups from original document
|
|
436
|
-
* @param revisedGroups - Groups from revised document
|
|
437
|
-
* @param similarityThreshold - Minimum TF-IDF cosine similarity for a match (default: 0.25)
|
|
438
|
-
* @param tfidfVectors - Precomputed TF-IDF vectors for all groups
|
|
439
|
-
*/
|
|
440
|
-
export function computeGroupLcs(originalGroups, revisedGroups, similarityThreshold = DEFAULT_PARAGRAPH_SIMILARITY_THRESHOLD, tfidfVectors) {
|
|
441
|
-
const n = originalGroups.length;
|
|
442
|
-
const m = revisedGroups.length;
|
|
443
|
-
// === Pass 1: LCS with exact hash and normalized-hash matching ===
|
|
444
|
-
const dp = Array(n + 1)
|
|
445
|
-
.fill(null)
|
|
446
|
-
.map(() => Array(m + 1).fill(0));
|
|
447
|
-
for (let i = 1; i <= n; i++) {
|
|
448
|
-
for (let j = 1; j <= m; j++) {
|
|
449
|
-
if (groupsCoarselyEqual(originalGroups[i - 1], revisedGroups[j - 1])) {
|
|
450
|
-
dp[i][j] = dp[i - 1][j - 1] + 1;
|
|
451
|
-
}
|
|
452
|
-
else {
|
|
453
|
-
dp[i][j] = Math.max(dp[i - 1][j], dp[i][j - 1]);
|
|
454
|
-
}
|
|
455
|
-
}
|
|
456
|
-
}
|
|
457
|
-
// Backtrack to find matched groups
|
|
458
|
-
const matchedGroups = [];
|
|
459
|
-
let i = n;
|
|
460
|
-
let j = m;
|
|
461
|
-
while (i > 0 && j > 0) {
|
|
462
|
-
if (groupsCoarselyEqual(originalGroups[i - 1], revisedGroups[j - 1])) {
|
|
463
|
-
matchedGroups.unshift({ originalIndex: i - 1, revisedIndex: j - 1 });
|
|
464
|
-
i--;
|
|
465
|
-
j--;
|
|
466
|
-
}
|
|
467
|
-
else if (dp[i - 1][j] > dp[i][j - 1]) {
|
|
468
|
-
i--;
|
|
469
|
-
}
|
|
470
|
-
else {
|
|
471
|
-
j--;
|
|
472
|
-
}
|
|
473
|
-
}
|
|
474
|
-
// Find initially unmatched indices
|
|
475
|
-
const matchedOriginal = new Set(matchedGroups.map((m) => m.originalIndex));
|
|
476
|
-
const matchedRevised = new Set(matchedGroups.map((m) => m.revisedIndex));
|
|
477
|
-
let unmatchedOriginal = [];
|
|
478
|
-
for (let idx = 0; idx < n; idx++) {
|
|
479
|
-
if (!matchedOriginal.has(idx)) {
|
|
480
|
-
unmatchedOriginal.push(idx);
|
|
481
|
-
}
|
|
482
|
-
}
|
|
483
|
-
let unmatchedRevised = [];
|
|
484
|
-
for (let idx = 0; idx < m; idx++) {
|
|
485
|
-
if (!matchedRevised.has(idx)) {
|
|
486
|
-
unmatchedRevised.push(idx);
|
|
487
|
-
}
|
|
488
|
-
}
|
|
489
|
-
// === Pass 2: Order-constrained similarity matching via gap-scoped LCS ===
|
|
490
|
-
//
|
|
491
|
-
// Pass 1 anchors divide both documents into "gaps" — regions between consecutive
|
|
492
|
-
// exact matches. Similarity matching is scoped to each gap: a source paragraph can
|
|
493
|
-
// only match a revised paragraph within the same gap. Within each gap, a mini-LCS
|
|
494
|
-
// using TF-IDF cosine similarity preserves document order.
|
|
495
|
-
//
|
|
496
|
-
// This prevents two classes of bugs:
|
|
497
|
-
// 1. Cross-anchor matches: Source[45] stealing Revised[20] across an anchor boundary
|
|
498
|
-
// 2. Non-monotonic matches within a gap: greedy best-match could reorder paragraphs
|
|
499
|
-
//
|
|
500
|
-
// TF-IDF (instead of Jaccard) down-weights legal boilerplate words ("holders",
|
|
501
|
-
// "Preferred Stock", "Corporation") that appear in many paragraphs, preventing
|
|
502
|
-
// false matches on shared vocabulary.
|
|
503
|
-
const similarityMatches = [];
|
|
504
|
-
// Build gaps between consecutive Pass 1 anchors
|
|
505
|
-
const gaps = buildGaps(matchedGroups, unmatchedOriginal, unmatchedRevised, n, m);
|
|
506
|
-
// Run mini-LCS within each gap using TF-IDF similarity (if vectors available),
|
|
507
|
-
// then fall back to Jaccard for any groups TF-IDF left unmatched.
|
|
508
|
-
// TF-IDF degenerates when document frequency is very low (e.g. 1-2 paragraphs):
|
|
509
|
-
// common words get IDF=0, making cosine similarity ≈ 0 even for paragraphs that
|
|
510
|
-
// share most of their content. Jaccard word overlap handles this correctly. (#78)
|
|
511
|
-
const tfidfMatchedOrig = new Set();
|
|
512
|
-
const tfidfMatchedRev = new Set();
|
|
513
|
-
if (tfidfVectors) {
|
|
514
|
-
for (const gap of gaps) {
|
|
515
|
-
if (gap.origIndices.length === 0 || gap.revIndices.length === 0)
|
|
516
|
-
continue;
|
|
517
|
-
const gapMatches = similarityLcs(gap.origIndices, gap.revIndices, originalGroups, revisedGroups, tfidfVectors, similarityThreshold);
|
|
518
|
-
for (const m of gapMatches) {
|
|
519
|
-
tfidfMatchedOrig.add(m.originalIndex);
|
|
520
|
-
tfidfMatchedRev.add(m.revisedIndex);
|
|
521
|
-
}
|
|
522
|
-
similarityMatches.push(...gapMatches);
|
|
523
|
-
}
|
|
524
|
-
}
|
|
525
|
-
// Jaccard fallback: match any groups that TF-IDF left unmatched (gap-scoped)
|
|
526
|
-
for (const gap of gaps) {
|
|
527
|
-
if (gap.origIndices.length === 0 || gap.revIndices.length === 0)
|
|
528
|
-
continue;
|
|
529
|
-
const candidates = [];
|
|
530
|
-
for (const origIdx of gap.origIndices) {
|
|
531
|
-
if (matchedOriginal.has(origIdx) || tfidfMatchedOrig.has(origIdx))
|
|
532
|
-
continue;
|
|
533
|
-
for (const revIdx of gap.revIndices) {
|
|
534
|
-
if (matchedRevised.has(revIdx) || tfidfMatchedRev.has(revIdx))
|
|
535
|
-
continue;
|
|
536
|
-
const similarity = computeGroupSimilarity(originalGroups[origIdx], revisedGroups[revIdx]);
|
|
537
|
-
if (similarity >= similarityThreshold) {
|
|
538
|
-
candidates.push({ originalIndex: origIdx, revisedIndex: revIdx, similarity });
|
|
539
|
-
}
|
|
540
|
-
}
|
|
541
|
-
}
|
|
542
|
-
candidates.sort((a, b) => b.similarity - a.similarity);
|
|
543
|
-
const assigned = new Set();
|
|
544
|
-
const assignedRev = new Set();
|
|
545
|
-
for (const c of candidates) {
|
|
546
|
-
if (assigned.has(c.originalIndex) || assignedRev.has(c.revisedIndex))
|
|
547
|
-
continue;
|
|
548
|
-
similarityMatches.push({ originalIndex: c.originalIndex, revisedIndex: c.revisedIndex });
|
|
549
|
-
assigned.add(c.originalIndex);
|
|
550
|
-
assignedRev.add(c.revisedIndex);
|
|
551
|
-
}
|
|
552
|
-
}
|
|
553
|
-
// Combine exact matches and similarity matches
|
|
554
|
-
const allMatches = [...matchedGroups, ...similarityMatches];
|
|
555
|
-
// Update matched sets
|
|
556
|
-
for (const match of similarityMatches) {
|
|
557
|
-
matchedOriginal.add(match.originalIndex);
|
|
558
|
-
matchedRevised.add(match.revisedIndex);
|
|
559
|
-
}
|
|
560
|
-
// === Pass 3 (issue #65): Container-position fallback ===
|
|
561
|
-
//
|
|
562
|
-
// After TF-IDF gap matching, some paragraphs remain unmatched because their
|
|
563
|
-
// cosine similarity is below the threshold. This happens when the only differing
|
|
564
|
-
// content is high-IDF words (e.g., company names in template fills).
|
|
565
|
-
//
|
|
566
|
-
// For unmatched paragraphs that are in the same structural container position
|
|
567
|
-
// (same table cell by table/row/cell index), force a match. This preserves
|
|
568
|
-
// paragraph alignment within table cells when the content is a template fill.
|
|
569
|
-
unmatchedOriginal = [];
|
|
570
|
-
for (let idx = 0; idx < n; idx++) {
|
|
571
|
-
if (!matchedOriginal.has(idx))
|
|
572
|
-
unmatchedOriginal.push(idx);
|
|
573
|
-
}
|
|
574
|
-
unmatchedRevised = [];
|
|
575
|
-
for (let idx = 0; idx < m; idx++) {
|
|
576
|
-
if (!matchedRevised.has(idx))
|
|
577
|
-
unmatchedRevised.push(idx);
|
|
578
|
-
}
|
|
579
|
-
if (unmatchedOriginal.length > 0 && unmatchedRevised.length > 0) {
|
|
580
|
-
// Build container keys for unmatched groups
|
|
581
|
-
const origContainerKeys = new Map();
|
|
582
|
-
for (const idx of unmatchedOriginal) {
|
|
583
|
-
const group = originalGroups[idx];
|
|
584
|
-
if (group.atoms.length > 0) {
|
|
585
|
-
origContainerKeys.set(idx, getGroupContainerKey(group));
|
|
586
|
-
}
|
|
587
|
-
}
|
|
588
|
-
const revContainerKeys = new Map();
|
|
589
|
-
for (const idx of unmatchedRevised) {
|
|
590
|
-
const group = revisedGroups[idx];
|
|
591
|
-
if (group.atoms.length > 0) {
|
|
592
|
-
revContainerKeys.set(idx, getGroupContainerKey(group));
|
|
593
|
-
}
|
|
594
|
-
}
|
|
595
|
-
// For each unmatched original in a table cell, find an unmatched revised
|
|
596
|
-
// in the same cell. Match greedily in document order.
|
|
597
|
-
const usedRevised = new Set();
|
|
598
|
-
for (const origIdx of unmatchedOriginal) {
|
|
599
|
-
const origKey = origContainerKeys.get(origIdx);
|
|
600
|
-
if (!origKey)
|
|
601
|
-
continue; // Not in a table cell
|
|
602
|
-
for (const revIdx of unmatchedRevised) {
|
|
603
|
-
if (usedRevised.has(revIdx))
|
|
604
|
-
continue;
|
|
605
|
-
const revKey = revContainerKeys.get(revIdx);
|
|
606
|
-
if (revKey === origKey) {
|
|
607
|
-
allMatches.push({ originalIndex: origIdx, revisedIndex: revIdx, containerMatch: true });
|
|
608
|
-
matchedOriginal.add(origIdx);
|
|
609
|
-
matchedRevised.add(revIdx);
|
|
610
|
-
usedRevised.add(revIdx);
|
|
611
|
-
break;
|
|
612
|
-
}
|
|
613
|
-
}
|
|
614
|
-
}
|
|
615
|
-
}
|
|
616
|
-
// Final deleted and inserted indices
|
|
617
|
-
const deletedGroupIndices = [];
|
|
618
|
-
for (let idx = 0; idx < n; idx++) {
|
|
619
|
-
if (!matchedOriginal.has(idx)) {
|
|
620
|
-
deletedGroupIndices.push(idx);
|
|
621
|
-
}
|
|
622
|
-
}
|
|
623
|
-
const insertedGroupIndices = [];
|
|
624
|
-
for (let idx = 0; idx < m; idx++) {
|
|
625
|
-
if (!matchedRevised.has(idx)) {
|
|
626
|
-
insertedGroupIndices.push(idx);
|
|
627
|
-
}
|
|
628
|
-
}
|
|
629
|
-
return { matchedGroups: allMatches, deletedGroupIndices, insertedGroupIndices };
|
|
630
|
-
}
|
|
631
|
-
/**
|
|
632
|
-
* Perform hierarchical LCS comparison.
|
|
633
|
-
*
|
|
634
|
-
* Pipeline:
|
|
635
|
-
* 1. Group atoms by paragraph
|
|
636
|
-
* 2. LCS on paragraph groups (coarse alignment) with similarity fallback
|
|
637
|
-
* 3. For matched groups: LCS on atoms within them
|
|
638
|
-
* 4. For unmatched groups: mark all atoms as deleted/inserted
|
|
639
|
-
*
|
|
640
|
-
* @param originalAtoms - Atoms from original document
|
|
641
|
-
* @param revisedAtoms - Atoms from revised document
|
|
642
|
-
* @param options - Comparison options including similarity threshold
|
|
643
|
-
* @returns Combined atom-level LCS result
|
|
644
|
-
*/
|
|
645
|
-
export function hierarchicalCompare(originalAtoms, revisedAtoms, options = {}) {
|
|
646
|
-
const { similarityThreshold = DEFAULT_PARAGRAPH_SIMILARITY_THRESHOLD } = options;
|
|
647
|
-
// Step 1: Group atoms by paragraph, splitting large paragraphs on w:br
|
|
648
|
-
const originalGroups = groupAtomsByParagraphAndBreaks(originalAtoms);
|
|
649
|
-
const revisedGroups = groupAtomsByParagraphAndBreaks(revisedAtoms);
|
|
650
|
-
// Count empty paragraph groups
|
|
651
|
-
const origEmptyGroups = originalGroups.filter(g => isEmptyParagraphGroup(g));
|
|
652
|
-
const revEmptyGroups = revisedGroups.filter(g => isEmptyParagraphGroup(g));
|
|
653
|
-
debug('hierarchicalLcs', `${originalGroups.length} original groups (${origEmptyGroups.length} empty), ${revisedGroups.length} revised groups (${revEmptyGroups.length} empty)`);
|
|
654
|
-
// Step 1b: Build TF-IDF vectors for all groups (computed once, used by Pass 2 + inline check)
|
|
655
|
-
const allGroups = [...originalGroups, ...revisedGroups];
|
|
656
|
-
const idfMap = buildIdfMap(allGroups);
|
|
657
|
-
const tfidfVectors = new Map();
|
|
658
|
-
for (const group of allGroups) {
|
|
659
|
-
tfidfVectors.set(group, buildTfidfVector(group, idfMap));
|
|
660
|
-
}
|
|
661
|
-
// Step 2: LCS on paragraph groups with order-constrained similarity fallback
|
|
662
|
-
const groupLcs = computeGroupLcs(originalGroups, revisedGroups, similarityThreshold, tfidfVectors);
|
|
663
|
-
// Count empty paragraphs in each category
|
|
664
|
-
const matchedEmptyCount = groupLcs.matchedGroups.filter(m => isEmptyParagraphGroup(originalGroups[m.originalIndex])).length;
|
|
665
|
-
const deletedEmptyCount = groupLcs.deletedGroupIndices.filter(i => isEmptyParagraphGroup(originalGroups[i])).length;
|
|
666
|
-
const insertedEmptyCount = groupLcs.insertedGroupIndices.filter(i => isEmptyParagraphGroup(revisedGroups[i])).length;
|
|
667
|
-
debug('hierarchicalLcs', `Group LCS: ${groupLcs.matchedGroups.length} matched (${matchedEmptyCount} empty), ${groupLcs.deletedGroupIndices.length} deleted (${deletedEmptyCount} empty), ${groupLcs.insertedGroupIndices.length} inserted (${insertedEmptyCount} empty)`);
|
|
668
|
-
// Step 3: Build combined atom-level result
|
|
669
|
-
const allMatches = [];
|
|
670
|
-
const deletedIndices = [];
|
|
671
|
-
const insertedIndices = [];
|
|
672
|
-
// Build atom index maps for quick lookup
|
|
673
|
-
const origAtomToIndex = new Map();
|
|
674
|
-
for (let i = 0; i < originalAtoms.length; i++) {
|
|
675
|
-
origAtomToIndex.set(originalAtoms[i], i);
|
|
676
|
-
}
|
|
677
|
-
const revAtomToIndex = new Map();
|
|
678
|
-
for (let i = 0; i < revisedAtoms.length; i++) {
|
|
679
|
-
revAtomToIndex.set(revisedAtoms[i], i);
|
|
680
|
-
}
|
|
681
|
-
// For matched groups: always run atom-level LCS within them.
|
|
682
|
-
// Group matching already determined these paragraphs correspond; the atom
|
|
683
|
-
// LCS determines which words within them changed. Skipping it based on a
|
|
684
|
-
// redundant TF-IDF similarity recheck was overly conservative and caused
|
|
685
|
-
// entire paragraphs to show as deleted+inserted instead of inline changes
|
|
686
|
-
// (see issue #78).
|
|
687
|
-
for (const match of groupLcs.matchedGroups) {
|
|
688
|
-
const origGroup = originalGroups[match.originalIndex];
|
|
689
|
-
const revGroup = revisedGroups[match.revisedIndex];
|
|
690
|
-
const withinLcs = computeAtomLcs(origGroup.atoms, revGroup.atoms);
|
|
691
|
-
for (const atomMatch of withinLcs.matches) {
|
|
692
|
-
const origAtom = origGroup.atoms[atomMatch.originalIndex];
|
|
693
|
-
const revAtom = revGroup.atoms[atomMatch.revisedIndex];
|
|
694
|
-
allMatches.push({
|
|
695
|
-
originalIndex: origAtomToIndex.get(origAtom),
|
|
696
|
-
revisedIndex: revAtomToIndex.get(revAtom),
|
|
697
|
-
});
|
|
698
|
-
}
|
|
699
|
-
for (const localIdx of withinLcs.deletedIndices) {
|
|
700
|
-
const origAtom = origGroup.atoms[localIdx];
|
|
701
|
-
deletedIndices.push(origAtomToIndex.get(origAtom));
|
|
702
|
-
}
|
|
703
|
-
for (const localIdx of withinLcs.insertedIndices) {
|
|
704
|
-
const revAtom = revGroup.atoms[localIdx];
|
|
705
|
-
insertedIndices.push(revAtomToIndex.get(revAtom));
|
|
706
|
-
}
|
|
707
|
-
}
|
|
708
|
-
// For deleted groups: mark all atoms as deleted
|
|
709
|
-
for (const groupIdx of groupLcs.deletedGroupIndices) {
|
|
710
|
-
const group = originalGroups[groupIdx];
|
|
711
|
-
for (const atom of group.atoms) {
|
|
712
|
-
deletedIndices.push(origAtomToIndex.get(atom));
|
|
713
|
-
}
|
|
714
|
-
}
|
|
715
|
-
// For inserted groups: mark all atoms as inserted
|
|
716
|
-
for (const groupIdx of groupLcs.insertedGroupIndices) {
|
|
717
|
-
const group = revisedGroups[groupIdx];
|
|
718
|
-
for (const atom of group.atoms) {
|
|
719
|
-
insertedIndices.push(revAtomToIndex.get(atom));
|
|
720
|
-
}
|
|
721
|
-
}
|
|
722
|
-
debug('hierarchicalLcs', `Hierarchical result: ${allMatches.length} matches, ${deletedIndices.length} deleted, ${insertedIndices.length} inserted`);
|
|
723
|
-
return {
|
|
724
|
-
matches: allMatches,
|
|
725
|
-
deletedIndices,
|
|
726
|
-
insertedIndices,
|
|
727
|
-
};
|
|
728
|
-
}
|
|
729
|
-
/**
|
|
730
|
-
* Mark correlation status using hierarchical comparison result.
|
|
731
|
-
*
|
|
732
|
-
* Same as regular markCorrelationStatus but uses hierarchical LCS result.
|
|
733
|
-
*/
|
|
734
|
-
export function markHierarchicalCorrelationStatus(original, revised, lcsResult) {
|
|
735
|
-
// Mark matched atoms as Equal and link them
|
|
736
|
-
for (const match of lcsResult.matches) {
|
|
737
|
-
const origAtom = original[match.originalIndex];
|
|
738
|
-
const revAtom = revised[match.revisedIndex];
|
|
739
|
-
origAtom.correlationStatus = CorrelationStatus.Equal;
|
|
740
|
-
revAtom.correlationStatus = CorrelationStatus.Equal;
|
|
741
|
-
// Link revised atom to original for format change detection
|
|
742
|
-
revAtom.comparisonUnitAtomBefore = origAtom;
|
|
743
|
-
}
|
|
744
|
-
// Mark deleted atoms
|
|
745
|
-
for (const idx of lcsResult.deletedIndices) {
|
|
746
|
-
original[idx].correlationStatus = CorrelationStatus.Deleted;
|
|
747
|
-
}
|
|
748
|
-
// Mark inserted atoms
|
|
749
|
-
for (const idx of lcsResult.insertedIndices) {
|
|
750
|
-
revised[idx].correlationStatus = CorrelationStatus.Inserted;
|
|
751
|
-
}
|
|
752
|
-
}
|
|
753
|
-
//# sourceMappingURL=hierarchicalLcs.js.map
|