@stll/folio-core 0.50.0 → 0.52.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/ai-edits/word-diff.js +134 -1
- package/dist/compare/content-alignment.js +220 -5
- package/dist/compare/content.d.ts +1 -0
- package/dist/compare/content.js +51 -20
- package/dist/compare/inline-provenance.js +18 -22
- package/dist/compare/section-boundary-properties.js +13 -0
- package/dist/docx/selectiveSave.d.ts +1 -1
- package/dist/docx/selectiveSave.js +16 -6
- package/dist/docx/structuralXmlPatch.d.ts +14 -0
- package/dist/docx/structuralXmlPatch.js +209 -0
- package/dist/generated/text_shaper.js +26 -0
- package/dist/generated/text_shaper_bg.wasm +0 -0
- package/dist/managers/editorShortcuts.d.ts +16 -3
- package/dist/managers/editorShortcuts.js +6 -4
- package/dist/prosemirror/extensions/StarterKit.d.ts +3 -0
- package/dist/prosemirror/extensions/StarterKit.js +2 -1
- package/dist/prosemirror/extensions/core/HistoryExtension.d.ts +9 -1
- package/dist/prosemirror/extensions/core/HistoryExtension.js +4 -3
- package/dist/prosemirror/extensions/features/ParagraphChangeTrackerExtension.d.ts +12 -5
- package/dist/prosemirror/extensions/features/ParagraphChangeTrackerExtension.js +91 -33
- package/dist/prosemirror/runStyleFormatting.d.ts +7 -1
- package/dist/prosemirror/runStyleFormatting.js +13 -7
- package/dist/shaping/shaper.d.ts +47 -4
- package/dist/shaping/shaper.js +46 -15
- package/dist/text-shaping.d.ts +4 -0
- package/dist/text-shaping.js +4 -0
- package/package.json +6 -1
|
@@ -836,6 +836,137 @@ const orderDeletionsFirst = (runs) => {
|
|
|
836
836
|
flush();
|
|
837
837
|
return ordered;
|
|
838
838
|
};
|
|
839
|
+
const WORD_CHARACTER = /^[\p{L}\p{N}\p{M}]$/u;
|
|
840
|
+
const LINE_BREAK = /^[\n\r\p{Zl}\p{Zp}]$/u;
|
|
841
|
+
/**
|
|
842
|
+
* How well a change edge between two code points reads, after
|
|
843
|
+
* diff-match-patch's semantic score: the string's edge (an empty side), then
|
|
844
|
+
* a line break, then the gap after a sentence or clause mark, then any space,
|
|
845
|
+
* then any other mark; inside a word is worst.
|
|
846
|
+
*/
|
|
847
|
+
const boundaryScore = (previous, next) => {
|
|
848
|
+
if (previous.length === 0 || next.length === 0) return 6;
|
|
849
|
+
if (LINE_BREAK.test(previous) || LINE_BREAK.test(next)) return 4;
|
|
850
|
+
const previousIsSpace = WHITESPACE.test(previous);
|
|
851
|
+
const nextIsSpace = WHITESPACE.test(next);
|
|
852
|
+
if (!previousIsSpace && !WORD_CHARACTER.test(previous) && nextIsSpace) return 3;
|
|
853
|
+
if (previousIsSpace || nextIsSpace) return 2;
|
|
854
|
+
return WORD_CHARACTER.test(previous) && WORD_CHARACTER.test(next) ? 0 : 1;
|
|
855
|
+
};
|
|
856
|
+
/** True when `position` falls between the two halves of a surrogate pair. */
|
|
857
|
+
const splitsSurrogatePair = (text, position) => {
|
|
858
|
+
const low = text.charCodeAt(position);
|
|
859
|
+
const high = text.charCodeAt(position - 1);
|
|
860
|
+
return low >= 56320 && low <= 57343 && high >= 55296 && high <= 56319;
|
|
861
|
+
};
|
|
862
|
+
/**
|
|
863
|
+
* Where, among the lossless positions of one change, it reads best. The
|
|
864
|
+
* change may slide by one character whenever the character it gives up
|
|
865
|
+
* equals the one it takes on, which keeps both strings intact. Only the
|
|
866
|
+
* window decides the result, so an insertion and the deletion that undoes it
|
|
867
|
+
* land on the same text; the leftmost of equally good positions wins.
|
|
868
|
+
*/
|
|
869
|
+
const bestSlideStart = (window) => {
|
|
870
|
+
const { text, start, length, minimumStart, maximumEnd } = window;
|
|
871
|
+
const scoreAt = (position) => boundaryScore(position === 0 ? window.outsideBefore : codePointBefore(text, position), position === text.length ? window.outsideAfter : codePointAt(text, position));
|
|
872
|
+
let leftmost = start;
|
|
873
|
+
while (leftmost > minimumStart && text[leftmost - 1] === text[leftmost + length - 1]) leftmost--;
|
|
874
|
+
let best = start;
|
|
875
|
+
let bestScore = -1;
|
|
876
|
+
for (let candidate = leftmost; candidate + length <= maximumEnd; candidate++) {
|
|
877
|
+
const end = candidate + length;
|
|
878
|
+
if (!splitsSurrogatePair(text, candidate) && !splitsSurrogatePair(text, end)) {
|
|
879
|
+
const startScore = scoreAt(candidate);
|
|
880
|
+
const endScore = scoreAt(end);
|
|
881
|
+
if (!(candidate !== start && (startScore === 0 || endScore === 0)) && startScore + endScore > bestScore) {
|
|
882
|
+
best = candidate;
|
|
883
|
+
bestScore = startScore + endScore;
|
|
884
|
+
}
|
|
885
|
+
}
|
|
886
|
+
if (end === text.length || text[candidate] !== text[end]) break;
|
|
887
|
+
}
|
|
888
|
+
return best;
|
|
889
|
+
};
|
|
890
|
+
/**
|
|
891
|
+
* Whether a change may end up directly beside `neighbour` once the equality
|
|
892
|
+
* between them empties: always beside nothing, an equality or its own kind,
|
|
893
|
+
* and a deletion may precede an insertion; an insertion before a deletion
|
|
894
|
+
* would break deletion-first order.
|
|
895
|
+
*/
|
|
896
|
+
const mayAbut = (first, second) => first === void 0 || second === void 0 || first === "equal" || second === "equal" || first === second || first === "del" && second === "ins";
|
|
897
|
+
/** The code point nearest `index`, walking by `step`, on the side `type` belongs to. */
|
|
898
|
+
const sideCodePoint = (segments, { from, step }, type) => {
|
|
899
|
+
for (let index = from; index >= 0 && index < segments.length; index += step) {
|
|
900
|
+
const segment = segments[index];
|
|
901
|
+
if (segment === void 0 || segment.text.length === 0) continue;
|
|
902
|
+
if (segment.type === "equal" || segment.type === type) return step === 1 ? codePointAt(segment.text, 0) : codePointBefore(segment.text, segment.text.length);
|
|
903
|
+
}
|
|
904
|
+
return "";
|
|
905
|
+
};
|
|
906
|
+
const mergeAdjacentSegments = (segments) => {
|
|
907
|
+
const merged = [];
|
|
908
|
+
for (const segment of segments) {
|
|
909
|
+
const last = merged.at(-1);
|
|
910
|
+
if (segment.text.length === 0) continue;
|
|
911
|
+
if (last?.type === segment.type) {
|
|
912
|
+
last.text += segment.text;
|
|
913
|
+
continue;
|
|
914
|
+
}
|
|
915
|
+
merged.push(segment);
|
|
916
|
+
}
|
|
917
|
+
return merged;
|
|
918
|
+
};
|
|
919
|
+
/**
|
|
920
|
+
* Slide every insertion or deletion that sits between equalities to the
|
|
921
|
+
* position that reads best.
|
|
922
|
+
*
|
|
923
|
+
* An LCS places a change arbitrarily among equally long alignments: an
|
|
924
|
+
* appended sentence can be marked as `". New sentence"` before the old
|
|
925
|
+
* sentence's full stop, not `" New sentence."` after it. The redline then
|
|
926
|
+
* marks a stop the author never touched and leaves the new sentence's own
|
|
927
|
+
* unmarked. Sliding moves only where a change's edges fall, so both strings
|
|
928
|
+
* still reconstruct.
|
|
929
|
+
*/
|
|
930
|
+
const slideChangesToReadableBoundaries = (segments) => {
|
|
931
|
+
const slid = [
|
|
932
|
+
{
|
|
933
|
+
type: "equal",
|
|
934
|
+
text: ""
|
|
935
|
+
},
|
|
936
|
+
...segments.map((segment) => ({ ...segment })),
|
|
937
|
+
{
|
|
938
|
+
type: "equal",
|
|
939
|
+
text: ""
|
|
940
|
+
}
|
|
941
|
+
];
|
|
942
|
+
for (let index = 1; index < slid.length - 1; index++) {
|
|
943
|
+
const change = slid[index];
|
|
944
|
+
const previous = slid[index - 1];
|
|
945
|
+
const next = slid[index + 1];
|
|
946
|
+
if (change === void 0 || previous === void 0 || next === void 0 || change.type === "equal" || previous.type !== "equal" || next.type !== "equal") continue;
|
|
947
|
+
const text = previous.text + change.text + next.text;
|
|
948
|
+
const start = bestSlideStart({
|
|
949
|
+
text,
|
|
950
|
+
start: previous.text.length,
|
|
951
|
+
length: change.text.length,
|
|
952
|
+
minimumStart: mayAbut(slid[index - 2]?.type, change.type) ? 0 : 1,
|
|
953
|
+
maximumEnd: mayAbut(change.type, slid[index + 2]?.type) ? text.length : text.length - 1,
|
|
954
|
+
outsideBefore: sideCodePoint(slid, {
|
|
955
|
+
from: index - 2,
|
|
956
|
+
step: -1
|
|
957
|
+
}, change.type),
|
|
958
|
+
outsideAfter: sideCodePoint(slid, {
|
|
959
|
+
from: index + 2,
|
|
960
|
+
step: 1
|
|
961
|
+
}, change.type)
|
|
962
|
+
});
|
|
963
|
+
const end = start + change.text.length;
|
|
964
|
+
previous.text = text.slice(0, start);
|
|
965
|
+
change.text = text.slice(start, end);
|
|
966
|
+
next.text = text.slice(end);
|
|
967
|
+
}
|
|
968
|
+
return mergeAdjacentSegments(slid);
|
|
969
|
+
};
|
|
839
970
|
const wholeStringReplacement = (before, after) => {
|
|
840
971
|
const segments = [];
|
|
841
972
|
if (before.length > 0) segments.push({
|
|
@@ -903,7 +1034,9 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
|
|
|
903
1034
|
const monotoneRuns = alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection);
|
|
904
1035
|
marked = whitespaceIsSignificant ? separateWhitespaceChanges(monotoneRuns) : monotoneRuns;
|
|
905
1036
|
}
|
|
906
|
-
|
|
1037
|
+
const segments = toSegments(orderDeletionsFirst(marked));
|
|
1038
|
+
const equalRunsAreExact = requestedNormalization.case !== true && requestedNormalization.whitespace !== true;
|
|
1039
|
+
return granularity === "word" && equalRunsAreExact ? slideChangesToReadableBoundaries(segments) : segments;
|
|
907
1040
|
};
|
|
908
1041
|
/**
|
|
909
1042
|
* One internal comparison/apply scope. Every diff shares the same quadratic
|
|
@@ -260,6 +260,44 @@ const pairByUniqueExactText = ({ base, revised, baseFrom, baseTo, revisedFrom, r
|
|
|
260
260
|
}
|
|
261
261
|
return pairs.toReversed();
|
|
262
262
|
};
|
|
263
|
+
/**
|
|
264
|
+
* Unique exact text anchors, then again inside every sub-gap they leave:
|
|
265
|
+
* wording repeated elsewhere in a document is still identity evidence within
|
|
266
|
+
* the one gap where it is unique, so a section's pairing does not depend on
|
|
267
|
+
* whether a neighbouring section happens to repeat it. Each nested pass is
|
|
268
|
+
* charged to the structural allowance; a refused pass leaves its sub-gap to
|
|
269
|
+
* the similarity pairing.
|
|
270
|
+
*/
|
|
271
|
+
const pairByNestedUniqueExactText = ({ workSession, ...gap }) => {
|
|
272
|
+
const anchors = [];
|
|
273
|
+
const pending = [gap];
|
|
274
|
+
for (let current = pending.pop(); current !== void 0; current = pending.pop()) {
|
|
275
|
+
const found = pairByUniqueExactText(current);
|
|
276
|
+
if (found.length === 0) continue;
|
|
277
|
+
anchors.push(...found);
|
|
278
|
+
let baseFrom = current.baseFrom;
|
|
279
|
+
let revisedFrom = current.revisedFrom;
|
|
280
|
+
for (const anchor of [...found, {
|
|
281
|
+
baseIndex: current.baseTo,
|
|
282
|
+
revisedIndex: current.revisedTo
|
|
283
|
+
}]) {
|
|
284
|
+
const size = anchor.baseIndex - baseFrom + (anchor.revisedIndex - revisedFrom);
|
|
285
|
+
if (anchor.baseIndex > baseFrom && anchor.revisedIndex > revisedFrom && size <= workSession.remainingStructuralTokenLookups) {
|
|
286
|
+
workSession.remainingStructuralTokenLookups -= size;
|
|
287
|
+
pending.push({
|
|
288
|
+
...current,
|
|
289
|
+
baseFrom,
|
|
290
|
+
baseTo: anchor.baseIndex,
|
|
291
|
+
revisedFrom,
|
|
292
|
+
revisedTo: anchor.revisedIndex
|
|
293
|
+
});
|
|
294
|
+
}
|
|
295
|
+
baseFrom = anchor.baseIndex + 1;
|
|
296
|
+
revisedFrom = anchor.revisedIndex + 1;
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
return anchors.toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
|
|
300
|
+
};
|
|
263
301
|
const pairsInAnchorGaps = ({ baseLength, revisedLength, anchors, pairGap }) => {
|
|
264
302
|
const pairs = [];
|
|
265
303
|
let baseFrom = 0;
|
|
@@ -312,9 +350,122 @@ const crossedExactTextBlocks = ({ base, revised, anchors, canPair }) => {
|
|
|
312
350
|
revised: crossedRevised
|
|
313
351
|
};
|
|
314
352
|
};
|
|
353
|
+
/**
|
|
354
|
+
* Minimum multiset Dice similarity for two blocks to pair inside a gap. At
|
|
355
|
+
* 0.5 a paragraph still pairs after gaining up to twice its own length
|
|
356
|
+
* (2n / (n + 3n)); below it, more of the pair would read as changed than
|
|
357
|
+
* kept, which a removal beside an insertion says better.
|
|
358
|
+
*/
|
|
359
|
+
const GAP_PAIR_SIMILARITY_THRESHOLD = .5;
|
|
360
|
+
/**
|
|
361
|
+
* Gap similarities are compared as integers, so a tie is exact rather than an
|
|
362
|
+
* accident of floating-point summation, and a tie-break can sit below the
|
|
363
|
+
* smallest similarity step.
|
|
364
|
+
*/
|
|
365
|
+
const GAP_PAIR_SIMILARITY_SCALE = 1e3;
|
|
366
|
+
/**
|
|
367
|
+
* Words as case-folded runs of letters, marks and digits: punctuation glued to
|
|
368
|
+
* a word ("paragraph." against "paragraph") and a capital at a sentence start
|
|
369
|
+
* would otherwise count a kept word as changed.
|
|
370
|
+
*/
|
|
371
|
+
const gapBlockTokens = (text) => {
|
|
372
|
+
const counts = /* @__PURE__ */ new Map();
|
|
373
|
+
let total = 0;
|
|
374
|
+
for (const match of text.toLowerCase().matchAll(/[\p{L}\p{M}\p{N}]+/gu)) {
|
|
375
|
+
total += 1;
|
|
376
|
+
counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
|
|
377
|
+
}
|
|
378
|
+
return {
|
|
379
|
+
counts,
|
|
380
|
+
total
|
|
381
|
+
};
|
|
382
|
+
};
|
|
383
|
+
const gapBlockSimilarity = (base, revised, workSession) => {
|
|
384
|
+
if (base.total === 0 && revised.total === 0) return {
|
|
385
|
+
status: "measured",
|
|
386
|
+
value: 1
|
|
387
|
+
};
|
|
388
|
+
if (base.total === 0 || revised.total === 0) return {
|
|
389
|
+
status: "measured",
|
|
390
|
+
value: GAP_PAIR_SIMILARITY_THRESHOLD
|
|
391
|
+
};
|
|
392
|
+
const [tokens, counterparts] = base.counts.size <= revised.counts.size ? [base.counts, revised.counts] : [revised.counts, base.counts];
|
|
393
|
+
if (tokens.size > workSession.remainingStructuralTokenLookups) return { status: "budget-exceeded" };
|
|
394
|
+
workSession.remainingStructuralTokenLookups -= tokens.size;
|
|
395
|
+
let shared = 0;
|
|
396
|
+
for (const [token, count] of tokens) shared += Math.min(count, counterparts.get(token) ?? 0);
|
|
397
|
+
return {
|
|
398
|
+
status: "measured",
|
|
399
|
+
value: 2 * shared / (base.total + revised.total)
|
|
400
|
+
};
|
|
401
|
+
};
|
|
402
|
+
/**
|
|
403
|
+
* The order-preserving pairs of one gap that maximise their summed
|
|
404
|
+
* similarity, each at least `GAP_PAIR_SIMILARITY_THRESHOLD`; offsets are into
|
|
405
|
+
* the gap's slices. Pairing by position instead fuses an inserted block with
|
|
406
|
+
* the neighbour it pushed down, and every block after it with the next one's
|
|
407
|
+
* wording. Null when the work budget refuses the gap.
|
|
408
|
+
*/
|
|
409
|
+
const pairGapBySimilarity = ({ base, revised, canPair, workSession }) => {
|
|
410
|
+
const baseCount = base.length;
|
|
411
|
+
const revisedCount = revised.length;
|
|
412
|
+
if (!claimFolioContentAlignmentCells(baseCount, revisedCount, workSession)) return null;
|
|
413
|
+
const revisedTokens = revised.map((block) => gapBlockTokens(block.block.text));
|
|
414
|
+
const similarityStep = Math.min(baseCount, revisedCount) + 1;
|
|
415
|
+
const similarity = new Float64Array(baseCount * revisedCount).fill(-1);
|
|
416
|
+
const candidateBase = /* @__PURE__ */ new Set();
|
|
417
|
+
const candidateRevised = /* @__PURE__ */ new Set();
|
|
418
|
+
for (const [baseOffset, baseBlock] of base.entries()) {
|
|
419
|
+
const baseTokens = gapBlockTokens(baseBlock.block.text);
|
|
420
|
+
for (const [revisedOffset, revisedBlock] of revised.entries()) {
|
|
421
|
+
if (!canPair(baseBlock, revisedBlock)) continue;
|
|
422
|
+
const tokens = revisedTokens[revisedOffset] ?? panic("A gap block has no token profile");
|
|
423
|
+
const measured = baseBlock.block.text === revisedBlock.block.text ? {
|
|
424
|
+
status: "measured",
|
|
425
|
+
value: 1
|
|
426
|
+
} : gapBlockSimilarity(baseTokens, tokens, workSession);
|
|
427
|
+
if (measured.status === "budget-exceeded") return null;
|
|
428
|
+
if (measured.value >= GAP_PAIR_SIMILARITY_THRESHOLD) {
|
|
429
|
+
const sameLabel = baseBlock.block.displayLabel !== void 0 && baseBlock.block.displayLabel === revisedBlock.block.displayLabel;
|
|
430
|
+
similarity[baseOffset * revisedCount + revisedOffset] = Math.round(measured.value * GAP_PAIR_SIMILARITY_SCALE) * similarityStep + (sameLabel ? 1 : 0);
|
|
431
|
+
candidateBase.add(baseOffset);
|
|
432
|
+
candidateRevised.add(revisedOffset);
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
const width = revisedCount + 1;
|
|
437
|
+
const scores = new Float64Array((baseCount + 1) * width);
|
|
438
|
+
const cellSimilarity = (baseOffset, revisedOffset) => similarity[baseOffset * revisedCount + revisedOffset] ?? -1;
|
|
439
|
+
const score = (row, column) => scores[row * width + column] ?? 0;
|
|
440
|
+
for (let row = baseCount - 1; row >= 0; row--) for (let column = revisedCount - 1; column >= 0; column--) {
|
|
441
|
+
const cell = cellSimilarity(row, column);
|
|
442
|
+
scores[row * width + column] = Math.max(score(row + 1, column), score(row, column + 1), cell < 0 ? 0 : score(row + 1, column + 1) + cell);
|
|
443
|
+
}
|
|
444
|
+
const pairs = [];
|
|
445
|
+
let row = 0;
|
|
446
|
+
let column = 0;
|
|
447
|
+
while (row < baseCount && column < revisedCount) {
|
|
448
|
+
const cell = cellSimilarity(row, column);
|
|
449
|
+
if (cell >= 0 && score(row, column) === score(row + 1, column + 1) + cell) {
|
|
450
|
+
pairs.push({
|
|
451
|
+
baseIndex: row,
|
|
452
|
+
revisedIndex: column
|
|
453
|
+
});
|
|
454
|
+
row += 1;
|
|
455
|
+
column += 1;
|
|
456
|
+
} else if (score(row, column) === score(row + 1, column)) row += 1;
|
|
457
|
+
else column += 1;
|
|
458
|
+
}
|
|
459
|
+
return {
|
|
460
|
+
pairs,
|
|
461
|
+
candidateBase,
|
|
462
|
+
candidateRevised
|
|
463
|
+
};
|
|
464
|
+
};
|
|
315
465
|
const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
|
|
316
466
|
const stableIdMismatch = options.stableIdMismatch ?? "separate";
|
|
317
467
|
const idStability = options.idStability ?? folioContentIdStability;
|
|
468
|
+
const workSession = options.workSession ?? createFolioContentAlignmentWorkSession();
|
|
318
469
|
const prepared = prepareAlignmentBlocks({
|
|
319
470
|
baseBlocks,
|
|
320
471
|
revisedBlocks,
|
|
@@ -335,14 +486,15 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
|
|
|
335
486
|
baseLength: prepared.base.length,
|
|
336
487
|
revisedLength: prepared.revised.length,
|
|
337
488
|
anchors: stableIdAnchors,
|
|
338
|
-
pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) =>
|
|
489
|
+
pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) => pairByNestedUniqueExactText({
|
|
339
490
|
base: prepared.base,
|
|
340
491
|
revised: prepared.revised,
|
|
341
492
|
baseFrom,
|
|
342
493
|
baseTo,
|
|
343
494
|
revisedFrom,
|
|
344
495
|
revisedTo,
|
|
345
|
-
canPair
|
|
496
|
+
canPair,
|
|
497
|
+
workSession
|
|
346
498
|
})
|
|
347
499
|
});
|
|
348
500
|
const exactAndStableAnchors = [...stableIdAnchors, ...exactTextAnchors].toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
|
|
@@ -368,12 +520,13 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
|
|
|
368
520
|
canPair
|
|
369
521
|
});
|
|
370
522
|
const events = [];
|
|
523
|
+
const fusesACrossing = (baseBlock, revisedBlock) => baseBlock.block.text !== revisedBlock.block.text && crossed.base.has(baseBlock.index) && crossed.revised.has(revisedBlock.index);
|
|
371
524
|
const emitPositionalGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
|
|
372
525
|
const pairedCount = Math.min(baseTo - baseFrom, revisedTo - revisedFrom);
|
|
373
526
|
for (let offset = 0; offset < pairedCount; offset++) {
|
|
374
527
|
const baseBlock = prepared.base[baseFrom + offset];
|
|
375
528
|
const revisedBlock = prepared.revised[revisedFrom + offset];
|
|
376
|
-
if (baseBlock && revisedBlock) if (
|
|
529
|
+
if (baseBlock && revisedBlock) if (fusesACrossing(baseBlock, revisedBlock) || !canPair(baseBlock, revisedBlock)) {
|
|
377
530
|
events.push({
|
|
378
531
|
type: "baseOnly",
|
|
379
532
|
block: baseBlock.block
|
|
@@ -403,10 +556,72 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
|
|
|
403
556
|
});
|
|
404
557
|
}
|
|
405
558
|
};
|
|
559
|
+
/**
|
|
560
|
+
* A gap's blocks pair by similarity, whether or not its sides are equal in
|
|
561
|
+
* length: an equal gap can hide an insertion beside a deletion, and pairing
|
|
562
|
+
* it by position reads each kept block as a rewrite of its neighbour. The
|
|
563
|
+
* similar pairs anchor the rest; blocks between two anchors pair by position
|
|
564
|
+
* only when neither side offered any candidate and the counts match, which
|
|
565
|
+
* reads a block rewritten beyond recognition as the modification it is.
|
|
566
|
+
*/
|
|
567
|
+
const emitGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
|
|
568
|
+
const pairing = pairGapBySimilarity({
|
|
569
|
+
base: prepared.base.slice(baseFrom, baseTo),
|
|
570
|
+
revised: prepared.revised.slice(revisedFrom, revisedTo),
|
|
571
|
+
canPair: (baseBlock, revisedBlock) => !fusesACrossing(baseBlock, revisedBlock) && canPair(baseBlock, revisedBlock),
|
|
572
|
+
workSession
|
|
573
|
+
});
|
|
574
|
+
if (pairing === null) {
|
|
575
|
+
emitPositionalGap(baseFrom, baseTo, revisedFrom, revisedTo);
|
|
576
|
+
return;
|
|
577
|
+
}
|
|
578
|
+
const { pairs, candidateBase, candidateRevised } = pairing;
|
|
579
|
+
let baseCursor = baseFrom;
|
|
580
|
+
let revisedCursor = revisedFrom;
|
|
581
|
+
const hasCandidate = (candidates, from, to) => {
|
|
582
|
+
for (let offset = from; offset < to; offset++) if (candidates.has(offset)) return true;
|
|
583
|
+
return false;
|
|
584
|
+
};
|
|
585
|
+
const emitUnpairedUntil = (baseEnd, revisedEnd) => {
|
|
586
|
+
if (baseEnd - baseCursor === revisedEnd - revisedCursor && !hasCandidate(candidateBase, baseCursor - baseFrom, baseEnd - baseFrom) && !hasCandidate(candidateRevised, revisedCursor - revisedFrom, revisedEnd - revisedFrom)) {
|
|
587
|
+
emitPositionalGap(baseCursor, baseEnd, revisedCursor, revisedEnd);
|
|
588
|
+
baseCursor = baseEnd;
|
|
589
|
+
revisedCursor = revisedEnd;
|
|
590
|
+
return;
|
|
591
|
+
}
|
|
592
|
+
for (; baseCursor < baseEnd; baseCursor++) {
|
|
593
|
+
const block = prepared.base[baseCursor]?.block;
|
|
594
|
+
if (block) events.push({
|
|
595
|
+
type: "baseOnly",
|
|
596
|
+
block
|
|
597
|
+
});
|
|
598
|
+
}
|
|
599
|
+
for (; revisedCursor < revisedEnd; revisedCursor++) {
|
|
600
|
+
const block = prepared.revised[revisedCursor]?.block;
|
|
601
|
+
if (block) events.push({
|
|
602
|
+
type: "revisedOnly",
|
|
603
|
+
block
|
|
604
|
+
});
|
|
605
|
+
}
|
|
606
|
+
};
|
|
607
|
+
for (const pair of pairs) {
|
|
608
|
+
emitUnpairedUntil(baseFrom + pair.baseIndex, revisedFrom + pair.revisedIndex);
|
|
609
|
+
const baseBlock = prepared.base[baseCursor]?.block;
|
|
610
|
+
const revisedBlock = prepared.revised[revisedCursor]?.block;
|
|
611
|
+
if (baseBlock && revisedBlock) events.push({
|
|
612
|
+
type: "pair",
|
|
613
|
+
baseBlock,
|
|
614
|
+
revisedBlock
|
|
615
|
+
});
|
|
616
|
+
baseCursor += 1;
|
|
617
|
+
revisedCursor += 1;
|
|
618
|
+
}
|
|
619
|
+
emitUnpairedUntil(baseTo, revisedTo);
|
|
620
|
+
};
|
|
406
621
|
let baseCursor = 0;
|
|
407
622
|
let revisedCursor = 0;
|
|
408
623
|
for (const anchor of anchors) {
|
|
409
|
-
|
|
624
|
+
emitGap(baseCursor, anchor.baseIndex, revisedCursor, anchor.revisedIndex);
|
|
410
625
|
const baseBlock = prepared.base[anchor.baseIndex]?.block;
|
|
411
626
|
const revisedBlock = prepared.revised[anchor.revisedIndex]?.block;
|
|
412
627
|
if (baseBlock && revisedBlock) events.push({
|
|
@@ -417,7 +632,7 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
|
|
|
417
632
|
baseCursor = anchor.baseIndex + 1;
|
|
418
633
|
revisedCursor = anchor.revisedIndex + 1;
|
|
419
634
|
}
|
|
420
|
-
|
|
635
|
+
emitGap(baseCursor, prepared.base.length, revisedCursor, prepared.revised.length);
|
|
421
636
|
return events;
|
|
422
637
|
};
|
|
423
638
|
const alignFolioContentBlocks = (baseBlocks, revisedBlocks, options = {}) => alignFolioContentBlocksInScope(baseBlocks, revisedBlocks, {
|
|
@@ -186,6 +186,7 @@ type ParagraphMarkPlan<Block extends FolioContentBlock> = {
|
|
|
186
186
|
revisedBlock: Block;
|
|
187
187
|
separator: string;
|
|
188
188
|
};
|
|
189
|
+
/** Plans keyed by their first step; each consumes the step after it. */
|
|
189
190
|
declare const detectFolioContentParagraphMarkPlans: <Block extends FolioContentBlock>(steps: readonly FolioContentAlignmentStep<Block>[]) => ReadonlyMap<number, ParagraphMarkPlan<Block>>;
|
|
190
191
|
type MovePair<Block extends FolioContentBlock> = {
|
|
191
192
|
baseBlock: Block;
|
package/dist/compare/content.js
CHANGED
|
@@ -431,30 +431,61 @@ const separatorBetween = (whole, head, tail) => {
|
|
|
431
431
|
const separator = whole.slice(head.length, whole.length - tail.length);
|
|
432
432
|
return separator.length === 0 || /^\s+$/u.test(separator) ? separator : null;
|
|
433
433
|
};
|
|
434
|
+
const splitPlan = ({ baseBlock, head, tail }) => {
|
|
435
|
+
const separator = separatorBetween(baseBlock.text, head.text, tail.text);
|
|
436
|
+
return separator !== null && contentBlocksShareContainer(head, tail) ? {
|
|
437
|
+
type: "split",
|
|
438
|
+
baseBlock,
|
|
439
|
+
revisedBlocks: [head, tail],
|
|
440
|
+
offset: head.text.length,
|
|
441
|
+
separator
|
|
442
|
+
} : null;
|
|
443
|
+
};
|
|
444
|
+
const mergePlan = ({ revisedBlock, head, tail }) => {
|
|
445
|
+
const separator = separatorBetween(revisedBlock.text, head.text, tail.text);
|
|
446
|
+
return separator !== null && contentBlocksShareContainer(head, tail) ? {
|
|
447
|
+
type: "merge",
|
|
448
|
+
baseBlocks: [head, tail],
|
|
449
|
+
revisedBlock,
|
|
450
|
+
separator
|
|
451
|
+
} : null;
|
|
452
|
+
};
|
|
453
|
+
/**
|
|
454
|
+
* A pair next to a one-sided block its text spells together with the pair's
|
|
455
|
+
* other side. Alignment pairs a split paragraph with whichever half reads more
|
|
456
|
+
* like it, so the unpaired half may stand before the pair or after it.
|
|
457
|
+
*/
|
|
458
|
+
const paragraphMarkPlan = (step, next) => {
|
|
459
|
+
if (step.type === "pair" && next.type === "revisedOnly") return splitPlan({
|
|
460
|
+
baseBlock: step.baseBlock,
|
|
461
|
+
head: step.revisedBlock,
|
|
462
|
+
tail: next.block
|
|
463
|
+
});
|
|
464
|
+
if (step.type === "pair" && next.type === "baseOnly") return mergePlan({
|
|
465
|
+
revisedBlock: step.revisedBlock,
|
|
466
|
+
head: step.baseBlock,
|
|
467
|
+
tail: next.block
|
|
468
|
+
});
|
|
469
|
+
if (step.type === "revisedOnly" && next.type === "pair") return splitPlan({
|
|
470
|
+
baseBlock: next.baseBlock,
|
|
471
|
+
head: step.block,
|
|
472
|
+
tail: next.revisedBlock
|
|
473
|
+
});
|
|
474
|
+
if (step.type === "baseOnly" && next.type === "pair") return mergePlan({
|
|
475
|
+
revisedBlock: next.revisedBlock,
|
|
476
|
+
head: step.block,
|
|
477
|
+
tail: next.baseBlock
|
|
478
|
+
});
|
|
479
|
+
return null;
|
|
480
|
+
};
|
|
481
|
+
/** Plans keyed by their first step; each consumes the step after it. */
|
|
434
482
|
const detectFolioContentParagraphMarkPlans = (steps) => {
|
|
435
483
|
const plans = /* @__PURE__ */ new Map();
|
|
436
484
|
for (const [index, step] of steps.entries()) {
|
|
437
485
|
const next = steps[index + 1];
|
|
438
|
-
if (
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
if (separator !== null && contentBlocksShareContainer(step.revisedBlock, next.block)) plans.set(index, {
|
|
442
|
-
type: "split",
|
|
443
|
-
baseBlock: step.baseBlock,
|
|
444
|
-
revisedBlocks: [step.revisedBlock, next.block],
|
|
445
|
-
offset: step.revisedBlock.text.length,
|
|
446
|
-
separator
|
|
447
|
-
});
|
|
448
|
-
continue;
|
|
449
|
-
}
|
|
450
|
-
if (next.type !== "baseOnly") continue;
|
|
451
|
-
const separator = separatorBetween(step.revisedBlock.text, step.baseBlock.text, next.block.text);
|
|
452
|
-
if (separator !== null && contentBlocksShareContainer(step.baseBlock, next.block)) plans.set(index, {
|
|
453
|
-
type: "merge",
|
|
454
|
-
baseBlocks: [step.baseBlock, next.block],
|
|
455
|
-
revisedBlock: step.revisedBlock,
|
|
456
|
-
separator
|
|
457
|
-
});
|
|
486
|
+
if (next === void 0 || plans.has(index - 1)) continue;
|
|
487
|
+
const plan = paragraphMarkPlan(step, next);
|
|
488
|
+
if (plan !== null) plans.set(index, plan);
|
|
458
489
|
}
|
|
459
490
|
return plans;
|
|
460
491
|
};
|
|
@@ -3,7 +3,7 @@ import { expectHyperlinkMarkAttrs, expectRunPropertyChangeMarkAttrs, expectTable
|
|
|
3
3
|
import { getDocumentStyleResolver } from "../prosemirror/plugins/documentStyles.js";
|
|
4
4
|
import { applyMarksToRunFormattingRepresentation, expandRunFormattingCarrier, runFormattingCarrierReviewText } from "../prosemirror/runFormattingInlineCarriers.js";
|
|
5
5
|
import { readAuthoredRunFormatting, reconcileRunFormattingMarks } from "../prosemirror/runFormattingReconciliation.js";
|
|
6
|
-
import {
|
|
6
|
+
import { nearestParagraphRunStyleContext } from "../prosemirror/runStyleFormatting.js";
|
|
7
7
|
import { isTableCellRetainedInReviewView } from "../prosemirror/tableCellRevisionVisibility.js";
|
|
8
8
|
import { canonicalJson } from "../utils/canonicalJson.js";
|
|
9
9
|
//#region src/compare/inline-provenance.ts
|
|
@@ -19,10 +19,11 @@ const hyperlinkIdentity = (node) => {
|
|
|
19
19
|
});
|
|
20
20
|
};
|
|
21
21
|
const isInserted = (node) => node.marks.some(({ type }) => type.name === "insertion");
|
|
22
|
-
const carrierRepresentations = (carrier) => carrier.representations.map((representation) => ({
|
|
22
|
+
const carrierRepresentations = ({ carrier, paragraph }) => carrier.representations.map((representation) => ({
|
|
23
23
|
...representation,
|
|
24
24
|
from: representation.position,
|
|
25
|
-
to: representation.position + representation.node.nodeSize
|
|
25
|
+
to: representation.position + representation.node.nodeSize,
|
|
26
|
+
paragraph
|
|
26
27
|
}));
|
|
27
28
|
const targetBlockIdLookup = (anchors) => {
|
|
28
29
|
const ordered = Object.values(anchors).toSorted((left, right) => left.from - right.from || left.to - right.to);
|
|
@@ -53,7 +54,13 @@ const collectCarriers = ({ doc, targetSnapshot }) => {
|
|
|
53
54
|
const carriers = [];
|
|
54
55
|
const targetBlockIdAt = targetSnapshot ? targetBlockIdLookup(targetSnapshot.anchors) : void 0;
|
|
55
56
|
let unanchoredTargetCarrier = false;
|
|
57
|
+
const openParagraphs = [];
|
|
56
58
|
doc.descendants((node, position) => {
|
|
59
|
+
while ((openParagraphs.at(-1)?.end ?? Number.POSITIVE_INFINITY) <= position) openParagraphs.pop();
|
|
60
|
+
if (node.type.name === "paragraph") openParagraphs.push({
|
|
61
|
+
node,
|
|
62
|
+
end: position + node.nodeSize
|
|
63
|
+
});
|
|
57
64
|
if ((node.type.name === "tableCell" || node.type.name === "tableHeader") && !isTableCellRetainedInReviewView(expectTableCellAttrs(node).cellMarker?.kind, "final")) return false;
|
|
58
65
|
if (!node.isInline) return true;
|
|
59
66
|
const carrier = expandRunFormattingCarrier(node, position);
|
|
@@ -69,6 +76,7 @@ const collectCarriers = ({ doc, targetSnapshot }) => {
|
|
|
69
76
|
}
|
|
70
77
|
carriers.push({
|
|
71
78
|
carrier,
|
|
79
|
+
paragraph: openParagraphs.at(-1)?.node ?? null,
|
|
72
80
|
text: runFormattingCarrierReviewText(carrier),
|
|
73
81
|
targetBlockId
|
|
74
82
|
});
|
|
@@ -106,8 +114,8 @@ const matchCarrierStreams = ({ live, target }) => {
|
|
|
106
114
|
const targetIsText = targetCarrier.carrier.disposition === "text-run";
|
|
107
115
|
if (!liveIsText || !targetIsText) {
|
|
108
116
|
if (liveOffset !== 0 || targetOffset !== 0 || !sameCarrierShape(liveCarrier, targetCarrier)) return null;
|
|
109
|
-
const liveRepresentations = carrierRepresentations(liveCarrier
|
|
110
|
-
const targetRepresentations = carrierRepresentations(targetCarrier
|
|
117
|
+
const liveRepresentations = carrierRepresentations(liveCarrier);
|
|
118
|
+
const targetRepresentations = carrierRepresentations(targetCarrier);
|
|
111
119
|
for (const [index, liveRepresentation] of liveRepresentations.entries()) {
|
|
112
120
|
const targetRepresentation = targetRepresentations.at(index);
|
|
113
121
|
if (!targetRepresentation) return null;
|
|
@@ -123,8 +131,8 @@ const matchCarrierStreams = ({ live, target }) => {
|
|
|
123
131
|
}
|
|
124
132
|
const sharedLength = Math.min(liveText.length, targetText.length);
|
|
125
133
|
if (sharedLength === 0 || liveText.slice(0, sharedLength) !== targetText.slice(0, sharedLength)) return null;
|
|
126
|
-
const liveRepresentation = carrierRepresentations(liveCarrier
|
|
127
|
-
const targetRepresentation = carrierRepresentations(targetCarrier
|
|
134
|
+
const liveRepresentation = carrierRepresentations(liveCarrier).at(0);
|
|
135
|
+
const targetRepresentation = carrierRepresentations(targetCarrier).at(0);
|
|
128
136
|
if (!liveRepresentation || !targetRepresentation) return null;
|
|
129
137
|
matched.push({
|
|
130
138
|
live: {
|
|
@@ -152,12 +160,8 @@ const matchCarrierStreams = ({ live, target }) => {
|
|
|
152
160
|
}
|
|
153
161
|
return matched;
|
|
154
162
|
};
|
|
155
|
-
const authoredFormattingAt = ({
|
|
156
|
-
context:
|
|
157
|
-
doc,
|
|
158
|
-
pos: representation.from,
|
|
159
|
-
styleResolver
|
|
160
|
-
}),
|
|
163
|
+
const authoredFormattingAt = ({ representation, styleResolver }) => readAuthoredRunFormatting({
|
|
164
|
+
context: nearestParagraphRunStyleContext(representation.paragraph, styleResolver),
|
|
161
165
|
marks: representation.node.marks,
|
|
162
166
|
styleResolver
|
|
163
167
|
});
|
|
@@ -190,12 +194,10 @@ const matchInlineProvenance = ({ state, targetSnapshot, revisionStamp, originalR
|
|
|
190
194
|
const planned = [];
|
|
191
195
|
for (const segment of matched) {
|
|
192
196
|
const liveFormatting = authoredFormattingAt({
|
|
193
|
-
doc: state.doc,
|
|
194
197
|
representation: segment.live,
|
|
195
198
|
styleResolver: liveStyleResolver
|
|
196
199
|
});
|
|
197
200
|
const targetFormatting = authoredFormattingAt({
|
|
198
|
-
doc: targetDocument,
|
|
199
201
|
representation: segment.target,
|
|
200
202
|
styleResolver: targetStyleResolver
|
|
201
203
|
});
|
|
@@ -220,11 +222,7 @@ const matchInlineProvenance = ({ state, targetSnapshot, revisionStamp, originalR
|
|
|
220
222
|
let nextRevisionId = revisionStamp.idSeed;
|
|
221
223
|
const changedTargetBlockIds = /* @__PURE__ */ new Set();
|
|
222
224
|
for (const change of planned.toReversed()) {
|
|
223
|
-
const context =
|
|
224
|
-
doc: state.doc,
|
|
225
|
-
pos: change.live.from,
|
|
226
|
-
styleResolver: liveStyleResolver
|
|
227
|
-
});
|
|
225
|
+
const context = nearestParagraphRunStyleContext(change.live.paragraph, liveStyleResolver);
|
|
228
226
|
const formattingMarks = reconcileRunFormattingMarks({
|
|
229
227
|
authoredFormatting: change.formatting,
|
|
230
228
|
context,
|
|
@@ -321,11 +319,9 @@ const sameAuthoredInlineProvenance = (baseSnapshot, targetSnapshot) => {
|
|
|
321
319
|
const baseStyleResolver = styleResolverOf(baseSnapshot);
|
|
322
320
|
const targetStyleResolver = styleResolverOf(targetSnapshot);
|
|
323
321
|
return matched.every(({ live, target }) => hyperlinkIdentity(live.node) === hyperlinkIdentity(target.node) && canonicalJson(authoredFormattingAt({
|
|
324
|
-
doc: baseDocument,
|
|
325
322
|
representation: live,
|
|
326
323
|
styleResolver: baseStyleResolver
|
|
327
324
|
})) === canonicalJson(authoredFormattingAt({
|
|
328
|
-
doc: targetDocument,
|
|
329
325
|
representation: target,
|
|
330
326
|
styleResolver: targetStyleResolver
|
|
331
327
|
})));
|
|
@@ -108,9 +108,22 @@ const boundaryParagraphEntriesOf = (document) => {
|
|
|
108
108
|
if (!children) return null;
|
|
109
109
|
return children.flatMap((child) => child.type === "paragraph" ? [child.paragraph] : []);
|
|
110
110
|
};
|
|
111
|
+
const hasSectionEndpoint = (document) => {
|
|
112
|
+
let found = false;
|
|
113
|
+
document.forEach((node) => {
|
|
114
|
+
if (node.type.name === "paragraph" && expectParagraphAttrs(node)._sectionProperties !== void 0) found = true;
|
|
115
|
+
});
|
|
116
|
+
return found;
|
|
117
|
+
};
|
|
111
118
|
/** Stage section deltas after document operations, against the accepted projection. */
|
|
112
119
|
const stageSectionBoundaryProperties = ({ state, target, originalRevisionIdSeed, revisionStamp, author, maxRanges, mapTargetProperties }) => {
|
|
113
120
|
if (!Number.isSafeInteger(maxRanges) || maxRanges < 0) return { status: "budget-exceeded" };
|
|
121
|
+
if (!hasSectionEndpoint(target)) return {
|
|
122
|
+
status: "matched",
|
|
123
|
+
transaction: state.tr,
|
|
124
|
+
nextRevisionId: revisionStamp.idSeed,
|
|
125
|
+
rangeCount: 0
|
|
126
|
+
};
|
|
114
127
|
const reviewedState = resolveAllChangesInHeadlessStateWithMapping(state, "accept");
|
|
115
128
|
const reviewed = boundaryParagraphEntriesOf(reviewedState.state.doc);
|
|
116
129
|
const targets = boundaryParagraphEntriesOf(target);
|
|
@@ -3,7 +3,7 @@ import { document_d_exports } from "../types/document.js";
|
|
|
3
3
|
type SelectiveSaveOptions = {
|
|
4
4
|
/** Changed paragraph IDs to selectively patch */
|
|
5
5
|
changedParaIds: Set<string>;
|
|
6
|
-
/** Whether
|
|
6
|
+
/** Whether paragraph membership, order, or block structure changed. */
|
|
7
7
|
structuralChange: boolean;
|
|
8
8
|
/** Whether any changes affected paragraphs without paraId */
|
|
9
9
|
hasUntrackedChanges: boolean;
|