@stll/folio-core 0.50.0 → 0.52.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -836,6 +836,137 @@ const orderDeletionsFirst = (runs) => {
836
836
  flush();
837
837
  return ordered;
838
838
  };
839
+ const WORD_CHARACTER = /^[\p{L}\p{N}\p{M}]$/u;
840
+ const LINE_BREAK = /^[\n\r\p{Zl}\p{Zp}]$/u;
841
+ /**
842
+ * How well a change edge between two code points reads, after
843
+ * diff-match-patch's semantic score: the string's edge (an empty side), then
844
+ * a line break, then the gap after a sentence or clause mark, then any space,
845
+ * then any other mark; inside a word is worst.
846
+ */
847
+ const boundaryScore = (previous, next) => {
848
+ if (previous.length === 0 || next.length === 0) return 6;
849
+ if (LINE_BREAK.test(previous) || LINE_BREAK.test(next)) return 4;
850
+ const previousIsSpace = WHITESPACE.test(previous);
851
+ const nextIsSpace = WHITESPACE.test(next);
852
+ if (!previousIsSpace && !WORD_CHARACTER.test(previous) && nextIsSpace) return 3;
853
+ if (previousIsSpace || nextIsSpace) return 2;
854
+ return WORD_CHARACTER.test(previous) && WORD_CHARACTER.test(next) ? 0 : 1;
855
+ };
856
+ /** True when `position` falls between the two halves of a surrogate pair. */
857
+ const splitsSurrogatePair = (text, position) => {
858
+ const low = text.charCodeAt(position);
859
+ const high = text.charCodeAt(position - 1);
860
+ return low >= 56320 && low <= 57343 && high >= 55296 && high <= 56319;
861
+ };
862
+ /**
863
+ * Where, among the lossless positions of one change, it reads best. The
864
+ * change may slide by one character whenever the character it gives up
865
+ * equals the one it takes on, which keeps both strings intact. Only the
866
+ * window decides the result, so an insertion and the deletion that undoes it
867
+ * land on the same text; the leftmost of equally good positions wins.
868
+ */
869
+ const bestSlideStart = (window) => {
870
+ const { text, start, length, minimumStart, maximumEnd } = window;
871
+ const scoreAt = (position) => boundaryScore(position === 0 ? window.outsideBefore : codePointBefore(text, position), position === text.length ? window.outsideAfter : codePointAt(text, position));
872
+ let leftmost = start;
873
+ while (leftmost > minimumStart && text[leftmost - 1] === text[leftmost + length - 1]) leftmost--;
874
+ let best = start;
875
+ let bestScore = -1;
876
+ for (let candidate = leftmost; candidate + length <= maximumEnd; candidate++) {
877
+ const end = candidate + length;
878
+ if (!splitsSurrogatePair(text, candidate) && !splitsSurrogatePair(text, end)) {
879
+ const startScore = scoreAt(candidate);
880
+ const endScore = scoreAt(end);
881
+ if (!(candidate !== start && (startScore === 0 || endScore === 0)) && startScore + endScore > bestScore) {
882
+ best = candidate;
883
+ bestScore = startScore + endScore;
884
+ }
885
+ }
886
+ if (end === text.length || text[candidate] !== text[end]) break;
887
+ }
888
+ return best;
889
+ };
890
+ /**
891
+ * Whether a change may end up directly beside `neighbour` once the equality
892
+ * between them empties: always beside nothing, an equality or its own kind,
893
+ * and a deletion may precede an insertion; an insertion before a deletion
894
+ * would break deletion-first order.
895
+ */
896
+ const mayAbut = (first, second) => first === void 0 || second === void 0 || first === "equal" || second === "equal" || first === second || first === "del" && second === "ins";
897
+ /** The code point nearest `index`, walking by `step`, on the side `type` belongs to. */
898
+ const sideCodePoint = (segments, { from, step }, type) => {
899
+ for (let index = from; index >= 0 && index < segments.length; index += step) {
900
+ const segment = segments[index];
901
+ if (segment === void 0 || segment.text.length === 0) continue;
902
+ if (segment.type === "equal" || segment.type === type) return step === 1 ? codePointAt(segment.text, 0) : codePointBefore(segment.text, segment.text.length);
903
+ }
904
+ return "";
905
+ };
906
+ const mergeAdjacentSegments = (segments) => {
907
+ const merged = [];
908
+ for (const segment of segments) {
909
+ const last = merged.at(-1);
910
+ if (segment.text.length === 0) continue;
911
+ if (last?.type === segment.type) {
912
+ last.text += segment.text;
913
+ continue;
914
+ }
915
+ merged.push(segment);
916
+ }
917
+ return merged;
918
+ };
919
+ /**
920
+ * Slide every insertion or deletion that sits between equalities to the
921
+ * position that reads best.
922
+ *
923
+ * An LCS places a change arbitrarily among equally long alignments: an
924
+ * appended sentence can be marked as `". New sentence"` before the old
925
+ * sentence's full stop, not `" New sentence."` after it. The redline then
926
+ * marks a stop the author never touched and leaves the new sentence's own
927
+ * unmarked. Sliding moves only where a change's edges fall, so both strings
928
+ * still reconstruct.
929
+ */
930
+ const slideChangesToReadableBoundaries = (segments) => {
931
+ const slid = [
932
+ {
933
+ type: "equal",
934
+ text: ""
935
+ },
936
+ ...segments.map((segment) => ({ ...segment })),
937
+ {
938
+ type: "equal",
939
+ text: ""
940
+ }
941
+ ];
942
+ for (let index = 1; index < slid.length - 1; index++) {
943
+ const change = slid[index];
944
+ const previous = slid[index - 1];
945
+ const next = slid[index + 1];
946
+ if (change === void 0 || previous === void 0 || next === void 0 || change.type === "equal" || previous.type !== "equal" || next.type !== "equal") continue;
947
+ const text = previous.text + change.text + next.text;
948
+ const start = bestSlideStart({
949
+ text,
950
+ start: previous.text.length,
951
+ length: change.text.length,
952
+ minimumStart: mayAbut(slid[index - 2]?.type, change.type) ? 0 : 1,
953
+ maximumEnd: mayAbut(change.type, slid[index + 2]?.type) ? text.length : text.length - 1,
954
+ outsideBefore: sideCodePoint(slid, {
955
+ from: index - 2,
956
+ step: -1
957
+ }, change.type),
958
+ outsideAfter: sideCodePoint(slid, {
959
+ from: index + 2,
960
+ step: 1
961
+ }, change.type)
962
+ });
963
+ const end = start + change.text.length;
964
+ previous.text = text.slice(0, start);
965
+ change.text = text.slice(start, end);
966
+ next.text = text.slice(end);
967
+ }
968
+ return mergeAdjacentSegments(slid);
969
+ };
839
970
  const wholeStringReplacement = (before, after) => {
840
971
  const segments = [];
841
972
  if (before.length > 0) segments.push({
@@ -903,7 +1034,9 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
903
1034
  const monotoneRuns = alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection);
904
1035
  marked = whitespaceIsSignificant ? separateWhitespaceChanges(monotoneRuns) : monotoneRuns;
905
1036
  }
906
- return toSegments(orderDeletionsFirst(marked));
1037
+ const segments = toSegments(orderDeletionsFirst(marked));
1038
+ const equalRunsAreExact = requestedNormalization.case !== true && requestedNormalization.whitespace !== true;
1039
+ return granularity === "word" && equalRunsAreExact ? slideChangesToReadableBoundaries(segments) : segments;
907
1040
  };
908
1041
  /**
909
1042
  * One internal comparison/apply scope. Every diff shares the same quadratic
@@ -260,6 +260,44 @@ const pairByUniqueExactText = ({ base, revised, baseFrom, baseTo, revisedFrom, r
260
260
  }
261
261
  return pairs.toReversed();
262
262
  };
263
+ /**
264
+ * Unique exact text anchors, then again inside every sub-gap they leave:
265
+ * wording repeated elsewhere in a document is still identity evidence within
266
+ * the one gap where it is unique, so a section's pairing does not depend on
267
+ * whether a neighbouring section happens to repeat it. Each nested pass is
268
+ * charged to the structural allowance; a refused pass leaves its sub-gap to
269
+ * the similarity pairing.
270
+ */
271
+ const pairByNestedUniqueExactText = ({ workSession, ...gap }) => {
272
+ const anchors = [];
273
+ const pending = [gap];
274
+ for (let current = pending.pop(); current !== void 0; current = pending.pop()) {
275
+ const found = pairByUniqueExactText(current);
276
+ if (found.length === 0) continue;
277
+ anchors.push(...found);
278
+ let baseFrom = current.baseFrom;
279
+ let revisedFrom = current.revisedFrom;
280
+ for (const anchor of [...found, {
281
+ baseIndex: current.baseTo,
282
+ revisedIndex: current.revisedTo
283
+ }]) {
284
+ const size = anchor.baseIndex - baseFrom + (anchor.revisedIndex - revisedFrom);
285
+ if (anchor.baseIndex > baseFrom && anchor.revisedIndex > revisedFrom && size <= workSession.remainingStructuralTokenLookups) {
286
+ workSession.remainingStructuralTokenLookups -= size;
287
+ pending.push({
288
+ ...current,
289
+ baseFrom,
290
+ baseTo: anchor.baseIndex,
291
+ revisedFrom,
292
+ revisedTo: anchor.revisedIndex
293
+ });
294
+ }
295
+ baseFrom = anchor.baseIndex + 1;
296
+ revisedFrom = anchor.revisedIndex + 1;
297
+ }
298
+ }
299
+ return anchors.toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
300
+ };
263
301
  const pairsInAnchorGaps = ({ baseLength, revisedLength, anchors, pairGap }) => {
264
302
  const pairs = [];
265
303
  let baseFrom = 0;
@@ -312,9 +350,122 @@ const crossedExactTextBlocks = ({ base, revised, anchors, canPair }) => {
312
350
  revised: crossedRevised
313
351
  };
314
352
  };
353
+ /**
354
+ * Minimum multiset Dice similarity for two blocks to pair inside a gap. At
355
+ * 0.5 a paragraph still pairs after gaining up to twice its own length
356
+ * (2n / (n + 3n)); below it, more of the pair would read as changed than
357
+ * kept, which a removal beside an insertion says better.
358
+ */
359
+ const GAP_PAIR_SIMILARITY_THRESHOLD = .5;
360
+ /**
361
+ * Gap similarities are compared as integers, so a tie is exact rather than an
362
+ * accident of floating-point summation, and a tie-break can sit below the
363
+ * smallest similarity step.
364
+ */
365
+ const GAP_PAIR_SIMILARITY_SCALE = 1e3;
366
+ /**
367
+ * Words as case-folded runs of letters, marks and digits: punctuation glued to
368
+ * a word ("paragraph." against "paragraph") and a capital at a sentence start
369
+ * would otherwise count a kept word as changed.
370
+ */
371
+ const gapBlockTokens = (text) => {
372
+ const counts = /* @__PURE__ */ new Map();
373
+ let total = 0;
374
+ for (const match of text.toLowerCase().matchAll(/[\p{L}\p{M}\p{N}]+/gu)) {
375
+ total += 1;
376
+ counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
377
+ }
378
+ return {
379
+ counts,
380
+ total
381
+ };
382
+ };
383
+ const gapBlockSimilarity = (base, revised, workSession) => {
384
+ if (base.total === 0 && revised.total === 0) return {
385
+ status: "measured",
386
+ value: 1
387
+ };
388
+ if (base.total === 0 || revised.total === 0) return {
389
+ status: "measured",
390
+ value: GAP_PAIR_SIMILARITY_THRESHOLD
391
+ };
392
+ const [tokens, counterparts] = base.counts.size <= revised.counts.size ? [base.counts, revised.counts] : [revised.counts, base.counts];
393
+ if (tokens.size > workSession.remainingStructuralTokenLookups) return { status: "budget-exceeded" };
394
+ workSession.remainingStructuralTokenLookups -= tokens.size;
395
+ let shared = 0;
396
+ for (const [token, count] of tokens) shared += Math.min(count, counterparts.get(token) ?? 0);
397
+ return {
398
+ status: "measured",
399
+ value: 2 * shared / (base.total + revised.total)
400
+ };
401
+ };
402
+ /**
403
+ * The order-preserving pairs of one gap that maximise their summed
404
+ * similarity, each at least `GAP_PAIR_SIMILARITY_THRESHOLD`; offsets are into
405
+ * the gap's slices. Pairing by position instead fuses an inserted block with
406
+ * the neighbour it pushed down, and every block after it with the next one's
407
+ * wording. Null when the work budget refuses the gap.
408
+ */
409
+ const pairGapBySimilarity = ({ base, revised, canPair, workSession }) => {
410
+ const baseCount = base.length;
411
+ const revisedCount = revised.length;
412
+ if (!claimFolioContentAlignmentCells(baseCount, revisedCount, workSession)) return null;
413
+ const revisedTokens = revised.map((block) => gapBlockTokens(block.block.text));
414
+ const similarityStep = Math.min(baseCount, revisedCount) + 1;
415
+ const similarity = new Float64Array(baseCount * revisedCount).fill(-1);
416
+ const candidateBase = /* @__PURE__ */ new Set();
417
+ const candidateRevised = /* @__PURE__ */ new Set();
418
+ for (const [baseOffset, baseBlock] of base.entries()) {
419
+ const baseTokens = gapBlockTokens(baseBlock.block.text);
420
+ for (const [revisedOffset, revisedBlock] of revised.entries()) {
421
+ if (!canPair(baseBlock, revisedBlock)) continue;
422
+ const tokens = revisedTokens[revisedOffset] ?? panic("A gap block has no token profile");
423
+ const measured = baseBlock.block.text === revisedBlock.block.text ? {
424
+ status: "measured",
425
+ value: 1
426
+ } : gapBlockSimilarity(baseTokens, tokens, workSession);
427
+ if (measured.status === "budget-exceeded") return null;
428
+ if (measured.value >= GAP_PAIR_SIMILARITY_THRESHOLD) {
429
+ const sameLabel = baseBlock.block.displayLabel !== void 0 && baseBlock.block.displayLabel === revisedBlock.block.displayLabel;
430
+ similarity[baseOffset * revisedCount + revisedOffset] = Math.round(measured.value * GAP_PAIR_SIMILARITY_SCALE) * similarityStep + (sameLabel ? 1 : 0);
431
+ candidateBase.add(baseOffset);
432
+ candidateRevised.add(revisedOffset);
433
+ }
434
+ }
435
+ }
436
+ const width = revisedCount + 1;
437
+ const scores = new Float64Array((baseCount + 1) * width);
438
+ const cellSimilarity = (baseOffset, revisedOffset) => similarity[baseOffset * revisedCount + revisedOffset] ?? -1;
439
+ const score = (row, column) => scores[row * width + column] ?? 0;
440
+ for (let row = baseCount - 1; row >= 0; row--) for (let column = revisedCount - 1; column >= 0; column--) {
441
+ const cell = cellSimilarity(row, column);
442
+ scores[row * width + column] = Math.max(score(row + 1, column), score(row, column + 1), cell < 0 ? 0 : score(row + 1, column + 1) + cell);
443
+ }
444
+ const pairs = [];
445
+ let row = 0;
446
+ let column = 0;
447
+ while (row < baseCount && column < revisedCount) {
448
+ const cell = cellSimilarity(row, column);
449
+ if (cell >= 0 && score(row, column) === score(row + 1, column + 1) + cell) {
450
+ pairs.push({
451
+ baseIndex: row,
452
+ revisedIndex: column
453
+ });
454
+ row += 1;
455
+ column += 1;
456
+ } else if (score(row, column) === score(row + 1, column)) row += 1;
457
+ else column += 1;
458
+ }
459
+ return {
460
+ pairs,
461
+ candidateBase,
462
+ candidateRevised
463
+ };
464
+ };
315
465
  const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
316
466
  const stableIdMismatch = options.stableIdMismatch ?? "separate";
317
467
  const idStability = options.idStability ?? folioContentIdStability;
468
+ const workSession = options.workSession ?? createFolioContentAlignmentWorkSession();
318
469
  const prepared = prepareAlignmentBlocks({
319
470
  baseBlocks,
320
471
  revisedBlocks,
@@ -335,14 +486,15 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
335
486
  baseLength: prepared.base.length,
336
487
  revisedLength: prepared.revised.length,
337
488
  anchors: stableIdAnchors,
338
- pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) => pairByUniqueExactText({
489
+ pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) => pairByNestedUniqueExactText({
339
490
  base: prepared.base,
340
491
  revised: prepared.revised,
341
492
  baseFrom,
342
493
  baseTo,
343
494
  revisedFrom,
344
495
  revisedTo,
345
- canPair
496
+ canPair,
497
+ workSession
346
498
  })
347
499
  });
348
500
  const exactAndStableAnchors = [...stableIdAnchors, ...exactTextAnchors].toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
@@ -368,12 +520,13 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
368
520
  canPair
369
521
  });
370
522
  const events = [];
523
+ const fusesACrossing = (baseBlock, revisedBlock) => baseBlock.block.text !== revisedBlock.block.text && crossed.base.has(baseBlock.index) && crossed.revised.has(revisedBlock.index);
371
524
  const emitPositionalGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
372
525
  const pairedCount = Math.min(baseTo - baseFrom, revisedTo - revisedFrom);
373
526
  for (let offset = 0; offset < pairedCount; offset++) {
374
527
  const baseBlock = prepared.base[baseFrom + offset];
375
528
  const revisedBlock = prepared.revised[revisedFrom + offset];
376
- if (baseBlock && revisedBlock) if (baseBlock.block.text !== revisedBlock.block.text && crossed.base.has(baseBlock.index) && crossed.revised.has(revisedBlock.index) || !canPair(baseBlock, revisedBlock)) {
529
+ if (baseBlock && revisedBlock) if (fusesACrossing(baseBlock, revisedBlock) || !canPair(baseBlock, revisedBlock)) {
377
530
  events.push({
378
531
  type: "baseOnly",
379
532
  block: baseBlock.block
@@ -403,10 +556,72 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
403
556
  });
404
557
  }
405
558
  };
559
+ /**
560
+ * A gap's blocks pair by similarity, whether or not its sides are equal in
561
+ * length: an equal gap can hide an insertion beside a deletion, and pairing
562
+ * it by position reads each kept block as a rewrite of its neighbour. The
563
+ * similar pairs anchor the rest; blocks between two anchors pair by position
564
+ * only when neither side offered any candidate and the counts match, which
565
+ * reads a block rewritten beyond recognition as the modification it is.
566
+ */
567
+ const emitGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
568
+ const pairing = pairGapBySimilarity({
569
+ base: prepared.base.slice(baseFrom, baseTo),
570
+ revised: prepared.revised.slice(revisedFrom, revisedTo),
571
+ canPair: (baseBlock, revisedBlock) => !fusesACrossing(baseBlock, revisedBlock) && canPair(baseBlock, revisedBlock),
572
+ workSession
573
+ });
574
+ if (pairing === null) {
575
+ emitPositionalGap(baseFrom, baseTo, revisedFrom, revisedTo);
576
+ return;
577
+ }
578
+ const { pairs, candidateBase, candidateRevised } = pairing;
579
+ let baseCursor = baseFrom;
580
+ let revisedCursor = revisedFrom;
581
+ const hasCandidate = (candidates, from, to) => {
582
+ for (let offset = from; offset < to; offset++) if (candidates.has(offset)) return true;
583
+ return false;
584
+ };
585
+ const emitUnpairedUntil = (baseEnd, revisedEnd) => {
586
+ if (baseEnd - baseCursor === revisedEnd - revisedCursor && !hasCandidate(candidateBase, baseCursor - baseFrom, baseEnd - baseFrom) && !hasCandidate(candidateRevised, revisedCursor - revisedFrom, revisedEnd - revisedFrom)) {
587
+ emitPositionalGap(baseCursor, baseEnd, revisedCursor, revisedEnd);
588
+ baseCursor = baseEnd;
589
+ revisedCursor = revisedEnd;
590
+ return;
591
+ }
592
+ for (; baseCursor < baseEnd; baseCursor++) {
593
+ const block = prepared.base[baseCursor]?.block;
594
+ if (block) events.push({
595
+ type: "baseOnly",
596
+ block
597
+ });
598
+ }
599
+ for (; revisedCursor < revisedEnd; revisedCursor++) {
600
+ const block = prepared.revised[revisedCursor]?.block;
601
+ if (block) events.push({
602
+ type: "revisedOnly",
603
+ block
604
+ });
605
+ }
606
+ };
607
+ for (const pair of pairs) {
608
+ emitUnpairedUntil(baseFrom + pair.baseIndex, revisedFrom + pair.revisedIndex);
609
+ const baseBlock = prepared.base[baseCursor]?.block;
610
+ const revisedBlock = prepared.revised[revisedCursor]?.block;
611
+ if (baseBlock && revisedBlock) events.push({
612
+ type: "pair",
613
+ baseBlock,
614
+ revisedBlock
615
+ });
616
+ baseCursor += 1;
617
+ revisedCursor += 1;
618
+ }
619
+ emitUnpairedUntil(baseTo, revisedTo);
620
+ };
406
621
  let baseCursor = 0;
407
622
  let revisedCursor = 0;
408
623
  for (const anchor of anchors) {
409
- emitPositionalGap(baseCursor, anchor.baseIndex, revisedCursor, anchor.revisedIndex);
624
+ emitGap(baseCursor, anchor.baseIndex, revisedCursor, anchor.revisedIndex);
410
625
  const baseBlock = prepared.base[anchor.baseIndex]?.block;
411
626
  const revisedBlock = prepared.revised[anchor.revisedIndex]?.block;
412
627
  if (baseBlock && revisedBlock) events.push({
@@ -417,7 +632,7 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
417
632
  baseCursor = anchor.baseIndex + 1;
418
633
  revisedCursor = anchor.revisedIndex + 1;
419
634
  }
420
- emitPositionalGap(baseCursor, prepared.base.length, revisedCursor, prepared.revised.length);
635
+ emitGap(baseCursor, prepared.base.length, revisedCursor, prepared.revised.length);
421
636
  return events;
422
637
  };
423
638
  const alignFolioContentBlocks = (baseBlocks, revisedBlocks, options = {}) => alignFolioContentBlocksInScope(baseBlocks, revisedBlocks, {
@@ -186,6 +186,7 @@ type ParagraphMarkPlan<Block extends FolioContentBlock> = {
186
186
  revisedBlock: Block;
187
187
  separator: string;
188
188
  };
189
+ /** Plans keyed by their first step; each consumes the step after it. */
189
190
  declare const detectFolioContentParagraphMarkPlans: <Block extends FolioContentBlock>(steps: readonly FolioContentAlignmentStep<Block>[]) => ReadonlyMap<number, ParagraphMarkPlan<Block>>;
190
191
  type MovePair<Block extends FolioContentBlock> = {
191
192
  baseBlock: Block;
@@ -431,30 +431,61 @@ const separatorBetween = (whole, head, tail) => {
431
431
  const separator = whole.slice(head.length, whole.length - tail.length);
432
432
  return separator.length === 0 || /^\s+$/u.test(separator) ? separator : null;
433
433
  };
434
+ const splitPlan = ({ baseBlock, head, tail }) => {
435
+ const separator = separatorBetween(baseBlock.text, head.text, tail.text);
436
+ return separator !== null && contentBlocksShareContainer(head, tail) ? {
437
+ type: "split",
438
+ baseBlock,
439
+ revisedBlocks: [head, tail],
440
+ offset: head.text.length,
441
+ separator
442
+ } : null;
443
+ };
444
+ const mergePlan = ({ revisedBlock, head, tail }) => {
445
+ const separator = separatorBetween(revisedBlock.text, head.text, tail.text);
446
+ return separator !== null && contentBlocksShareContainer(head, tail) ? {
447
+ type: "merge",
448
+ baseBlocks: [head, tail],
449
+ revisedBlock,
450
+ separator
451
+ } : null;
452
+ };
453
+ /**
454
+ * A pair next to a one-sided block its text spells together with the pair's
455
+ * other side. Alignment pairs a split paragraph with whichever half reads more
456
+ * like it, so the unpaired half may stand before the pair or after it.
457
+ */
458
+ const paragraphMarkPlan = (step, next) => {
459
+ if (step.type === "pair" && next.type === "revisedOnly") return splitPlan({
460
+ baseBlock: step.baseBlock,
461
+ head: step.revisedBlock,
462
+ tail: next.block
463
+ });
464
+ if (step.type === "pair" && next.type === "baseOnly") return mergePlan({
465
+ revisedBlock: step.revisedBlock,
466
+ head: step.baseBlock,
467
+ tail: next.block
468
+ });
469
+ if (step.type === "revisedOnly" && next.type === "pair") return splitPlan({
470
+ baseBlock: next.baseBlock,
471
+ head: step.block,
472
+ tail: next.revisedBlock
473
+ });
474
+ if (step.type === "baseOnly" && next.type === "pair") return mergePlan({
475
+ revisedBlock: next.revisedBlock,
476
+ head: step.block,
477
+ tail: next.baseBlock
478
+ });
479
+ return null;
480
+ };
481
+ /** Plans keyed by their first step; each consumes the step after it. */
434
482
  const detectFolioContentParagraphMarkPlans = (steps) => {
435
483
  const plans = /* @__PURE__ */ new Map();
436
484
  for (const [index, step] of steps.entries()) {
437
485
  const next = steps[index + 1];
438
- if (step.type !== "pair" || next === void 0) continue;
439
- if (next.type === "revisedOnly") {
440
- const separator = separatorBetween(step.baseBlock.text, step.revisedBlock.text, next.block.text);
441
- if (separator !== null && contentBlocksShareContainer(step.revisedBlock, next.block)) plans.set(index, {
442
- type: "split",
443
- baseBlock: step.baseBlock,
444
- revisedBlocks: [step.revisedBlock, next.block],
445
- offset: step.revisedBlock.text.length,
446
- separator
447
- });
448
- continue;
449
- }
450
- if (next.type !== "baseOnly") continue;
451
- const separator = separatorBetween(step.revisedBlock.text, step.baseBlock.text, next.block.text);
452
- if (separator !== null && contentBlocksShareContainer(step.baseBlock, next.block)) plans.set(index, {
453
- type: "merge",
454
- baseBlocks: [step.baseBlock, next.block],
455
- revisedBlock: step.revisedBlock,
456
- separator
457
- });
486
+ if (next === void 0 || plans.has(index - 1)) continue;
487
+ const plan = paragraphMarkPlan(step, next);
488
+ if (plan !== null) plans.set(index, plan);
458
489
  }
459
490
  return plans;
460
491
  };
@@ -3,7 +3,7 @@ import { expectHyperlinkMarkAttrs, expectRunPropertyChangeMarkAttrs, expectTable
3
3
  import { getDocumentStyleResolver } from "../prosemirror/plugins/documentStyles.js";
4
4
  import { applyMarksToRunFormattingRepresentation, expandRunFormattingCarrier, runFormattingCarrierReviewText } from "../prosemirror/runFormattingInlineCarriers.js";
5
5
  import { readAuthoredRunFormatting, reconcileRunFormattingMarks } from "../prosemirror/runFormattingReconciliation.js";
6
- import { paragraphRunStyleContextAt } from "../prosemirror/runStyleFormatting.js";
6
+ import { nearestParagraphRunStyleContext } from "../prosemirror/runStyleFormatting.js";
7
7
  import { isTableCellRetainedInReviewView } from "../prosemirror/tableCellRevisionVisibility.js";
8
8
  import { canonicalJson } from "../utils/canonicalJson.js";
9
9
  //#region src/compare/inline-provenance.ts
@@ -19,10 +19,11 @@ const hyperlinkIdentity = (node) => {
19
19
  });
20
20
  };
21
21
  const isInserted = (node) => node.marks.some(({ type }) => type.name === "insertion");
22
- const carrierRepresentations = (carrier) => carrier.representations.map((representation) => ({
22
+ const carrierRepresentations = ({ carrier, paragraph }) => carrier.representations.map((representation) => ({
23
23
  ...representation,
24
24
  from: representation.position,
25
- to: representation.position + representation.node.nodeSize
25
+ to: representation.position + representation.node.nodeSize,
26
+ paragraph
26
27
  }));
27
28
  const targetBlockIdLookup = (anchors) => {
28
29
  const ordered = Object.values(anchors).toSorted((left, right) => left.from - right.from || left.to - right.to);
@@ -53,7 +54,13 @@ const collectCarriers = ({ doc, targetSnapshot }) => {
53
54
  const carriers = [];
54
55
  const targetBlockIdAt = targetSnapshot ? targetBlockIdLookup(targetSnapshot.anchors) : void 0;
55
56
  let unanchoredTargetCarrier = false;
57
+ const openParagraphs = [];
56
58
  doc.descendants((node, position) => {
59
+ while ((openParagraphs.at(-1)?.end ?? Number.POSITIVE_INFINITY) <= position) openParagraphs.pop();
60
+ if (node.type.name === "paragraph") openParagraphs.push({
61
+ node,
62
+ end: position + node.nodeSize
63
+ });
57
64
  if ((node.type.name === "tableCell" || node.type.name === "tableHeader") && !isTableCellRetainedInReviewView(expectTableCellAttrs(node).cellMarker?.kind, "final")) return false;
58
65
  if (!node.isInline) return true;
59
66
  const carrier = expandRunFormattingCarrier(node, position);
@@ -69,6 +76,7 @@ const collectCarriers = ({ doc, targetSnapshot }) => {
69
76
  }
70
77
  carriers.push({
71
78
  carrier,
79
+ paragraph: openParagraphs.at(-1)?.node ?? null,
72
80
  text: runFormattingCarrierReviewText(carrier),
73
81
  targetBlockId
74
82
  });
@@ -106,8 +114,8 @@ const matchCarrierStreams = ({ live, target }) => {
106
114
  const targetIsText = targetCarrier.carrier.disposition === "text-run";
107
115
  if (!liveIsText || !targetIsText) {
108
116
  if (liveOffset !== 0 || targetOffset !== 0 || !sameCarrierShape(liveCarrier, targetCarrier)) return null;
109
- const liveRepresentations = carrierRepresentations(liveCarrier.carrier);
110
- const targetRepresentations = carrierRepresentations(targetCarrier.carrier);
117
+ const liveRepresentations = carrierRepresentations(liveCarrier);
118
+ const targetRepresentations = carrierRepresentations(targetCarrier);
111
119
  for (const [index, liveRepresentation] of liveRepresentations.entries()) {
112
120
  const targetRepresentation = targetRepresentations.at(index);
113
121
  if (!targetRepresentation) return null;
@@ -123,8 +131,8 @@ const matchCarrierStreams = ({ live, target }) => {
123
131
  }
124
132
  const sharedLength = Math.min(liveText.length, targetText.length);
125
133
  if (sharedLength === 0 || liveText.slice(0, sharedLength) !== targetText.slice(0, sharedLength)) return null;
126
- const liveRepresentation = carrierRepresentations(liveCarrier.carrier).at(0);
127
- const targetRepresentation = carrierRepresentations(targetCarrier.carrier).at(0);
134
+ const liveRepresentation = carrierRepresentations(liveCarrier).at(0);
135
+ const targetRepresentation = carrierRepresentations(targetCarrier).at(0);
128
136
  if (!liveRepresentation || !targetRepresentation) return null;
129
137
  matched.push({
130
138
  live: {
@@ -152,12 +160,8 @@ const matchCarrierStreams = ({ live, target }) => {
152
160
  }
153
161
  return matched;
154
162
  };
155
- const authoredFormattingAt = ({ doc, representation, styleResolver }) => readAuthoredRunFormatting({
156
- context: paragraphRunStyleContextAt({
157
- doc,
158
- pos: representation.from,
159
- styleResolver
160
- }),
163
+ const authoredFormattingAt = ({ representation, styleResolver }) => readAuthoredRunFormatting({
164
+ context: nearestParagraphRunStyleContext(representation.paragraph, styleResolver),
161
165
  marks: representation.node.marks,
162
166
  styleResolver
163
167
  });
@@ -190,12 +194,10 @@ const matchInlineProvenance = ({ state, targetSnapshot, revisionStamp, originalR
190
194
  const planned = [];
191
195
  for (const segment of matched) {
192
196
  const liveFormatting = authoredFormattingAt({
193
- doc: state.doc,
194
197
  representation: segment.live,
195
198
  styleResolver: liveStyleResolver
196
199
  });
197
200
  const targetFormatting = authoredFormattingAt({
198
- doc: targetDocument,
199
201
  representation: segment.target,
200
202
  styleResolver: targetStyleResolver
201
203
  });
@@ -220,11 +222,7 @@ const matchInlineProvenance = ({ state, targetSnapshot, revisionStamp, originalR
220
222
  let nextRevisionId = revisionStamp.idSeed;
221
223
  const changedTargetBlockIds = /* @__PURE__ */ new Set();
222
224
  for (const change of planned.toReversed()) {
223
- const context = paragraphRunStyleContextAt({
224
- doc: state.doc,
225
- pos: change.live.from,
226
- styleResolver: liveStyleResolver
227
- });
225
+ const context = nearestParagraphRunStyleContext(change.live.paragraph, liveStyleResolver);
228
226
  const formattingMarks = reconcileRunFormattingMarks({
229
227
  authoredFormatting: change.formatting,
230
228
  context,
@@ -321,11 +319,9 @@ const sameAuthoredInlineProvenance = (baseSnapshot, targetSnapshot) => {
321
319
  const baseStyleResolver = styleResolverOf(baseSnapshot);
322
320
  const targetStyleResolver = styleResolverOf(targetSnapshot);
323
321
  return matched.every(({ live, target }) => hyperlinkIdentity(live.node) === hyperlinkIdentity(target.node) && canonicalJson(authoredFormattingAt({
324
- doc: baseDocument,
325
322
  representation: live,
326
323
  styleResolver: baseStyleResolver
327
324
  })) === canonicalJson(authoredFormattingAt({
328
- doc: targetDocument,
329
325
  representation: target,
330
326
  styleResolver: targetStyleResolver
331
327
  })));
@@ -108,9 +108,22 @@ const boundaryParagraphEntriesOf = (document) => {
108
108
  if (!children) return null;
109
109
  return children.flatMap((child) => child.type === "paragraph" ? [child.paragraph] : []);
110
110
  };
111
+ const hasSectionEndpoint = (document) => {
112
+ let found = false;
113
+ document.forEach((node) => {
114
+ if (node.type.name === "paragraph" && expectParagraphAttrs(node)._sectionProperties !== void 0) found = true;
115
+ });
116
+ return found;
117
+ };
111
118
  /** Stage section deltas after document operations, against the accepted projection. */
112
119
  const stageSectionBoundaryProperties = ({ state, target, originalRevisionIdSeed, revisionStamp, author, maxRanges, mapTargetProperties }) => {
113
120
  if (!Number.isSafeInteger(maxRanges) || maxRanges < 0) return { status: "budget-exceeded" };
121
+ if (!hasSectionEndpoint(target)) return {
122
+ status: "matched",
123
+ transaction: state.tr,
124
+ nextRevisionId: revisionStamp.idSeed,
125
+ rangeCount: 0
126
+ };
114
127
  const reviewedState = resolveAllChangesInHeadlessStateWithMapping(state, "accept");
115
128
  const reviewed = boundaryParagraphEntriesOf(reviewedState.state.doc);
116
129
  const targets = boundaryParagraphEntriesOf(target);
@@ -3,7 +3,7 @@ import { document_d_exports } from "../types/document.js";
3
3
  type SelectiveSaveOptions = {
4
4
  /** Changed paragraph IDs to selectively patch */
5
5
  changedParaIds: Set<string>;
6
- /** Whether structural changes occurred (paragraph add/delete) */
6
+ /** Whether paragraph membership, order, or block structure changed. */
7
7
  structuralChange: boolean;
8
8
  /** Whether any changes affected paragraphs without paraId */
9
9
  hasUntrackedChanges: boolean;