@stll/folio-core 0.50.0 → 0.51.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -836,6 +836,137 @@ const orderDeletionsFirst = (runs) => {
836
836
  flush();
837
837
  return ordered;
838
838
  };
839
+ const WORD_CHARACTER = /^[\p{L}\p{N}\p{M}]$/u;
840
+ const LINE_BREAK = /^[\n\r\p{Zl}\p{Zp}]$/u;
841
+ /**
842
+ * How well a change edge between two code points reads, after
843
+ * diff-match-patch's semantic score: the string's edge (an empty side), then
844
+ * a line break, then the gap after a sentence or clause mark, then any space,
845
+ * then any other mark; inside a word is worst.
846
+ */
847
+ const boundaryScore = (previous, next) => {
848
+ if (previous.length === 0 || next.length === 0) return 6;
849
+ if (LINE_BREAK.test(previous) || LINE_BREAK.test(next)) return 4;
850
+ const previousIsSpace = WHITESPACE.test(previous);
851
+ const nextIsSpace = WHITESPACE.test(next);
852
+ if (!previousIsSpace && !WORD_CHARACTER.test(previous) && nextIsSpace) return 3;
853
+ if (previousIsSpace || nextIsSpace) return 2;
854
+ return WORD_CHARACTER.test(previous) && WORD_CHARACTER.test(next) ? 0 : 1;
855
+ };
856
+ /** True when `position` falls between the two halves of a surrogate pair. */
857
+ const splitsSurrogatePair = (text, position) => {
858
+ const low = text.charCodeAt(position);
859
+ const high = text.charCodeAt(position - 1);
860
+ return low >= 56320 && low <= 57343 && high >= 55296 && high <= 56319;
861
+ };
862
+ /**
863
+ * Where, among the lossless positions of one change, it reads best. The
864
+ * change may slide by one character whenever the character it gives up
865
+ * equals the one it takes on, which keeps both strings intact. Only the
866
+ * window decides the result, so an insertion and the deletion that undoes it
867
+ * land on the same text; the leftmost of equally good positions wins.
868
+ */
869
+ const bestSlideStart = (window) => {
870
+ const { text, start, length, minimumStart, maximumEnd } = window;
871
+ const scoreAt = (position) => boundaryScore(position === 0 ? window.outsideBefore : codePointBefore(text, position), position === text.length ? window.outsideAfter : codePointAt(text, position));
872
+ let leftmost = start;
873
+ while (leftmost > minimumStart && text[leftmost - 1] === text[leftmost + length - 1]) leftmost--;
874
+ let best = start;
875
+ let bestScore = -1;
876
+ for (let candidate = leftmost; candidate + length <= maximumEnd; candidate++) {
877
+ const end = candidate + length;
878
+ if (!splitsSurrogatePair(text, candidate) && !splitsSurrogatePair(text, end)) {
879
+ const startScore = scoreAt(candidate);
880
+ const endScore = scoreAt(end);
881
+ if (!(candidate !== start && (startScore === 0 || endScore === 0)) && startScore + endScore > bestScore) {
882
+ best = candidate;
883
+ bestScore = startScore + endScore;
884
+ }
885
+ }
886
+ if (end === text.length || text[candidate] !== text[end]) break;
887
+ }
888
+ return best;
889
+ };
890
+ /**
891
+ * Whether a change may end up directly beside `neighbour` once the equality
892
+ * between them empties: always beside nothing, an equality or its own kind,
893
+ * and a deletion may precede an insertion; an insertion before a deletion
894
+ * would break deletion-first order.
895
+ */
896
+ const mayAbut = (first, second) => first === void 0 || second === void 0 || first === "equal" || second === "equal" || first === second || first === "del" && second === "ins";
897
+ /** The code point nearest `index`, walking by `step`, on the side `type` belongs to. */
898
+ const sideCodePoint = (segments, { from, step }, type) => {
899
+ for (let index = from; index >= 0 && index < segments.length; index += step) {
900
+ const segment = segments[index];
901
+ if (segment === void 0 || segment.text.length === 0) continue;
902
+ if (segment.type === "equal" || segment.type === type) return step === 1 ? codePointAt(segment.text, 0) : codePointBefore(segment.text, segment.text.length);
903
+ }
904
+ return "";
905
+ };
906
+ const mergeAdjacentSegments = (segments) => {
907
+ const merged = [];
908
+ for (const segment of segments) {
909
+ const last = merged.at(-1);
910
+ if (segment.text.length === 0) continue;
911
+ if (last?.type === segment.type) {
912
+ last.text += segment.text;
913
+ continue;
914
+ }
915
+ merged.push(segment);
916
+ }
917
+ return merged;
918
+ };
919
+ /**
920
+ * Slide every insertion or deletion that sits between equalities to the
921
+ * position that reads best.
922
+ *
923
+ * An LCS places a change arbitrarily among equally long alignments: an
924
+ * appended sentence can be marked as `". New sentence"` before the old
925
+ * sentence's full stop, not `" New sentence."` after it. The redline then
926
+ * marks a stop the author never touched and leaves the new sentence's own
927
+ * unmarked. Sliding moves only where a change's edges fall, so both strings
928
+ * still reconstruct.
929
+ */
930
+ const slideChangesToReadableBoundaries = (segments) => {
931
+ const slid = [
932
+ {
933
+ type: "equal",
934
+ text: ""
935
+ },
936
+ ...segments.map((segment) => ({ ...segment })),
937
+ {
938
+ type: "equal",
939
+ text: ""
940
+ }
941
+ ];
942
+ for (let index = 1; index < slid.length - 1; index++) {
943
+ const change = slid[index];
944
+ const previous = slid[index - 1];
945
+ const next = slid[index + 1];
946
+ if (change === void 0 || previous === void 0 || next === void 0 || change.type === "equal" || previous.type !== "equal" || next.type !== "equal") continue;
947
+ const text = previous.text + change.text + next.text;
948
+ const start = bestSlideStart({
949
+ text,
950
+ start: previous.text.length,
951
+ length: change.text.length,
952
+ minimumStart: mayAbut(slid[index - 2]?.type, change.type) ? 0 : 1,
953
+ maximumEnd: mayAbut(change.type, slid[index + 2]?.type) ? text.length : text.length - 1,
954
+ outsideBefore: sideCodePoint(slid, {
955
+ from: index - 2,
956
+ step: -1
957
+ }, change.type),
958
+ outsideAfter: sideCodePoint(slid, {
959
+ from: index + 2,
960
+ step: 1
961
+ }, change.type)
962
+ });
963
+ const end = start + change.text.length;
964
+ previous.text = text.slice(0, start);
965
+ change.text = text.slice(start, end);
966
+ next.text = text.slice(end);
967
+ }
968
+ return mergeAdjacentSegments(slid);
969
+ };
839
970
  const wholeStringReplacement = (before, after) => {
840
971
  const segments = [];
841
972
  if (before.length > 0) segments.push({
@@ -903,7 +1034,9 @@ const diffWordSegmentsWithBudget = (before, after, options, budget) => {
903
1034
  const monotoneRuns = alignMonotoneChange(before, after, beforeTokens, afterTokens, normalization, monotoneDirection);
904
1035
  marked = whitespaceIsSignificant ? separateWhitespaceChanges(monotoneRuns) : monotoneRuns;
905
1036
  }
906
- return toSegments(orderDeletionsFirst(marked));
1037
+ const segments = toSegments(orderDeletionsFirst(marked));
1038
+ const equalRunsAreExact = requestedNormalization.case !== true && requestedNormalization.whitespace !== true;
1039
+ return granularity === "word" && equalRunsAreExact ? slideChangesToReadableBoundaries(segments) : segments;
907
1040
  };
908
1041
  /**
909
1042
  * One internal comparison/apply scope. Every diff shares the same quadratic
@@ -260,6 +260,44 @@ const pairByUniqueExactText = ({ base, revised, baseFrom, baseTo, revisedFrom, r
260
260
  }
261
261
  return pairs.toReversed();
262
262
  };
263
+ /**
264
+ * Unique exact text anchors, then again inside every sub-gap they leave:
265
+ * wording repeated elsewhere in a document is still identity evidence within
266
+ * the one gap where it is unique, so a section's pairing does not depend on
267
+ * whether a neighbouring section happens to repeat it. Each nested pass is
268
+ * charged to the structural allowance; a refused pass leaves its sub-gap to
269
+ * the similarity pairing.
270
+ */
271
+ const pairByNestedUniqueExactText = ({ workSession, ...gap }) => {
272
+ const anchors = [];
273
+ const pending = [gap];
274
+ for (let current = pending.pop(); current !== void 0; current = pending.pop()) {
275
+ const found = pairByUniqueExactText(current);
276
+ if (found.length === 0) continue;
277
+ anchors.push(...found);
278
+ let baseFrom = current.baseFrom;
279
+ let revisedFrom = current.revisedFrom;
280
+ for (const anchor of [...found, {
281
+ baseIndex: current.baseTo,
282
+ revisedIndex: current.revisedTo
283
+ }]) {
284
+ const size = anchor.baseIndex - baseFrom + (anchor.revisedIndex - revisedFrom);
285
+ if (anchor.baseIndex > baseFrom && anchor.revisedIndex > revisedFrom && size <= workSession.remainingStructuralTokenLookups) {
286
+ workSession.remainingStructuralTokenLookups -= size;
287
+ pending.push({
288
+ ...current,
289
+ baseFrom,
290
+ baseTo: anchor.baseIndex,
291
+ revisedFrom,
292
+ revisedTo: anchor.revisedIndex
293
+ });
294
+ }
295
+ baseFrom = anchor.baseIndex + 1;
296
+ revisedFrom = anchor.revisedIndex + 1;
297
+ }
298
+ }
299
+ return anchors.toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
300
+ };
263
301
  const pairsInAnchorGaps = ({ baseLength, revisedLength, anchors, pairGap }) => {
264
302
  const pairs = [];
265
303
  let baseFrom = 0;
@@ -312,9 +350,122 @@ const crossedExactTextBlocks = ({ base, revised, anchors, canPair }) => {
312
350
  revised: crossedRevised
313
351
  };
314
352
  };
353
+ /**
354
+ * Minimum multiset Dice similarity for two blocks to pair inside a gap. At
355
+ * 0.5 a paragraph still pairs after gaining up to twice its own length
356
+ * (2n / (n + 3n)); below it, more of the pair would read as changed than
357
+ * kept, which a removal beside an insertion says better.
358
+ */
359
+ const GAP_PAIR_SIMILARITY_THRESHOLD = .5;
360
+ /**
361
+ * Gap similarities are compared as integers, so a tie is exact rather than an
362
+ * accident of floating-point summation, and a tie-break can sit below the
363
+ * smallest similarity step.
364
+ */
365
+ const GAP_PAIR_SIMILARITY_SCALE = 1e3;
366
+ /**
367
+ * Words as case-folded runs of letters, marks and digits: punctuation glued to
368
+ * a word ("paragraph." against "paragraph") and a capital at a sentence start
369
+ * would otherwise count a kept word as changed.
370
+ */
371
+ const gapBlockTokens = (text) => {
372
+ const counts = /* @__PURE__ */ new Map();
373
+ let total = 0;
374
+ for (const match of text.toLowerCase().matchAll(/[\p{L}\p{M}\p{N}]+/gu)) {
375
+ total += 1;
376
+ counts.set(match[0], (counts.get(match[0]) ?? 0) + 1);
377
+ }
378
+ return {
379
+ counts,
380
+ total
381
+ };
382
+ };
383
+ const gapBlockSimilarity = (base, revised, workSession) => {
384
+ if (base.total === 0 && revised.total === 0) return {
385
+ status: "measured",
386
+ value: 1
387
+ };
388
+ if (base.total === 0 || revised.total === 0) return {
389
+ status: "measured",
390
+ value: GAP_PAIR_SIMILARITY_THRESHOLD
391
+ };
392
+ const [tokens, counterparts] = base.counts.size <= revised.counts.size ? [base.counts, revised.counts] : [revised.counts, base.counts];
393
+ if (tokens.size > workSession.remainingStructuralTokenLookups) return { status: "budget-exceeded" };
394
+ workSession.remainingStructuralTokenLookups -= tokens.size;
395
+ let shared = 0;
396
+ for (const [token, count] of tokens) shared += Math.min(count, counterparts.get(token) ?? 0);
397
+ return {
398
+ status: "measured",
399
+ value: 2 * shared / (base.total + revised.total)
400
+ };
401
+ };
402
+ /**
403
+ * The order-preserving pairs of one gap that maximise their summed
404
+ * similarity, each at least `GAP_PAIR_SIMILARITY_THRESHOLD`; offsets are into
405
+ * the gap's slices. Pairing by position instead fuses an inserted block with
406
+ * the neighbour it pushed down, and every block after it with the next one's
407
+ * wording. Null when the work budget refuses the gap.
408
+ */
409
+ const pairGapBySimilarity = ({ base, revised, canPair, workSession }) => {
410
+ const baseCount = base.length;
411
+ const revisedCount = revised.length;
412
+ if (!claimFolioContentAlignmentCells(baseCount, revisedCount, workSession)) return null;
413
+ const revisedTokens = revised.map((block) => gapBlockTokens(block.block.text));
414
+ const similarityStep = Math.min(baseCount, revisedCount) + 1;
415
+ const similarity = new Float64Array(baseCount * revisedCount).fill(-1);
416
+ const candidateBase = /* @__PURE__ */ new Set();
417
+ const candidateRevised = /* @__PURE__ */ new Set();
418
+ for (const [baseOffset, baseBlock] of base.entries()) {
419
+ const baseTokens = gapBlockTokens(baseBlock.block.text);
420
+ for (const [revisedOffset, revisedBlock] of revised.entries()) {
421
+ if (!canPair(baseBlock, revisedBlock)) continue;
422
+ const tokens = revisedTokens[revisedOffset] ?? panic("A gap block has no token profile");
423
+ const measured = baseBlock.block.text === revisedBlock.block.text ? {
424
+ status: "measured",
425
+ value: 1
426
+ } : gapBlockSimilarity(baseTokens, tokens, workSession);
427
+ if (measured.status === "budget-exceeded") return null;
428
+ if (measured.value >= GAP_PAIR_SIMILARITY_THRESHOLD) {
429
+ const sameLabel = baseBlock.block.displayLabel !== void 0 && baseBlock.block.displayLabel === revisedBlock.block.displayLabel;
430
+ similarity[baseOffset * revisedCount + revisedOffset] = Math.round(measured.value * GAP_PAIR_SIMILARITY_SCALE) * similarityStep + (sameLabel ? 1 : 0);
431
+ candidateBase.add(baseOffset);
432
+ candidateRevised.add(revisedOffset);
433
+ }
434
+ }
435
+ }
436
+ const width = revisedCount + 1;
437
+ const scores = new Float64Array((baseCount + 1) * width);
438
+ const cellSimilarity = (baseOffset, revisedOffset) => similarity[baseOffset * revisedCount + revisedOffset] ?? -1;
439
+ const score = (row, column) => scores[row * width + column] ?? 0;
440
+ for (let row = baseCount - 1; row >= 0; row--) for (let column = revisedCount - 1; column >= 0; column--) {
441
+ const cell = cellSimilarity(row, column);
442
+ scores[row * width + column] = Math.max(score(row + 1, column), score(row, column + 1), cell < 0 ? 0 : score(row + 1, column + 1) + cell);
443
+ }
444
+ const pairs = [];
445
+ let row = 0;
446
+ let column = 0;
447
+ while (row < baseCount && column < revisedCount) {
448
+ const cell = cellSimilarity(row, column);
449
+ if (cell >= 0 && score(row, column) === score(row + 1, column + 1) + cell) {
450
+ pairs.push({
451
+ baseIndex: row,
452
+ revisedIndex: column
453
+ });
454
+ row += 1;
455
+ column += 1;
456
+ } else if (score(row, column) === score(row + 1, column)) row += 1;
457
+ else column += 1;
458
+ }
459
+ return {
460
+ pairs,
461
+ candidateBase,
462
+ candidateRevised
463
+ };
464
+ };
315
465
  const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
316
466
  const stableIdMismatch = options.stableIdMismatch ?? "separate";
317
467
  const idStability = options.idStability ?? folioContentIdStability;
468
+ const workSession = options.workSession ?? createFolioContentAlignmentWorkSession();
318
469
  const prepared = prepareAlignmentBlocks({
319
470
  baseBlocks,
320
471
  revisedBlocks,
@@ -335,14 +486,15 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
335
486
  baseLength: prepared.base.length,
336
487
  revisedLength: prepared.revised.length,
337
488
  anchors: stableIdAnchors,
338
- pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) => pairByUniqueExactText({
489
+ pairGap: (baseFrom, baseTo, revisedFrom, revisedTo) => pairByNestedUniqueExactText({
339
490
  base: prepared.base,
340
491
  revised: prepared.revised,
341
492
  baseFrom,
342
493
  baseTo,
343
494
  revisedFrom,
344
495
  revisedTo,
345
- canPair
496
+ canPair,
497
+ workSession
346
498
  })
347
499
  });
348
500
  const exactAndStableAnchors = [...stableIdAnchors, ...exactTextAnchors].toSorted((left, right) => left.baseIndex - right.baseIndex || left.revisedIndex - right.revisedIndex);
@@ -368,12 +520,13 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
368
520
  canPair
369
521
  });
370
522
  const events = [];
523
+ const fusesACrossing = (baseBlock, revisedBlock) => baseBlock.block.text !== revisedBlock.block.text && crossed.base.has(baseBlock.index) && crossed.revised.has(revisedBlock.index);
371
524
  const emitPositionalGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
372
525
  const pairedCount = Math.min(baseTo - baseFrom, revisedTo - revisedFrom);
373
526
  for (let offset = 0; offset < pairedCount; offset++) {
374
527
  const baseBlock = prepared.base[baseFrom + offset];
375
528
  const revisedBlock = prepared.revised[revisedFrom + offset];
376
- if (baseBlock && revisedBlock) if (baseBlock.block.text !== revisedBlock.block.text && crossed.base.has(baseBlock.index) && crossed.revised.has(revisedBlock.index) || !canPair(baseBlock, revisedBlock)) {
529
+ if (baseBlock && revisedBlock) if (fusesACrossing(baseBlock, revisedBlock) || !canPair(baseBlock, revisedBlock)) {
377
530
  events.push({
378
531
  type: "baseOnly",
379
532
  block: baseBlock.block
@@ -403,10 +556,72 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
403
556
  });
404
557
  }
405
558
  };
559
+ /**
560
+ * A gap's blocks pair by similarity, whether or not its sides are equal in
561
+ * length: an equal gap can hide an insertion beside a deletion, and pairing
562
+ * it by position reads each kept block as a rewrite of its neighbour. The
563
+ * similar pairs anchor the rest; blocks between two anchors pair by position
564
+ * only when neither side offered any candidate and the counts match, which
565
+ * reads a block rewritten beyond recognition as the modification it is.
566
+ */
567
+ const emitGap = (baseFrom, baseTo, revisedFrom, revisedTo) => {
568
+ const pairing = pairGapBySimilarity({
569
+ base: prepared.base.slice(baseFrom, baseTo),
570
+ revised: prepared.revised.slice(revisedFrom, revisedTo),
571
+ canPair: (baseBlock, revisedBlock) => !fusesACrossing(baseBlock, revisedBlock) && canPair(baseBlock, revisedBlock),
572
+ workSession
573
+ });
574
+ if (pairing === null) {
575
+ emitPositionalGap(baseFrom, baseTo, revisedFrom, revisedTo);
576
+ return;
577
+ }
578
+ const { pairs, candidateBase, candidateRevised } = pairing;
579
+ let baseCursor = baseFrom;
580
+ let revisedCursor = revisedFrom;
581
+ const hasCandidate = (candidates, from, to) => {
582
+ for (let offset = from; offset < to; offset++) if (candidates.has(offset)) return true;
583
+ return false;
584
+ };
585
+ const emitUnpairedUntil = (baseEnd, revisedEnd) => {
586
+ if (baseEnd - baseCursor === revisedEnd - revisedCursor && !hasCandidate(candidateBase, baseCursor - baseFrom, baseEnd - baseFrom) && !hasCandidate(candidateRevised, revisedCursor - revisedFrom, revisedEnd - revisedFrom)) {
587
+ emitPositionalGap(baseCursor, baseEnd, revisedCursor, revisedEnd);
588
+ baseCursor = baseEnd;
589
+ revisedCursor = revisedEnd;
590
+ return;
591
+ }
592
+ for (; baseCursor < baseEnd; baseCursor++) {
593
+ const block = prepared.base[baseCursor]?.block;
594
+ if (block) events.push({
595
+ type: "baseOnly",
596
+ block
597
+ });
598
+ }
599
+ for (; revisedCursor < revisedEnd; revisedCursor++) {
600
+ const block = prepared.revised[revisedCursor]?.block;
601
+ if (block) events.push({
602
+ type: "revisedOnly",
603
+ block
604
+ });
605
+ }
606
+ };
607
+ for (const pair of pairs) {
608
+ emitUnpairedUntil(baseFrom + pair.baseIndex, revisedFrom + pair.revisedIndex);
609
+ const baseBlock = prepared.base[baseCursor]?.block;
610
+ const revisedBlock = prepared.revised[revisedCursor]?.block;
611
+ if (baseBlock && revisedBlock) events.push({
612
+ type: "pair",
613
+ baseBlock,
614
+ revisedBlock
615
+ });
616
+ baseCursor += 1;
617
+ revisedCursor += 1;
618
+ }
619
+ emitUnpairedUntil(baseTo, revisedTo);
620
+ };
406
621
  let baseCursor = 0;
407
622
  let revisedCursor = 0;
408
623
  for (const anchor of anchors) {
409
- emitPositionalGap(baseCursor, anchor.baseIndex, revisedCursor, anchor.revisedIndex);
624
+ emitGap(baseCursor, anchor.baseIndex, revisedCursor, anchor.revisedIndex);
410
625
  const baseBlock = prepared.base[anchor.baseIndex]?.block;
411
626
  const revisedBlock = prepared.revised[anchor.revisedIndex]?.block;
412
627
  if (baseBlock && revisedBlock) events.push({
@@ -417,7 +632,7 @@ const alignFolioContentBlocksInScope = (baseBlocks, revisedBlocks, options) => {
417
632
  baseCursor = anchor.baseIndex + 1;
418
633
  revisedCursor = anchor.revisedIndex + 1;
419
634
  }
420
- emitPositionalGap(baseCursor, prepared.base.length, revisedCursor, prepared.revised.length);
635
+ emitGap(baseCursor, prepared.base.length, revisedCursor, prepared.revised.length);
421
636
  return events;
422
637
  };
423
638
  const alignFolioContentBlocks = (baseBlocks, revisedBlocks, options = {}) => alignFolioContentBlocksInScope(baseBlocks, revisedBlocks, {
@@ -186,6 +186,7 @@ type ParagraphMarkPlan<Block extends FolioContentBlock> = {
186
186
  revisedBlock: Block;
187
187
  separator: string;
188
188
  };
189
+ /** Plans keyed by their first step; each consumes the step after it. */
189
190
  declare const detectFolioContentParagraphMarkPlans: <Block extends FolioContentBlock>(steps: readonly FolioContentAlignmentStep<Block>[]) => ReadonlyMap<number, ParagraphMarkPlan<Block>>;
190
191
  type MovePair<Block extends FolioContentBlock> = {
191
192
  baseBlock: Block;
@@ -431,30 +431,61 @@ const separatorBetween = (whole, head, tail) => {
431
431
  const separator = whole.slice(head.length, whole.length - tail.length);
432
432
  return separator.length === 0 || /^\s+$/u.test(separator) ? separator : null;
433
433
  };
434
+ const splitPlan = ({ baseBlock, head, tail }) => {
435
+ const separator = separatorBetween(baseBlock.text, head.text, tail.text);
436
+ return separator !== null && contentBlocksShareContainer(head, tail) ? {
437
+ type: "split",
438
+ baseBlock,
439
+ revisedBlocks: [head, tail],
440
+ offset: head.text.length,
441
+ separator
442
+ } : null;
443
+ };
444
+ const mergePlan = ({ revisedBlock, head, tail }) => {
445
+ const separator = separatorBetween(revisedBlock.text, head.text, tail.text);
446
+ return separator !== null && contentBlocksShareContainer(head, tail) ? {
447
+ type: "merge",
448
+ baseBlocks: [head, tail],
449
+ revisedBlock,
450
+ separator
451
+ } : null;
452
+ };
453
+ /**
454
+ * A pair next to a one-sided block its text spells together with the pair's
455
+ * other side. Alignment pairs a split paragraph with whichever half reads more
456
+ * like it, so the unpaired half may stand before the pair or after it.
457
+ */
458
+ const paragraphMarkPlan = (step, next) => {
459
+ if (step.type === "pair" && next.type === "revisedOnly") return splitPlan({
460
+ baseBlock: step.baseBlock,
461
+ head: step.revisedBlock,
462
+ tail: next.block
463
+ });
464
+ if (step.type === "pair" && next.type === "baseOnly") return mergePlan({
465
+ revisedBlock: step.revisedBlock,
466
+ head: step.baseBlock,
467
+ tail: next.block
468
+ });
469
+ if (step.type === "revisedOnly" && next.type === "pair") return splitPlan({
470
+ baseBlock: next.baseBlock,
471
+ head: step.block,
472
+ tail: next.revisedBlock
473
+ });
474
+ if (step.type === "baseOnly" && next.type === "pair") return mergePlan({
475
+ revisedBlock: next.revisedBlock,
476
+ head: step.block,
477
+ tail: next.baseBlock
478
+ });
479
+ return null;
480
+ };
481
+ /** Plans keyed by their first step; each consumes the step after it. */
434
482
  const detectFolioContentParagraphMarkPlans = (steps) => {
435
483
  const plans = /* @__PURE__ */ new Map();
436
484
  for (const [index, step] of steps.entries()) {
437
485
  const next = steps[index + 1];
438
- if (step.type !== "pair" || next === void 0) continue;
439
- if (next.type === "revisedOnly") {
440
- const separator = separatorBetween(step.baseBlock.text, step.revisedBlock.text, next.block.text);
441
- if (separator !== null && contentBlocksShareContainer(step.revisedBlock, next.block)) plans.set(index, {
442
- type: "split",
443
- baseBlock: step.baseBlock,
444
- revisedBlocks: [step.revisedBlock, next.block],
445
- offset: step.revisedBlock.text.length,
446
- separator
447
- });
448
- continue;
449
- }
450
- if (next.type !== "baseOnly") continue;
451
- const separator = separatorBetween(step.revisedBlock.text, step.baseBlock.text, next.block.text);
452
- if (separator !== null && contentBlocksShareContainer(step.baseBlock, next.block)) plans.set(index, {
453
- type: "merge",
454
- baseBlocks: [step.baseBlock, next.block],
455
- revisedBlock: step.revisedBlock,
456
- separator
457
- });
486
+ if (next === void 0 || plans.has(index - 1)) continue;
487
+ const plan = paragraphMarkPlan(step, next);
488
+ if (plan !== null) plans.set(index, plan);
458
489
  }
459
490
  return plans;
460
491
  };
@@ -3,7 +3,7 @@ import { document_d_exports } from "../types/document.js";
3
3
  type SelectiveSaveOptions = {
4
4
  /** Changed paragraph IDs to selectively patch */
5
5
  changedParaIds: Set<string>;
6
- /** Whether structural changes occurred (paragraph add/delete) */
6
+ /** Whether paragraph membership, order, or block structure changed. */
7
7
  structuralChange: boolean;
8
8
  /** Whether any changes affected paragraphs without paraId */
9
9
  hasUntrackedChanges: boolean;
@@ -8,12 +8,13 @@ import { isUnsafePackagePath } from "./packageParts.js";
8
8
  import { RELATIONSHIP_TYPES } from "./relsParser.js";
9
9
  import { COMMENTS_CONTENT_TYPE, COMMENTS_EXTENDED_PART_LOWER, addCommentsExtendedOverride, addCommentsExtendedRelationship, applyUpdatesToZip, collectHeaderFooterUpdates, findMaxRId, hasModelDrivenPictureWatermark, hasUnmaterializedHeaderFooter, hasUnmaterializedInlineResources, updateCoreProperties, withoutAttachedTemplate } from "./rezip.js";
10
10
  import "./selectiveSaveFlags.js";
11
- import { buildPatchedDocumentXml, buildPatchedNoteXml, collectParaIds, patchNumberingDefinitions } from "./selectiveXmlPatch.js";
11
+ import { buildPatchedDocumentXml, buildPatchedNoteXml, collectAddedNumberingDefs, collectParaIds, patchNumberingDefinitions } from "./selectiveXmlPatch.js";
12
12
  import { planCommentParts, serializeComments, serializeCommentsExtended } from "./serializer/commentSerializer.js";
13
13
  import { serializeDocument } from "./serializer/documentSerializer.js";
14
14
  import { serializeEndnotes, serializeFootnotes } from "./serializer/noteSerializer.js";
15
15
  import { serializeNumberingXml } from "./serializer/numberingSerializer.js";
16
16
  import { readRootNamespaceBindings } from "./serializer/partNamespaces.js";
17
+ import { buildStructuralDocumentPatch } from "./structuralXmlPatch.js";
17
18
  import { DOCX_CONFORMANCE_CLASSES } from "@stll/docx-core/model";
18
19
  //#region src/docx/selectiveSave.ts
19
20
  /**
@@ -174,7 +175,6 @@ const queueSettingsUpdates = async (zip, updates) => {
174
175
  async function attemptSelectiveSave(doc, originalBuffer, options) {
175
176
  const { changedParaIds, structuralChange, hasUntrackedChanges } = options;
176
177
  const maxBytes = options.maxBytes ?? 104857600;
177
- if (structuralChange) return null;
178
178
  if (hasUntrackedChanges) return null;
179
179
  if (originalBuffer.byteLength > maxBytes) return null;
180
180
  const content = doc.package.document.content;
@@ -197,22 +197,32 @@ async function attemptSelectiveSave(doc, originalBuffer, options) {
197
197
  const zip = await (await import("jszip")).default.loadAsync(originalBuffer);
198
198
  for (const [path, file] of Object.entries(zip.files)) if (!file.dir && isUnsafePackagePath(path)) return null;
199
199
  const updates = /* @__PURE__ */ new Map();
200
- if (changedParaIds.size > 0) {
200
+ if (changedParaIds.size > 0 || structuralChange) {
201
201
  const docXmlFile = zip.file("word/document.xml");
202
202
  if (!docXmlFile) return null;
203
203
  const originalDocXml = await docXmlFile.async("text");
204
204
  const serializedDocXml = serializeDocument(doc, readRootNamespaceBindings(originalDocXml));
205
205
  const bodyParaIds = collectParaIds(serializedDocXml);
206
+ const originalBodyParaIds = structuralChange ? collectParaIds(originalDocXml) : void 0;
207
+ if (structuralChange && doc.package.numbering) {
208
+ const numberingFile = findZipEntryCaseInsensitive(zip, "word/numbering.xml");
209
+ const added = collectAddedNumberingDefs(numberingFile ? serializeNumberingXml(parseNumbering(await numberingFile.async("text")).definitions) : "", serializeNumberingXml(doc.package.numbering));
210
+ if (added.abstractNums.size > 0 || added.nums.size > 0) return null;
211
+ }
206
212
  const bodyChangedIds = /* @__PURE__ */ new Set();
207
213
  const noteCandidateIds = /* @__PURE__ */ new Set();
208
- for (const id of changedParaIds) if (bodyParaIds.has(id)) bodyChangedIds.add(id);
214
+ for (const id of changedParaIds) if (bodyParaIds.has(id) || originalBodyParaIds?.has(id)) bodyChangedIds.add(id);
209
215
  else noteCandidateIds.add(id);
210
216
  if (noteCandidateIds.size > 0) {
211
217
  const unrouted = await patchNoteParts(zip, doc, noteCandidateIds, updates);
212
218
  if (unrouted === null || unrouted.size > 0) return null;
213
219
  }
214
- if (bodyChangedIds.size > 0) {
215
- const patchedDocXml = buildPatchedDocumentXml(originalDocXml, serializedDocXml, bodyChangedIds);
220
+ if (bodyChangedIds.size > 0 || structuralChange) {
221
+ const patchedDocXml = structuralChange ? buildStructuralDocumentPatch({
222
+ originalXml: originalDocXml,
223
+ serializedXml: serializedDocXml,
224
+ changedIds: bodyChangedIds
225
+ }) : buildPatchedDocumentXml(originalDocXml, serializedDocXml, bodyChangedIds);
216
226
  if (!patchedDocXml) return null;
217
227
  updates.set("word/document.xml", patchedDocXml);
218
228
  }
@@ -0,0 +1,14 @@
1
+ //#region src/docx/structuralXmlPatch.d.ts
2
+ type StructuralPatchOptions = {
3
+ originalXml: string;
4
+ serializedXml: string;
5
+ changedIds: ReadonlySet<string>;
6
+ };
7
+ /**
8
+ * Insert/delete direct body paragraphs between surviving paragraph/table anchors.
9
+ * Tables are opaque barriers: their modeled content and order must stay identical.
10
+ * Id-less sources need an explicit ensureParaIds ingest before structural editing.
11
+ */
12
+ declare const buildStructuralDocumentPatch: ({ originalXml, serializedXml, changedIds }: StructuralPatchOptions) => string | null;
13
+ //#endregion
14
+ export { buildStructuralDocumentPatch };
@@ -0,0 +1,209 @@
1
+ import { canonicalJson } from "../utils/canonicalJson.js";
2
+ import { parseDocumentBody } from "./documentParser.js";
3
+ import { paraIdAttribute } from "./paraIdAttribute.js";
4
+ import { spliceXml } from "./selectiveXmlPatch.js";
5
+ import { readRootNamespaceBindings, serializePartElement } from "./serializer/partNamespaces.js";
6
+ import { NAMESPACES, WORDPROCESSINGML_NAMESPACE_URIS, getChildElements, getLocalName, getNamespaceUri, parseXmlDocument } from "./xmlParser.js";
7
+ //#region src/docx/structuralXmlPatch.ts
8
+ /** Conservative body-paragraph splices; source offsets never come from reserialization. */
9
+ const isWordElement = (element, name) => getLocalName(element.name) === name && WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(element) ?? "");
10
+ /** Match XML tokens, skipping quoted delimiters, comments, CDATA and processing instructions. */
11
+ const XML_TOKEN = /<!--[\s\S]*?-->|<!\[CDATA\[[\s\S]*?\]\]>|<\?[\s\S]*?\?>|<(?:"[^"]*"|'[^']*'|[^'">])*>/gu;
12
+ const readBody = (xml) => {
13
+ const root = parseXmlDocument(xml);
14
+ if (!root || !isWordElement(root, "document")) return null;
15
+ if (getNamespaceUri(root) !== NAMESPACES.w) return null;
16
+ const body = getChildElements(root).find((child) => isWordElement(child, "body"));
17
+ if (!body) return null;
18
+ const children = getChildElements(body);
19
+ const blocks = [];
20
+ const stack = [];
21
+ let bodyDepth = -1;
22
+ let bodyEnd = -1;
23
+ let start = -1;
24
+ for (const token of xml.matchAll(XML_TOKEN)) {
25
+ const tag = token[0];
26
+ if (tag.startsWith("<!") || tag.startsWith("<?")) continue;
27
+ const closing = tag.startsWith("</");
28
+ const name = tag.slice(closing ? 2 : 1).split(/[\s/>]/u).at(0);
29
+ if (!name) return null;
30
+ if (closing) {
31
+ if (stack.pop() !== name) return null;
32
+ if (stack.length === bodyDepth) {
33
+ const element = children[blocks.length];
34
+ if (!element || start < 0) return null;
35
+ blocks.push({
36
+ element,
37
+ start,
38
+ end: token.index + tag.length
39
+ });
40
+ start = -1;
41
+ }
42
+ if (name === body.name && stack.length === bodyDepth - 1) bodyEnd = token.index;
43
+ continue;
44
+ }
45
+ if (name === body.name && stack.length === 1) bodyDepth = 2;
46
+ if (stack.length === bodyDepth) {
47
+ if (children[blocks.length]?.name !== name) return null;
48
+ start = token.index;
49
+ if (tag.endsWith("/>")) {
50
+ const element = children[blocks.length];
51
+ if (!element) return null;
52
+ blocks.push({
53
+ element,
54
+ start,
55
+ end: token.index + tag.length
56
+ });
57
+ start = -1;
58
+ }
59
+ }
60
+ if (!tag.endsWith("/>")) stack.push(name);
61
+ }
62
+ if (stack.length > 0 || blocks.length !== children.length || bodyEnd < 0) return null;
63
+ return {
64
+ root,
65
+ blocks,
66
+ bodyEnd
67
+ };
68
+ };
69
+ function* descendants(element) {
70
+ yield element;
71
+ for (const child of getChildElements(element)) yield* descendants(child);
72
+ }
73
+ /** Cross-paragraph ranges require a wider edit contract than paragraph identity. */
74
+ const RANGE_ELEMENTS = /* @__PURE__ */ new Set([
75
+ "commentRangeStart",
76
+ "commentRangeEnd",
77
+ "commentReference",
78
+ "bookmarkStart",
79
+ "bookmarkEnd",
80
+ "permStart",
81
+ "permEnd",
82
+ "fldChar",
83
+ "moveFromRangeStart",
84
+ "moveFromRangeEnd",
85
+ "moveToRangeStart",
86
+ "moveToRangeEnd"
87
+ ]);
88
+ const DEPENDENT_ELEMENTS = /* @__PURE__ */ new Set([
89
+ "sectPr",
90
+ "footnoteReference",
91
+ "endnoteReference",
92
+ "drawing",
93
+ "pict",
94
+ "object"
95
+ ]);
96
+ const paragraphIsSafe = (element, touched) => {
97
+ for (const child of descendants(element)) {
98
+ if (!WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(child) ?? "")) continue;
99
+ const name = getLocalName(child.name);
100
+ if (touched && DEPENDENT_ELEMENTS.has(name)) return false;
101
+ }
102
+ return true;
103
+ };
104
+ const paragraphIds = (root) => {
105
+ const ids = /* @__PURE__ */ new Set();
106
+ for (const element of descendants(root)) {
107
+ if (!isWordElement(element, "p")) continue;
108
+ const id = paraIdAttribute(element);
109
+ if (!id || !/^[0-9a-f]{8}$/iu.test(id) || /^0+$/u.test(id) || ids.has(id.toUpperCase())) return null;
110
+ ids.add(id.toUpperCase());
111
+ }
112
+ return ids;
113
+ };
114
+ /** Bind fragment prefixes locally: the source root may use entirely different aliases. */
115
+ const paragraphFragment = ({ xml, block, bindings }) => {
116
+ const fragment = xml.slice(block.start, block.end);
117
+ const openEnd = fragment.indexOf(">");
118
+ const name = block.element.name ?? "w:p";
119
+ const selfClosing = fragment[openEnd - 1] === "/";
120
+ return serializePartElement({
121
+ partPath: "word/document.xml",
122
+ rootName: name,
123
+ rootAttributes: fragment.slice(name.length + 1, selfClosing ? openEnd - 1 : openEnd).trim(),
124
+ baselinePrefixes: [],
125
+ sourceBindings: bindings,
126
+ body: selfClosing ? "" : fragment.slice(openEnd + 1, fragment.lastIndexOf("</"))
127
+ });
128
+ };
129
+ /**
130
+ * Insert/delete direct body paragraphs between surviving paragraph/table anchors.
131
+ * Tables are opaque barriers: their modeled content and order must stay identical.
132
+ * Id-less sources need an explicit ensureParaIds ingest before structural editing.
133
+ */
134
+ const buildStructuralDocumentPatch = ({ originalXml, serializedXml, changedIds }) => {
135
+ const source = readBody(originalXml);
136
+ const current = readBody(serializedXml);
137
+ if (!source || !current) return null;
138
+ for (const root of [source.root, current.root]) for (const element of descendants(root)) if (WORDPROCESSINGML_NAMESPACE_URIS.has(getNamespaceUri(element) ?? "") && RANGE_ELEMENTS.has(getLocalName(element.name))) return null;
139
+ const allSourceIds = paragraphIds(source.root);
140
+ const allCurrentIds = paragraphIds(current.root);
141
+ if (!allSourceIds || !allCurrentIds) return null;
142
+ const changed = new Set([...changedIds].map((id) => id.toUpperCase()));
143
+ for (const { element } of [...source.blocks, ...current.blocks]) if (isWordElement(element, "p")) {
144
+ const id = paraIdAttribute(element)?.toUpperCase();
145
+ if (!id) return null;
146
+ const touched = changed.has(id) || !allSourceIds.has(id) || !allCurrentIds.has(id);
147
+ if (!paragraphIsSafe(element, touched)) return null;
148
+ } else if (!isWordElement(element, "tbl") && !isWordElement(element, "sectPr")) return null;
149
+ const sourceBody = parseDocumentBody(originalXml);
150
+ const currentBody = parseDocumentBody(serializedXml);
151
+ const barriers = (body) => ({
152
+ blocks: body.content.filter((block) => block.type !== "paragraph"),
153
+ finalSectionProperties: body.finalSectionProperties,
154
+ background: body.background
155
+ });
156
+ if (canonicalJson(barriers(sourceBody)) !== canonicalJson(barriers(currentBody))) return null;
157
+ const indexBlocks = (blocks) => {
158
+ let barrier = 0;
159
+ const indexed = /* @__PURE__ */ new Map();
160
+ for (const block of blocks) {
161
+ const key = isWordElement(block.element, "p") ? `p:${paraIdAttribute(block.element)?.toUpperCase()}` : `barrier:${barrier++}`;
162
+ indexed.set(key, block);
163
+ }
164
+ return indexed;
165
+ };
166
+ const before = indexBlocks(source.blocks);
167
+ const after = indexBlocks(current.blocks);
168
+ const survivingBefore = [...before.keys()].filter((key) => after.has(key));
169
+ const survivingAfter = [...after.keys()].filter((key) => before.has(key));
170
+ if (canonicalJson(survivingBefore) !== canonicalJson(survivingAfter)) return null;
171
+ const bindings = readRootNamespaceBindings(serializedXml);
172
+ const fragmentFor = (block) => paragraphFragment({
173
+ xml: serializedXml,
174
+ block,
175
+ bindings
176
+ });
177
+ const splices = [];
178
+ for (const [key, block] of before) if (!after.has(key)) splices.push({
179
+ start: block.start,
180
+ end: block.end,
181
+ newXml: ""
182
+ });
183
+ let pending = [];
184
+ for (const [key, block] of after) {
185
+ const original = before.get(key);
186
+ if (!original) {
187
+ const id = paraIdAttribute(block.element)?.toUpperCase();
188
+ if (!id || allSourceIds.has(id)) return null;
189
+ pending.push(fragmentFor(block));
190
+ continue;
191
+ }
192
+ const id = paraIdAttribute(block.element)?.toUpperCase();
193
+ const replacement = id !== void 0 && changed.has(id) ? fragmentFor(block) : originalXml.slice(original.start, original.end);
194
+ if (pending.length > 0 || replacement !== originalXml.slice(original.start, original.end)) splices.push({
195
+ start: original.start,
196
+ end: original.end,
197
+ newXml: pending.join("") + replacement
198
+ });
199
+ pending = [];
200
+ }
201
+ if (pending.length > 0) splices.push({
202
+ start: source.bodyEnd,
203
+ end: source.bodyEnd,
204
+ newXml: pending.join("")
205
+ });
206
+ return spliceXml(originalXml, splices);
207
+ };
208
+ //#endregion
209
+ export { buildStructuralDocumentPatch };
@@ -1,3 +1,4 @@
1
+ import { HistoryShortcutOwner } from "../prosemirror/extensions/core/HistoryExtension.js";
1
2
  import { EditorState, Transaction } from "prosemirror-state";
2
3
  //#region src/managers/editorShortcuts.d.ts
3
4
  declare function isMacPlatform(): boolean;
@@ -17,19 +18,31 @@ type EditorKeydownIntent = {
17
18
  } | {
18
19
  type: "none";
19
20
  };
21
+ /**
22
+ * An editor shortcut the host binds itself. The editor leaves its keys alone,
23
+ * so the host's binding (an IDE's undo stack, its own print command) is the
24
+ * only one that runs.
25
+ * - `"history"`: undo and redo (Mod-z, Mod-y, Mod-Shift-z).
26
+ * - `"print"`: Cmd/Ctrl+P.
27
+ */
28
+ type HostShortcut = "history" | "print";
29
+ /** Who answers the undo and redo keys when the host owns `hostShortcuts`. */
30
+ declare const historyShortcutOwner: (hostShortcuts: readonly HostShortcut[]) => HistoryShortcutOwner;
20
31
  type ClassifyEditorKeydownOptions = {
21
32
  /** Whether the platform uses Cmd (Mac) rather than Ctrl as the primary modifier. */
22
33
  isMac: boolean;
23
34
  /** Whether focus is in a non-editor input/textarea/contenteditable. */
24
35
  isInputLike: boolean;
36
+ /** Shortcuts the host binds itself; their keys map to `none`. */
37
+ hostShortcuts: readonly HostShortcut[];
25
38
  };
26
39
  /**
27
40
  * Map a keydown to an editor intent:
28
41
  * - Cmd/Ctrl+F or Cmd/Ctrl+H → open find
29
- * - Cmd/Ctrl+P (no auto-repeat) → custom print
42
+ * - Cmd/Ctrl+P (no auto-repeat) → custom print, unless the host owns print
30
43
  * - Delete/Backspace (no modifiers, focus not in an input) → delete selected table
31
44
  */
32
- declare function classifyEditorKeydown(e: KeyboardEvent, { isMac, isInputLike }: ClassifyEditorKeydownOptions): EditorKeydownIntent;
45
+ declare function classifyEditorKeydown(e: Pick<KeyboardEvent, "key" | "metaKey" | "ctrlKey" | "shiftKey" | "altKey" | "repeat">, { isMac, isInputLike, hostShortcuts }: ClassifyEditorKeydownOptions): EditorKeydownIntent;
33
46
  /**
34
47
  * Which key presses the editor's page-level shortcuts answer.
35
48
  * - `"document"`: every press on the page.
@@ -69,4 +82,4 @@ declare function isKeydownInShortcutScope(event: {
69
82
  */
70
83
  declare function deleteSelectedTable(state: EditorState, dispatch: (tr: Transaction) => void): boolean;
71
84
  //#endregion
72
- export { ClassifyEditorKeydownOptions, EditorKeydownIntent, IsKeydownInShortcutScopeOptions, KeyboardShortcutScope, ShortcutScopeRoot, classifyEditorKeydown, deleteSelectedTable, isFocusInInputLike, isKeydownInShortcutScope, isMacPlatform };
85
+ export { ClassifyEditorKeydownOptions, EditorKeydownIntent, HostShortcut, IsKeydownInShortcutScopeOptions, KeyboardShortcutScope, ShortcutScopeRoot, classifyEditorKeydown, deleteSelectedTable, historyShortcutOwner, isFocusInInputLike, isKeydownInShortcutScope, isMacPlatform };
@@ -16,19 +16,21 @@ function isFocusInInputLike(target, editorDom) {
16
16
  if (target.isContentEditable && target !== editorDom) return true;
17
17
  return false;
18
18
  }
19
+ /** Who answers the undo and redo keys when the host owns `hostShortcuts`. */
20
+ const historyShortcutOwner = (hostShortcuts) => hostShortcuts.includes("history") ? "host" : "editor";
19
21
  /**
20
22
  * Map a keydown to an editor intent:
21
23
  * - Cmd/Ctrl+F or Cmd/Ctrl+H → open find
22
- * - Cmd/Ctrl+P (no auto-repeat) → custom print
24
+ * - Cmd/Ctrl+P (no auto-repeat) → custom print, unless the host owns print
23
25
  * - Delete/Backspace (no modifiers, focus not in an input) → delete selected table
24
26
  */
25
- function classifyEditorKeydown(e, { isMac, isInputLike }) {
27
+ function classifyEditorKeydown(e, { isMac, isInputLike, hostShortcuts }) {
26
28
  const cmdOrCtrl = isMac ? e.metaKey : e.ctrlKey;
27
29
  if (!cmdOrCtrl && !e.shiftKey && !e.altKey && (e.key === "Delete" || e.key === "Backspace") && !isInputLike) return { type: "deleteSelectedTable" };
28
30
  if (cmdOrCtrl && !e.shiftKey && !e.altKey) {
29
31
  const key = e.key.toLowerCase();
30
32
  if (key === "f" || key === "h") return { type: "openFind" };
31
- if (key === "p" && !e.repeat) return { type: "print" };
33
+ if (key === "p" && !e.repeat && !hostShortcuts.includes("print")) return { type: "print" };
32
34
  }
33
35
  return { type: "none" };
34
36
  }
@@ -76,4 +78,4 @@ function deleteSelectedTable(state, dispatch) {
76
78
  return false;
77
79
  }
78
80
  //#endregion
79
- export { classifyEditorKeydown, deleteSelectedTable, isFocusInInputLike, isKeydownInShortcutScope, isMacPlatform };
81
+ export { classifyEditorKeydown, deleteSelectedTable, historyShortcutOwner, isFocusInInputLike, isKeydownInShortcutScope, isMacPlatform };
@@ -1,4 +1,5 @@
1
1
  import { AnyExtension } from "./types.js";
2
+ import { HistoryShortcutOwner } from "./core/HistoryExtension.js";
2
3
  import { SelectionChangeCallback } from "../plugins/selectionTracker.js";
3
4
  //#region src/prosemirror/extensions/StarterKit.d.ts
4
5
  type StarterKitOptions = {
@@ -8,6 +9,8 @@ type StarterKitOptions = {
8
9
  historyDepth?: number;
9
10
  /** History new group delay (default: 500) */
10
11
  historyNewGroupDelay?: number;
12
+ /** Who answers the undo and redo keys (default: `"editor"`). */
13
+ historyShortcuts?: HistoryShortcutOwner;
11
14
  /** Selection change callback */
12
15
  onSelectionChange?: SelectionChangeCallback;
13
16
  };
@@ -61,7 +61,8 @@ function createStarterKit(options = {}) {
61
61
  add("paragraph", ParagraphExtension());
62
62
  add("history", HistoryExtension({
63
63
  ...options.historyDepth !== void 0 ? { depth: options.historyDepth } : {},
64
- ...options.historyNewGroupDelay !== void 0 ? { newGroupDelay: options.historyNewGroupDelay } : {}
64
+ ...options.historyNewGroupDelay !== void 0 ? { newGroupDelay: options.historyNewGroupDelay } : {},
65
+ ...options.historyShortcuts !== void 0 ? { shortcuts: options.historyShortcuts } : {}
65
66
  }));
66
67
  for (const name of MARK_NESTING_ORDER) add(name, MARK_EXTENSIONS[name]());
67
68
  add("bookmarkBoundary", BookmarkBoundaryExtension({ getInternalClipboardToken }));
@@ -3,10 +3,18 @@ import { Extension } from "../types.js";
3
3
  /**
4
4
  * History Extension — undo/redo via prosemirror-history
5
5
  */
6
+ /**
7
+ * Who answers the undo and redo keys (Mod-z, Mod-y, Mod-Shift-z).
8
+ * - `"editor"`: the extension binds them. Default.
9
+ * - `"host"`: nothing binds them. The host keeps its own undo stack and runs
10
+ * the `undo` / `redo` commands itself, so one press never undoes twice.
11
+ */
12
+ type HistoryShortcutOwner = "editor" | "host";
6
13
  type HistoryOptions = {
7
14
  depth: number;
8
15
  newGroupDelay: number;
16
+ shortcuts: HistoryShortcutOwner;
9
17
  };
10
18
  declare const HistoryExtension: (options?: Partial<HistoryOptions> | undefined) => Extension;
11
19
  //#endregion
12
- export { HistoryExtension };
20
+ export { HistoryExtension, HistoryShortcutOwner };
@@ -4,7 +4,8 @@ const HistoryExtension = createExtension({
4
4
  name: "history",
5
5
  defaultOptions: {
6
6
  depth: 100,
7
- newGroupDelay: 500
7
+ newGroupDelay: 500,
8
+ shortcuts: "editor"
8
9
  },
9
10
  onSchemaReady(_ctx, options) {
10
11
  return {
@@ -16,11 +17,11 @@ const HistoryExtension = createExtension({
16
17
  undo: () => undo,
17
18
  redo: () => redo
18
19
  },
19
- keyboardShortcuts: {
20
+ ...options.shortcuts === "editor" && { keyboardShortcuts: {
20
21
  "Mod-z": undo,
21
22
  "Mod-y": redo,
22
23
  "Mod-Shift-z": redo
23
- }
24
+ } }
24
25
  };
25
26
  }
26
27
  });
@@ -3,15 +3,15 @@ import { RemovedSectionReference, TrackedSectionEndpointRemoval } from "../../..
3
3
  import { EditorState, PluginKey, Transaction } from "prosemirror-state";
4
4
  import { Node } from "prosemirror-model";
5
5
  //#region src/prosemirror/extensions/features/ParagraphChangeTrackerExtension.d.ts
6
- declare const paragraphChangeTrackerKey: PluginKey<ParagraphChangeTrackerState>;
6
+ declare const paragraphChangeTrackerKey: PluginKey<InternalParagraphChangeTrackerState>;
7
7
  type ParagraphChangeTrackerState = {
8
8
  /** Set of paraIds that were modified since last clear */
9
9
  changedParaIds: Set<string>;
10
- /** Whether paragraphs were added or deleted (structural change) */
10
+ /** Whether paragraph order, membership, or block structure changed. */
11
11
  structuralChange: boolean;
12
- /** Whether any edited paragraph lacked a paraId */
12
+ /** Whether edited paragraphs still lack IDs or source identity was unavailable. */
13
13
  hasUntrackedChanges: boolean;
14
- /** Cached paragraph count to avoid full doc traversal on every transaction */
14
+ /** Paragraph count in the current tracked document. */
15
15
  paragraphCount: number;
16
16
  /** Cached section-endpoint count used to invalidate stale save authorization. */
17
17
  sectionEndpointCount: number;
@@ -20,6 +20,13 @@ type ParagraphChangeTrackerState = {
20
20
  /** Exact tracked-resolution transition that may reduce the saved section count. */
21
21
  sectionEndpointRemoval: TrackedSectionEndpointRemoval | null;
22
22
  };
23
+ type InternalParagraphChangeTrackerState = ParagraphChangeTrackerState & {
24
+ /** Edited paragraph positions retained across allocator-only transactions. */
25
+ affectedParagraphPositions: Set<number>;
26
+ /** Editing an already unidentified source paragraph cannot become selective. */
27
+ hasUntrackedSourceChanges: boolean;
28
+ blockStructureFingerprint: string;
29
+ };
23
30
  /**
24
31
  * Get the change tracker state from an EditorState
25
32
  */
@@ -29,7 +36,7 @@ declare function getChangeTrackerState(state: EditorState): ParagraphChangeTrack
29
36
  */
30
37
  declare function getChangedParagraphIds(state: EditorState): Set<string>;
31
38
  /**
32
- * Check if structural changes (paragraph add/delete) occurred
39
+ * Check if paragraph membership, order, or block structure changed
33
40
  */
34
41
  declare function hasStructuralChanges(state: EditorState): boolean;
35
42
  /**
@@ -18,9 +18,20 @@ const isTrackedSectionEndpointRemovalMeta = (value) => typeof value === "object"
18
18
  function countDocumentStructure(doc) {
19
19
  let paragraphs = 0;
20
20
  let sectionEndpoints = 0;
21
+ const blockRecords = [];
21
22
  const endpointRecords = [];
22
23
  const visit = (parent, parentPath) => {
23
24
  parent.forEach((node, _offset, index) => {
25
+ const path = parentPath.length === 0 ? `${index}` : `${parentPath}.${index}`;
26
+ blockRecords.push(node.type.name === "paragraph" ? [
27
+ path,
28
+ node.type.name,
29
+ node.attrs["paraId"] ?? null
30
+ ] : [
31
+ path,
32
+ node.type.name,
33
+ canonicalJson(node.attrs)
34
+ ]);
24
35
  if (node.type.name === "paragraph") {
25
36
  paragraphs++;
26
37
  const sectionProperties = sectionPropertiesOf(node);
@@ -34,34 +45,36 @@ function countDocumentStructure(doc) {
34
45
  }
35
46
  return;
36
47
  }
37
- if (node.childCount > 0) {
38
- const path = parentPath.length === 0 ? `${index}` : `${parentPath}.${index}`;
39
- visit(node, path);
40
- }
48
+ if (node.childCount > 0) visit(node, path);
41
49
  });
42
50
  };
43
51
  visit(doc, "");
44
52
  return {
53
+ blockStructureFingerprint: JSON.stringify(blockRecords),
45
54
  paragraphs,
46
55
  sectionEndpoints,
47
56
  sectionEndpointFingerprint: canonicalJson(endpointRecords)
48
57
  };
49
58
  }
59
+ const isUsableParaId = (value) => typeof value === "string" && value.length > 0 && value !== "00000000";
50
60
  /**
51
61
  * Collect paraIds of all paragraphs that overlap with the given range
52
62
  */
53
63
  function collectAffectedParaIds(doc, from, to) {
54
64
  const ids = /* @__PURE__ */ new Set();
65
+ const positions = /* @__PURE__ */ new Set();
55
66
  let hasUntracked = false;
56
- doc.nodesBetween(from, to, (node) => {
67
+ doc.nodesBetween(from, to, (node, pos) => {
57
68
  if (node.type.name === "paragraph") {
69
+ positions.add(pos);
58
70
  const paraId = node.attrs["paraId"];
59
- if (paraId) ids.add(paraId);
71
+ if (isUsableParaId(paraId)) ids.add(paraId);
60
72
  else hasUntracked = true;
61
73
  }
62
74
  });
63
75
  return {
64
76
  ids,
77
+ positions,
65
78
  hasUntracked
66
79
  };
67
80
  }
@@ -75,28 +88,40 @@ function collectAffectedParaIdsFromMarkLikeStep(doc, from, to) {
75
88
  const hi = Math.max(from, to);
76
89
  const primary = collectAffectedParaIds(doc, lo, hi > lo ? hi : lo + 1);
77
90
  if (primary.ids.size > 0 || primary.hasUntracked) return primary;
78
- try {
79
- const $p = doc.resolve(lo);
80
- for (let d = $p.depth; d >= 0; d--) {
81
- const n = $p.node(d);
82
- if (n.type.name === "paragraph") {
83
- const paraId = n.attrs["paraId"];
84
- if (paraId) return {
85
- ids: /* @__PURE__ */ new Set([paraId]),
86
- hasUntracked: false
87
- };
88
- return {
89
- ids: /* @__PURE__ */ new Set(),
90
- hasUntracked: true
91
- };
92
- }
91
+ const $p = doc.resolve(lo);
92
+ for (let depth = $p.depth; depth > 0; depth--) {
93
+ const node = $p.node(depth);
94
+ if (node.type.name === "paragraph") {
95
+ const paraId = node.attrs["paraId"];
96
+ return {
97
+ ids: new Set(isUsableParaId(paraId) ? [paraId] : []),
98
+ positions: /* @__PURE__ */ new Set([$p.before(depth)]),
99
+ hasUntracked: !isUsableParaId(paraId)
100
+ };
93
101
  }
94
- } catch {}
102
+ }
95
103
  return {
96
104
  ids: /* @__PURE__ */ new Set(),
105
+ positions: /* @__PURE__ */ new Set(),
97
106
  hasUntracked: false
98
107
  };
99
108
  }
109
+ /** Map an edited paragraph's interior so markup replacement retains its ownership. */
110
+ const mapParagraphPosition = (tr, position) => {
111
+ const mapped = tr.mapping.mapResult(position + 1, 1);
112
+ if (mapped.deletedAcross) return null;
113
+ const $position = tr.doc.resolve(mapped.pos);
114
+ for (let depth = $position.depth; depth > 0; depth--) if ($position.node(depth).type.name === "paragraph") return $position.before(depth);
115
+ return null;
116
+ };
117
+ const mapAffectedParagraphPositions = (tr, positions) => {
118
+ const mappedPositions = /* @__PURE__ */ new Set();
119
+ for (const position of positions) {
120
+ const mapped = mapParagraphPosition(tr, position);
121
+ if (mapped !== null) mappedPositions.add(mapped);
122
+ }
123
+ return mappedPositions;
124
+ };
100
125
  function mapStepPosition(remap, pos, assoc) {
101
126
  return remap.map(pos, assoc);
102
127
  }
@@ -107,6 +132,9 @@ function createParagraphChangeTrackerPlugin() {
107
132
  init(_config, state) {
108
133
  const counts = countDocumentStructure(state.doc);
109
134
  return {
135
+ affectedParagraphPositions: /* @__PURE__ */ new Set(),
136
+ hasUntrackedSourceChanges: false,
137
+ blockStructureFingerprint: counts.blockStructureFingerprint,
110
138
  changedParaIds: /* @__PURE__ */ new Set(),
111
139
  structuralChange: false,
112
140
  hasUntrackedChanges: false,
@@ -123,6 +151,9 @@ function createParagraphChangeTrackerPlugin() {
123
151
  if (meta === CLEAR_META) {
124
152
  const counts = countDocumentStructure(tr.doc);
125
153
  return {
154
+ affectedParagraphPositions: /* @__PURE__ */ new Set(),
155
+ hasUntrackedSourceChanges: false,
156
+ blockStructureFingerprint: counts.blockStructureFingerprint,
126
157
  changedParaIds: /* @__PURE__ */ new Set(),
127
158
  structuralChange: false,
128
159
  hasUntrackedChanges: false,
@@ -133,13 +164,22 @@ function createParagraphChangeTrackerPlugin() {
133
164
  };
134
165
  }
135
166
  if (meta === IGNORE_META) {
136
- const counts = tr.docChanged ? countDocumentStructure(tr.doc) : {
137
- paragraphs: prevState.paragraphCount,
138
- sectionEndpoints: prevState.sectionEndpointCount,
139
- sectionEndpointFingerprint: prevState.sectionEndpointFingerprint
140
- };
167
+ if (!tr.docChanged) return prevState;
168
+ const counts = countDocumentStructure(tr.doc);
169
+ const affectedParagraphPositions = mapAffectedParagraphPositions(tr, prevState.affectedParagraphPositions);
170
+ const changedParaIds = /* @__PURE__ */ new Set();
171
+ let unresolvedChanges = prevState.hasUntrackedSourceChanges;
172
+ for (const position of affectedParagraphPositions) {
173
+ const paraId = tr.doc.nodeAt(position)?.attrs["paraId"];
174
+ if (isUsableParaId(paraId)) changedParaIds.add(paraId);
175
+ else unresolvedChanges = true;
176
+ }
141
177
  return {
142
178
  ...prevState,
179
+ affectedParagraphPositions,
180
+ changedParaIds,
181
+ hasUntrackedChanges: unresolvedChanges,
182
+ blockStructureFingerprint: counts.blockStructureFingerprint,
143
183
  paragraphCount: counts.paragraphs,
144
184
  sectionEndpointCount: counts.sectionEndpoints,
145
185
  sectionEndpointRemoval: counts.sectionEndpointFingerprint === prevState.sectionEndpointFingerprint ? prevState.sectionEndpointRemoval : null,
@@ -162,6 +202,9 @@ function createParagraphChangeTrackerPlugin() {
162
202
  };
163
203
  } else sectionEndpointRemoval = null;
164
204
  const newState = {
205
+ affectedParagraphPositions: mapAffectedParagraphPositions(tr, prevState.affectedParagraphPositions),
206
+ hasUntrackedSourceChanges: prevState.hasUntrackedSourceChanges,
207
+ blockStructureFingerprint: counts.blockStructureFingerprint,
165
208
  changedParaIds: new Set(prevState.changedParaIds),
166
209
  structuralChange: prevState.structuralChange || meta === STRUCTURAL_META,
167
210
  hasUntrackedChanges: prevState.hasUntrackedChanges,
@@ -176,12 +219,13 @@ function createParagraphChangeTrackerPlugin() {
176
219
  const from = mapStepPosition(remap, range.from, 1);
177
220
  const to = mapStepPosition(remap, range.to, -1);
178
221
  if (to <= from) continue;
179
- const { ids, hasUntracked } = collectAffectedParaIdsFromMarkLikeStep(tr.doc, from, to);
222
+ const { ids, positions, hasUntracked } = collectAffectedParaIdsFromMarkLikeStep(tr.doc, from, to);
223
+ for (const position of positions) newState.affectedParagraphPositions.add(position);
180
224
  for (const id of ids) newState.changedParaIds.add(id);
181
225
  if (hasUntracked) newState.hasUntrackedChanges = true;
182
226
  }
183
227
  }
184
- if (prevState.paragraphCount !== newCount) newState.structuralChange = true;
228
+ if (prevState.blockStructureFingerprint !== counts.blockStructureFingerprint) newState.structuralChange = true;
185
229
  for (let stepIndex = 0; stepIndex < tr.steps.length; stepIndex++) {
186
230
  const step = tr.steps[stepIndex];
187
231
  const remap = tr.mapping.slice(stepIndex + 1);
@@ -189,7 +233,8 @@ function createParagraphChangeTrackerPlugin() {
189
233
  const from = mapStepPosition(remap, step.from, 1);
190
234
  const to = mapStepPosition(remap, step.to, -1);
191
235
  if (to <= from) continue;
192
- const { ids, hasUntracked } = collectAffectedParaIdsFromMarkLikeStep(tr.doc, from, to);
236
+ const { ids, positions, hasUntracked } = collectAffectedParaIdsFromMarkLikeStep(tr.doc, from, to);
237
+ for (const position of positions) newState.affectedParagraphPositions.add(position);
193
238
  for (const id of ids) newState.changedParaIds.add(id);
194
239
  if (hasUntracked) newState.hasUntrackedChanges = true;
195
240
  continue;
@@ -199,7 +244,8 @@ function createParagraphChangeTrackerPlugin() {
199
244
  const node = tr.doc.nodeAt(pos);
200
245
  if (!node) continue;
201
246
  const end = pos + node.nodeSize;
202
- const { ids, hasUntracked } = collectAffectedParaIds(tr.doc, pos, end);
247
+ const { ids, positions, hasUntracked } = collectAffectedParaIds(tr.doc, pos, end);
248
+ for (const position of positions) newState.affectedParagraphPositions.add(position);
203
249
  for (const id of ids) newState.changedParaIds.add(id);
204
250
  if (hasUntracked) newState.hasUntrackedChanges = true;
205
251
  continue;
@@ -208,11 +254,23 @@ function createParagraphChangeTrackerPlugin() {
208
254
  const from = mapStepPosition(remap, newStart, 1);
209
255
  const to = mapStepPosition(remap, newEnd, -1);
210
256
  if (to < from) return;
211
- const { ids, hasUntracked } = collectAffectedParaIds(tr.doc, from, to);
257
+ const { ids, positions, hasUntracked } = collectAffectedParaIds(tr.doc, from, to);
258
+ for (const position of positions) newState.affectedParagraphPositions.add(position);
212
259
  for (const id of ids) newState.changedParaIds.add(id);
213
260
  if (hasUntracked) newState.hasUntrackedChanges = true;
214
261
  });
215
262
  }
263
+ if (newState.hasUntrackedChanges || prevState.blockStructureFingerprint !== counts.blockStructureFingerprint) tr.before.descendants((node, position) => {
264
+ if (node.type.name !== "paragraph") return true;
265
+ if (!isUsableParaId(node.attrs["paraId"])) {
266
+ const mapped = mapParagraphPosition(tr, position);
267
+ if (mapped === null || newState.affectedParagraphPositions.has(mapped)) {
268
+ newState.hasUntrackedSourceChanges = true;
269
+ newState.hasUntrackedChanges = true;
270
+ }
271
+ }
272
+ return false;
273
+ });
216
274
  return newState;
217
275
  }
218
276
  }
@@ -231,7 +289,7 @@ function getChangedParagraphIds(state) {
231
289
  return getChangeTrackerState(state)?.changedParaIds ?? /* @__PURE__ */ new Set();
232
290
  }
233
291
  /**
234
- * Check if structural changes (paragraph add/delete) occurred
292
+ * Check if paragraph membership, order, or block structure changed
235
293
  */
236
294
  function hasStructuralChanges(state) {
237
295
  return getChangeTrackerState(state)?.structuralChange ?? false;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@stll/folio-core",
3
- "version": "0.50.0",
3
+ "version": "0.51.0",
4
4
  "description": "Headless, framework-neutral core of folio: the OOXML (.docx) parser, document model, ProseMirror integration, and page-layout engine. No React.",
5
5
  "keywords": [
6
6
  "document-model",