codex-overleaf-link 2.3.2 → 2.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,31 +1,54 @@
1
1
  'use strict';
2
2
 
3
+
3
4
  function computeTextPatches(oldText, newText) {
4
- const oldValue = String(oldText ?? '');
5
- const newValue = String(newText ?? '');
6
- if (oldValue === newValue) {
5
+ if (oldText === newText) {
7
6
  return [];
8
7
  }
9
8
 
10
- const groups = computeLineAnchoredChangeGroups(oldValue, newValue);
11
- if (!groups.length) {
12
- return [computeSingleTextPatch(oldValue, newValue)];
9
+ const groups = computeLineAnchoredChangeGroups(oldText, newText);
10
+ const naturalPatches = groups.flatMap(group => computeNaturalGroupPatches(group));
11
+ if (isValidNaturalPatchSet(oldText, newText, naturalPatches)) {
12
+ return naturalPatches;
13
13
  }
14
14
 
15
- const patches = [];
16
- for (const group of groups) {
17
- patches.push(...computeNaturalGroupPatches(group));
15
+ // Semantic fallback stays scoped to the line-anchored windows. A budget
16
+ // overflow must never turn sparse file-wide edits into one first-to-last
17
+ // replacement.
18
+ return groups.flatMap(singleGroupPatch);
19
+ }
20
+
21
+ function isValidNaturalPatchSet(oldText, newText, patches) {
22
+ if (!Array.isArray(patches) || patches.length === 0) {
23
+ return false;
18
24
  }
19
- return patches.length ? patches : [computeSingleTextPatch(oldValue, newValue)];
25
+
26
+ const ordered = [...patches].sort((left, right) => left.from - right.from || left.to - right.to);
27
+ let previousEnd = 0;
28
+ for (const patch of ordered) {
29
+ if (
30
+ !patch
31
+ || !Number.isInteger(patch.from)
32
+ || !Number.isInteger(patch.to)
33
+ || patch.from < previousEnd
34
+ || patch.to < patch.from
35
+ || oldText.slice(patch.from, patch.to) !== patch.expected
36
+ ) {
37
+ return false;
38
+ }
39
+ previousEnd = patch.to;
40
+ }
41
+
42
+ let applied = oldText;
43
+ for (let index = ordered.length - 1; index >= 0; index -= 1) {
44
+ const patch = ordered[index];
45
+ applied = applied.slice(0, patch.from) + patch.insert + applied.slice(patch.to);
46
+ }
47
+ return applied === newText;
20
48
  }
21
49
 
22
- // Computes the natural-granularity patches for one changed group (spec
23
- // "Algorithm sketch"). Builds token patches and metrics, classifies the group,
24
- // then dispatches to the matching builder. `singleGroupPatch` already returns
25
- // a one-element array; `computeParagraphPatches` / `computeSentencePatches`
26
- // return an array or `null`, so a null/empty result falls back to a single
27
- // group patch. `coalesceTokenPatches` always returns a non-empty array when it
28
- // receives non-empty token patches.
50
+
51
+
29
52
  function computeNaturalGroupPatches(group) {
30
53
  const tokenPatches = computeTokenAnchoredPatches(
31
54
  group.oldText,
@@ -33,27 +56,30 @@ function computeNaturalGroupPatches(group) {
33
56
  group.oldStart
34
57
  );
35
58
  const metrics = computeGroupMetrics(group, tokenPatches);
36
- const { type } = classifyChangedGroup(group, tokenPatches, metrics);
59
+ const classification = classifyChangedGroup(group, tokenPatches, metrics);
37
60
 
38
- if (type === 'annotated_block') {
61
+ if (classification.type === 'annotated_block') {
39
62
  return singleGroupPatch(group);
40
63
  }
41
- if (type === 'paragraph_rewrite') {
64
+ if (classification.type === 'paragraph_rewrite') {
42
65
  const paragraphPatches = computeParagraphPatches(group);
43
- return (paragraphPatches && paragraphPatches.length)
66
+ return paragraphPatches !== null
44
67
  ? paragraphPatches
45
68
  : singleGroupPatch(group);
46
69
  }
47
- if (type === 'sentence_rewrite') {
70
+ if (classification.type === 'sentence_rewrite') {
48
71
  const sentencePatches = computeSentencePatches(group, tokenPatches);
49
- return (sentencePatches && sentencePatches.length)
72
+ return sentencePatches !== null
50
73
  ? sentencePatches
51
- : singleGroupPatch(group);
74
+ : coalesceTokenPatches(group, tokenPatches);
52
75
  }
53
- if (type === 'small_edit' && tokenPatches && tokenPatches.length) {
76
+ if (classification.type === 'small_edit') {
54
77
  return coalesceTokenPatches(group, tokenPatches);
55
78
  }
56
- return singleGroupPatch(group);
79
+
80
+ return tokenPatches.length > 0
81
+ ? coalesceTokenPatches(group, tokenPatches)
82
+ : singleGroupPatch(group);
57
83
  }
58
84
 
59
85
  function computeSingleTextPatch(oldValue, newValue, offset = 0) {
@@ -82,149 +108,586 @@ function computeSingleTextPatch(oldValue, newValue, offset = 0) {
82
108
  };
83
109
  }
84
110
 
111
+
85
112
  function computeLineAnchoredChangeGroups(oldValue, newValue) {
113
+ if (oldValue === newValue) {
114
+ return [];
115
+ }
116
+
86
117
  const oldParts = splitTextParts(oldValue);
87
118
  const newParts = splitTextParts(newValue);
119
+ const oldOffsets = computeNaturalPartOffsets(oldParts);
120
+ const groups = [];
121
+
122
+ collectNaturalLineGroups({
123
+ oldParts,
124
+ newParts,
125
+ oldOffsets,
126
+ oldFrom: 0,
127
+ oldTo: oldParts.length,
128
+ newFrom: 0,
129
+ newTo: newParts.length,
130
+ groups,
131
+ depth: 0
132
+ });
133
+
134
+ if (groups.length === 0) {
135
+ groups.push({
136
+ oldStart: 0,
137
+ oldText: oldValue,
138
+ newText: newValue
139
+ });
140
+ }
141
+ return groups;
142
+ }
143
+
144
+ function computeNaturalPartOffsets(parts) {
145
+ const offsets = new Array(parts.length + 1);
146
+ offsets[0] = 0;
147
+ for (let index = 0; index < parts.length; index += 1) {
148
+ offsets[index + 1] = offsets[index] + parts[index].length;
149
+ }
150
+ return offsets;
151
+ }
152
+
153
+ function pushNaturalLineGroup(state, oldFrom, oldTo, newFrom, newTo) {
154
+ const oldText = state.oldParts.slice(oldFrom, oldTo).join('');
155
+ const newText = state.newParts.slice(newFrom, newTo).join('');
156
+ if (oldText === newText) {
157
+ return;
158
+ }
159
+ state.groups.push({
160
+ oldStart: state.oldOffsets[oldFrom],
161
+ oldText,
162
+ newText
163
+ });
164
+ }
165
+
166
+ function collectNaturalLineGroups(state) {
167
+ let { oldFrom, oldTo, newFrom, newTo } = state;
168
+
169
+ while (
170
+ oldFrom < oldTo
171
+ && newFrom < newTo
172
+ && state.oldParts[oldFrom] === state.newParts[newFrom]
173
+ ) {
174
+ oldFrom += 1;
175
+ newFrom += 1;
176
+ }
177
+ while (
178
+ oldFrom < oldTo
179
+ && newFrom < newTo
180
+ && state.oldParts[oldTo - 1] === state.newParts[newTo - 1]
181
+ ) {
182
+ oldTo -= 1;
183
+ newTo -= 1;
184
+ }
185
+
186
+ if (oldFrom === oldTo || newFrom === newTo) {
187
+ pushNaturalLineGroup(state, oldFrom, oldTo, newFrom, newTo);
188
+ return;
189
+ }
190
+
191
+ const oldCount = oldTo - oldFrom;
192
+ const newCount = newTo - newFrom;
88
193
  const MAX_PARTS = 5000;
89
- const MAX_PRODUCT = 4000000;
194
+ const MAX_PRODUCT = 4_000_000;
195
+ if (
196
+ oldCount <= MAX_PARTS
197
+ && newCount <= MAX_PARTS
198
+ && oldCount * newCount <= MAX_PRODUCT
199
+ ) {
200
+ appendNaturalExactLineGroups(state, oldFrom, oldTo, newFrom, newTo);
201
+ return;
202
+ }
203
+
204
+ let anchors = discoverNaturalLineAnchors(
205
+ state.oldParts,
206
+ state.newParts,
207
+ oldFrom,
208
+ oldTo,
209
+ newFrom,
210
+ newTo,
211
+ 4
212
+ );
213
+ if (anchors.length === 0) {
214
+ anchors = discoverNaturalLineAnchors(
215
+ state.oldParts,
216
+ state.newParts,
217
+ oldFrom,
218
+ oldTo,
219
+ newFrom,
220
+ newTo,
221
+ 8
222
+ );
223
+ }
224
+
225
+ if (anchors.length > 0 && state.depth < 10) {
226
+ let oldCursor = oldFrom;
227
+ let newCursor = newFrom;
228
+ for (const anchor of anchors) {
229
+ collectNaturalLineGroups({
230
+ ...state,
231
+ oldFrom: oldCursor,
232
+ oldTo: anchor.oldIndex,
233
+ newFrom: newCursor,
234
+ newTo: anchor.newIndex,
235
+ depth: state.depth + 1
236
+ });
237
+ oldCursor = anchor.oldIndex + 1;
238
+ newCursor = anchor.newIndex + 1;
239
+ }
240
+ collectNaturalLineGroups({
241
+ ...state,
242
+ oldFrom: oldCursor,
243
+ oldTo,
244
+ newFrom: newCursor,
245
+ newTo,
246
+ depth: state.depth + 1
247
+ });
248
+ return;
249
+ }
90
250
 
91
251
  if (
92
- oldParts.length === 0
93
- || newParts.length === 0
94
- || oldParts.length > MAX_PARTS
95
- || newParts.length > MAX_PARTS
96
- || oldParts.length * newParts.length > MAX_PRODUCT
252
+ oldCount === newCount
253
+ && appendNaturalPositionalLineGroups(state, oldFrom, oldTo, newFrom, newTo)
97
254
  ) {
98
- return [];
255
+ return;
99
256
  }
100
257
 
101
- const edits = computePartEdits(oldParts, newParts);
102
- const groups = [];
103
- let oldOffset = 0;
104
- let newOffset = 0;
105
- let group = null;
106
-
107
- for (const edit of edits) {
108
- if (edit.type === 'equal') {
109
- flushGroup();
110
- oldOffset += oldParts[edit.oldIndex].length;
111
- newOffset += newParts[edit.newIndex].length;
258
+ pushNaturalLineGroup(state, oldFrom, oldTo, newFrom, newTo);
259
+ }
260
+
261
+ function appendNaturalExactLineGroups(state, oldFrom, oldTo, newFrom, newTo) {
262
+ const matches = computeNaturalLcsMatches(
263
+ state.oldParts.slice(oldFrom, oldTo),
264
+ state.newParts.slice(newFrom, newTo)
265
+ );
266
+ let oldCursor = oldFrom;
267
+ let newCursor = newFrom;
268
+ for (const match of matches) {
269
+ const oldIndex = oldFrom + match.oldIndex;
270
+ const newIndex = newFrom + match.newIndex;
271
+ pushNaturalLineGroup(state, oldCursor, oldIndex, newCursor, newIndex);
272
+ oldCursor = oldIndex + 1;
273
+ newCursor = newIndex + 1;
274
+ }
275
+ pushNaturalLineGroup(state, oldCursor, oldTo, newCursor, newTo);
276
+ }
277
+
278
+ function computeNaturalLcsMatches(oldItems, newItems) {
279
+ const oldCount = oldItems.length;
280
+ const newCount = newItems.length;
281
+ const width = newCount + 1;
282
+ const table = new Uint32Array((oldCount + 1) * width);
283
+
284
+ for (let oldIndex = oldCount - 1; oldIndex >= 0; oldIndex -= 1) {
285
+ for (let newIndex = newCount - 1; newIndex >= 0; newIndex -= 1) {
286
+ const cell = oldIndex * width + newIndex;
287
+ table[cell] = oldItems[oldIndex] === newItems[newIndex]
288
+ ? table[(oldIndex + 1) * width + newIndex + 1] + 1
289
+ : Math.max(
290
+ table[(oldIndex + 1) * width + newIndex],
291
+ table[oldIndex * width + newIndex + 1]
292
+ );
293
+ }
294
+ }
295
+
296
+ const matches = [];
297
+ let oldIndex = 0;
298
+ let newIndex = 0;
299
+ while (oldIndex < oldCount && newIndex < newCount) {
300
+ if (oldItems[oldIndex] === newItems[newIndex]) {
301
+ matches.push({ oldIndex, newIndex });
302
+ oldIndex += 1;
303
+ newIndex += 1;
304
+ } else if (
305
+ table[(oldIndex + 1) * width + newIndex]
306
+ >= table[oldIndex * width + newIndex + 1]
307
+ ) {
308
+ oldIndex += 1;
309
+ } else {
310
+ newIndex += 1;
311
+ }
312
+ }
313
+ return matches;
314
+ }
315
+
316
+ function discoverNaturalLineAnchors(
317
+ oldParts,
318
+ newParts,
319
+ oldFrom,
320
+ oldTo,
321
+ newFrom,
322
+ newTo,
323
+ occurrenceLimit
324
+ ) {
325
+ const oldOccurrences = collectNaturalOccurrences(oldParts, oldFrom, oldTo, 1);
326
+ const newOccurrences = collectNaturalOccurrences(newParts, newFrom, newTo, 1);
327
+ const candidates = [];
328
+
329
+ for (const [signature, oldIndexes] of oldOccurrences) {
330
+ const newIndexes = newOccurrences.get(signature);
331
+ if (
332
+ !newIndexes
333
+ || oldIndexes.length > occurrenceLimit
334
+ || newIndexes.length > occurrenceLimit
335
+ || signature.trim() === ''
336
+ ) {
112
337
  continue;
113
338
  }
339
+ for (const oldIndex of oldIndexes) {
340
+ for (const newIndex of newIndexes) {
341
+ candidates.push({ oldIndex, newIndex });
342
+ }
343
+ }
344
+ }
345
+ return selectNaturalMonotonicAnchors(candidates, 1);
346
+ }
114
347
 
115
- if (!group) {
116
- group = {
117
- oldStart: oldOffset,
118
- oldText: '',
119
- newText: ''
120
- };
348
+ function collectNaturalOccurrences(items, from, to, span) {
349
+ const occurrences = new Map();
350
+ for (let index = from; index + span <= to; index += 1) {
351
+ const signature = span === 1
352
+ ? items[index]
353
+ : JSON.stringify(items.slice(index, index + span));
354
+ const indexes = occurrences.get(signature);
355
+ if (indexes) {
356
+ indexes.push(index);
357
+ } else {
358
+ occurrences.set(signature, [index]);
121
359
  }
360
+ }
361
+ return occurrences;
362
+ }
122
363
 
123
- if (edit.type === 'remove') {
124
- const text = oldParts[edit.oldIndex];
125
- group.oldText += text;
126
- oldOffset += text.length;
127
- } else if (edit.type === 'add') {
128
- const text = newParts[edit.newIndex];
129
- group.newText += text;
130
- newOffset += text.length;
364
+ function selectNaturalMonotonicAnchors(candidates, span) {
365
+ if (candidates.length === 0) {
366
+ return [];
367
+ }
368
+
369
+ candidates.sort((left, right) => (
370
+ left.oldIndex - right.oldIndex || right.newIndex - left.newIndex
371
+ ));
372
+ const tailValues = [];
373
+ const tailCandidateIndexes = [];
374
+ const predecessors = new Int32Array(candidates.length);
375
+ predecessors.fill(-1);
376
+
377
+ for (let candidateIndex = 0; candidateIndex < candidates.length; candidateIndex += 1) {
378
+ const candidate = candidates[candidateIndex];
379
+ let low = 0;
380
+ let high = tailValues.length;
381
+ while (low < high) {
382
+ const middle = (low + high) >> 1;
383
+ if (tailValues[middle] < candidate.newIndex) {
384
+ low = middle + 1;
385
+ } else {
386
+ high = middle;
387
+ }
131
388
  }
389
+ if (low > 0) {
390
+ predecessors[candidateIndex] = tailCandidateIndexes[low - 1];
391
+ }
392
+ tailValues[low] = candidate.newIndex;
393
+ tailCandidateIndexes[low] = candidateIndex;
132
394
  }
133
- flushGroup();
134
395
 
135
- return groups;
396
+ const selected = [];
397
+ let cursor = tailCandidateIndexes[tailCandidateIndexes.length - 1];
398
+ while (cursor >= 0) {
399
+ selected.push(candidates[cursor]);
400
+ cursor = predecessors[cursor];
401
+ }
402
+ selected.reverse();
403
+
404
+ const nonOverlapping = [];
405
+ let previousOldEnd = -1;
406
+ let previousNewEnd = -1;
407
+ for (const anchor of selected) {
408
+ if (
409
+ anchor.oldIndex >= previousOldEnd
410
+ && anchor.newIndex >= previousNewEnd
411
+ ) {
412
+ nonOverlapping.push(anchor);
413
+ previousOldEnd = anchor.oldIndex + span;
414
+ previousNewEnd = anchor.newIndex + span;
415
+ }
416
+ }
417
+ return nonOverlapping;
418
+ }
419
+
420
+ function appendNaturalPositionalLineGroups(state, oldFrom, oldTo, newFrom, newTo) {
421
+ if (oldTo - oldFrom !== newTo - newFrom) {
422
+ return false;
423
+ }
136
424
 
137
- function flushGroup() {
138
- if (!group) {
425
+ let changed = false;
426
+ let lowSimilarityOldStart = -1;
427
+ let lowSimilarityNewStart = -1;
428
+
429
+ const flushLowSimilarityRun = (oldEnd, newEnd) => {
430
+ if (lowSimilarityOldStart < 0) {
139
431
  return;
140
432
  }
141
- groups.push(group);
142
- group = null;
433
+ pushNaturalLineGroup(
434
+ state,
435
+ lowSimilarityOldStart,
436
+ oldEnd,
437
+ lowSimilarityNewStart,
438
+ newEnd
439
+ );
440
+ lowSimilarityOldStart = -1;
441
+ lowSimilarityNewStart = -1;
442
+ };
443
+
444
+ for (let offset = 0; offset < oldTo - oldFrom; offset += 1) {
445
+ const oldIndex = oldFrom + offset;
446
+ const newIndex = newFrom + offset;
447
+ const oldLine = state.oldParts[oldIndex];
448
+ const newLine = state.newParts[newIndex];
449
+
450
+ if (oldLine === newLine) {
451
+ flushLowSimilarityRun(oldIndex, newIndex);
452
+ continue;
453
+ }
454
+
455
+ changed = true;
456
+ if (computeNaturalLineSimilarity(oldLine, newLine) >= 0.55) {
457
+ flushLowSimilarityRun(oldIndex, newIndex);
458
+ pushNaturalLineGroup(state, oldIndex, oldIndex + 1, newIndex, newIndex + 1);
459
+ } else if (lowSimilarityOldStart < 0) {
460
+ lowSimilarityOldStart = oldIndex;
461
+ lowSimilarityNewStart = newIndex;
462
+ }
143
463
  }
464
+ flushLowSimilarityRun(oldTo, newTo);
465
+ return changed;
144
466
  }
145
467
 
146
- function computeTokenAnchoredPatches(oldValue, newValue, offset = 0) {
147
- const MAX_GROUP_CHARS = 20000;
148
- const MAX_TOKENS = 3000;
149
- const MAX_PRODUCT = 4000000;
150
- const MAX_PATCHES = 80;
468
+ function computeNaturalLineSimilarity(oldLine, newLine) {
469
+ if (oldLine === newLine) {
470
+ return 1;
471
+ }
472
+ const denominator = Math.max(oldLine.length, newLine.length, 1);
473
+ let prefix = 0;
474
+ while (
475
+ prefix < oldLine.length
476
+ && prefix < newLine.length
477
+ && oldLine[prefix] === newLine[prefix]
478
+ ) {
479
+ prefix += 1;
480
+ }
151
481
 
152
- if (
153
- oldValue.length > MAX_GROUP_CHARS
154
- || newValue.length > MAX_GROUP_CHARS
482
+ let suffix = 0;
483
+ while (
484
+ suffix < oldLine.length - prefix
485
+ && suffix < newLine.length - prefix
486
+ && oldLine[oldLine.length - 1 - suffix] === newLine[newLine.length - 1 - suffix]
155
487
  ) {
156
- return null;
488
+ suffix += 1;
489
+ }
490
+ return (prefix + suffix) / denominator;
491
+ }
492
+
493
+
494
+ function computeTokenAnchoredPatches(oldValue, newValue, offset = 0) {
495
+ if (oldValue === newValue) {
496
+ return [];
157
497
  }
158
498
 
159
499
  const oldTokens = splitTextTokens(oldValue);
160
500
  const newTokens = splitTextTokens(newValue);
501
+ const oldOffsets = computeNaturalTokenOffsets(oldTokens, oldValue.length);
502
+ const newOffsets = computeNaturalTokenOffsets(newTokens, newValue.length);
503
+ const patches = [];
504
+
505
+ collectNaturalTokenPatches({
506
+ oldValue,
507
+ newValue,
508
+ oldTokens,
509
+ newTokens,
510
+ oldTokenTexts: oldTokens.map(token => token.text),
511
+ newTokenTexts: newTokens.map(token => token.text),
512
+ oldOffsets,
513
+ newOffsets,
514
+ offset,
515
+ patches,
516
+ oldFrom: 0,
517
+ oldTo: oldTokens.length,
518
+ newFrom: 0,
519
+ newTo: newTokens.length,
520
+ depth: 0
521
+ });
522
+
523
+ return patches.sort((left, right) => left.from - right.from || left.to - right.to);
524
+ }
525
+
526
+
527
+ function computeNaturalTokenOffsets(tokens, textLength) {
528
+ const offsets = new Array(tokens.length + 1);
529
+ for (let index = 0; index < tokens.length; index += 1) {
530
+ offsets[index] = tokens[index].start;
531
+ }
532
+ offsets[tokens.length] = textLength;
533
+ return offsets;
534
+ }
535
+
536
+ function collectNaturalTokenPatches(state) {
537
+ let { oldFrom, oldTo, newFrom, newTo } = state;
538
+
539
+ while (
540
+ oldFrom < oldTo
541
+ && newFrom < newTo
542
+ && state.oldTokenTexts[oldFrom] === state.newTokenTexts[newFrom]
543
+ ) {
544
+ oldFrom += 1;
545
+ newFrom += 1;
546
+ }
547
+ while (
548
+ oldFrom < oldTo
549
+ && newFrom < newTo
550
+ && state.oldTokenTexts[oldTo - 1] === state.newTokenTexts[newTo - 1]
551
+ ) {
552
+ oldTo -= 1;
553
+ newTo -= 1;
554
+ }
555
+
556
+ if (oldFrom === oldTo || newFrom === newTo) {
557
+ pushNaturalTokenPatch(state, oldFrom, oldTo, newFrom, newTo);
558
+ return;
559
+ }
560
+
561
+ const oldCount = oldTo - oldFrom;
562
+ const newCount = newTo - newFrom;
563
+ const oldChars = state.oldOffsets[oldTo] - state.oldOffsets[oldFrom];
564
+ const newChars = state.newOffsets[newTo] - state.newOffsets[newFrom];
565
+ const MAX_GROUP_CHARS = 20_000;
566
+ const MAX_TOKENS = 3000;
567
+ const MAX_PRODUCT = 4_000_000;
568
+
161
569
  if (
162
- oldTokens.length === 0
163
- || newTokens.length === 0
164
- || oldTokens.length > MAX_TOKENS
165
- || newTokens.length > MAX_TOKENS
166
- || oldTokens.length * newTokens.length > MAX_PRODUCT
570
+ oldChars + newChars <= MAX_GROUP_CHARS
571
+ && oldCount <= MAX_TOKENS
572
+ && newCount <= MAX_TOKENS
573
+ && oldCount * newCount <= MAX_PRODUCT
167
574
  ) {
168
- return null;
575
+ appendNaturalExactTokenPatches(state, oldFrom, oldTo, newFrom, newTo);
576
+ return;
169
577
  }
170
578
 
171
- const edits = computePartEdits(
172
- oldTokens.map(token => token.text),
173
- newTokens.map(token => token.text)
579
+ const anchors = discoverNaturalTokenAnchors(
580
+ state.oldTokenTexts,
581
+ state.newTokenTexts,
582
+ oldFrom,
583
+ oldTo,
584
+ newFrom,
585
+ newTo
174
586
  );
175
- const patches = [];
176
- let oldOffset = 0;
177
- let newOffset = 0;
178
- let group = null;
179
-
180
- for (const edit of edits) {
181
- if (edit.type === 'equal') {
182
- flushGroup();
183
- oldOffset = oldTokens[edit.oldIndex].end;
184
- newOffset = newTokens[edit.newIndex].end;
185
- continue;
587
+ if (anchors.length > 0 && state.depth < 8) {
588
+ const ANCHOR_SPAN = 3;
589
+ let oldCursor = oldFrom;
590
+ let newCursor = newFrom;
591
+ for (const anchor of anchors) {
592
+ collectNaturalTokenPatches({
593
+ ...state,
594
+ oldFrom: oldCursor,
595
+ oldTo: anchor.oldIndex,
596
+ newFrom: newCursor,
597
+ newTo: anchor.newIndex,
598
+ depth: state.depth + 1
599
+ });
600
+ oldCursor = anchor.oldIndex + ANCHOR_SPAN;
601
+ newCursor = anchor.newIndex + ANCHOR_SPAN;
186
602
  }
603
+ collectNaturalTokenPatches({
604
+ ...state,
605
+ oldFrom: oldCursor,
606
+ oldTo,
607
+ newFrom: newCursor,
608
+ newTo,
609
+ depth: state.depth + 1
610
+ });
611
+ return;
612
+ }
187
613
 
188
- if (!group) {
189
- group = {
190
- oldStart: oldOffset,
191
- newStart: newOffset,
192
- oldEnd: oldOffset,
193
- newEnd: newOffset
194
- };
195
- }
614
+ pushNaturalTokenPatch(state, oldFrom, oldTo, newFrom, newTo);
615
+ }
196
616
 
197
- if (edit.type === 'remove') {
198
- group.oldEnd = oldTokens[edit.oldIndex].end;
199
- oldOffset = group.oldEnd;
200
- } else if (edit.type === 'add') {
201
- group.newEnd = newTokens[edit.newIndex].end;
202
- newOffset = group.newEnd;
203
- }
617
+ function pushNaturalTokenPatch(state, oldFrom, oldTo, newFrom, newTo) {
618
+ const oldStart = state.oldOffsets[oldFrom];
619
+ const oldEnd = state.oldOffsets[oldTo];
620
+ const newStart = state.newOffsets[newFrom];
621
+ const newEnd = state.newOffsets[newTo];
622
+ const expected = state.oldValue.slice(oldStart, oldEnd);
623
+ const insert = state.newValue.slice(newStart, newEnd);
624
+ if (expected === insert) {
625
+ return;
204
626
  }
205
- flushGroup();
627
+ state.patches.push({
628
+ from: state.offset + oldStart,
629
+ to: state.offset + oldEnd,
630
+ expected,
631
+ insert
632
+ });
633
+ }
206
634
 
207
- if (!patches.length || patches.length > MAX_PATCHES) {
208
- return null;
635
+ function appendNaturalExactTokenPatches(state, oldFrom, oldTo, newFrom, newTo) {
636
+ const matches = computeNaturalLcsMatches(
637
+ state.oldTokenTexts.slice(oldFrom, oldTo),
638
+ state.newTokenTexts.slice(newFrom, newTo)
639
+ );
640
+ let oldCursor = oldFrom;
641
+ let newCursor = newFrom;
642
+ for (const match of matches) {
643
+ const oldIndex = oldFrom + match.oldIndex;
644
+ const newIndex = newFrom + match.newIndex;
645
+ pushNaturalTokenPatch(state, oldCursor, oldIndex, newCursor, newIndex);
646
+ oldCursor = oldIndex + 1;
647
+ newCursor = newIndex + 1;
209
648
  }
210
- return patches;
649
+ pushNaturalTokenPatch(state, oldCursor, oldTo, newCursor, newTo);
650
+ }
211
651
 
212
- function flushGroup() {
213
- if (!group) {
214
- return;
652
+ function discoverNaturalTokenAnchors(
653
+ oldTokens,
654
+ newTokens,
655
+ oldFrom,
656
+ oldTo,
657
+ newFrom,
658
+ newTo
659
+ ) {
660
+ const ANCHOR_SPAN = 3;
661
+ const oldOccurrences = collectNaturalOccurrences(
662
+ oldTokens,
663
+ oldFrom,
664
+ oldTo,
665
+ ANCHOR_SPAN
666
+ );
667
+ const newOccurrences = collectNaturalOccurrences(
668
+ newTokens,
669
+ newFrom,
670
+ newTo,
671
+ ANCHOR_SPAN
672
+ );
673
+ const candidates = [];
674
+
675
+ for (const [signature, oldIndexes] of oldOccurrences) {
676
+ const newIndexes = newOccurrences.get(signature);
677
+ if (
678
+ !newIndexes
679
+ || oldIndexes.length > 4
680
+ || newIndexes.length > 4
681
+ ) {
682
+ continue;
215
683
  }
216
- const oldText = oldValue.slice(group.oldStart, group.oldEnd);
217
- const newText = newValue.slice(group.newStart, group.newEnd);
218
- if (oldText !== newText) {
219
- patches.push({
220
- from: offset + group.oldStart,
221
- to: offset + group.oldEnd,
222
- expected: oldText,
223
- insert: newText
224
- });
684
+ for (const oldIndex of oldIndexes) {
685
+ for (const newIndex of newIndexes) {
686
+ candidates.push({ oldIndex, newIndex });
687
+ }
225
688
  }
226
- group = null;
227
689
  }
690
+ return selectNaturalMonotonicAnchors(candidates, ANCHOR_SPAN);
228
691
  }
229
692
 
230
693
  function splitTextTokens(text) {
@@ -519,40 +982,28 @@ function splitSentences(text) {
519
982
  return spans;
520
983
  }
521
984
 
985
+
986
+
522
987
  function computeGroupMetrics(group, tokenPatches) {
523
988
  const oldNonEmptyLineCount = countNonEmptyLines(group.oldText);
524
989
  const newNonEmptyLineCount = countNonEmptyLines(group.newText);
525
-
526
990
  return {
527
991
  oldNonEmptyLineCount,
528
992
  newNonEmptyLineCount,
529
993
  maxNonEmptyLineCount: Math.max(oldNonEmptyLineCount, newNonEmptyLineCount),
530
994
  changedSpanChars: Math.max(group.oldText.length, group.newText.length),
531
- tokenPatchCount: tokenPatches === null ? null : tokenPatches.length,
532
- totalTokenChangedChars: tokenPatches === null
533
- ? null
534
- : tokenPatches.reduce((sum, patch) => (
535
- sum + Math.max(patch.to - patch.from, patch.insert.length)
536
- ), 0),
995
+ tokenPatchCount: Array.isArray(tokenPatches) ? tokenPatches.length : null,
996
+ totalTokenChangedChars: Array.isArray(tokenPatches)
997
+ ? tokenPatches.reduce(
998
+ (sum, patch) => sum + Math.max(patch.expected.length, patch.insert.length),
999
+ 0
1000
+ )
1001
+ : null,
537
1002
  oldSentenceTerminatorCount: countSentenceTerminators(group.oldText),
538
1003
  newSentenceTerminatorCount: countSentenceTerminators(group.newText)
539
1004
  };
540
1005
  }
541
1006
 
542
- // Resolves the sentence-span quantities used by the `isSentenceRewrite`
543
- // predicate (the design spec leaves them undefined). It segments the changed
544
- // group's OLD text into sentence spans and checks whether every token patch's
545
- // old range maps within a single span.
546
- //
547
- // Returns:
548
- // - `fitsOneSpan`: true iff exactly one sentence span contains every token
549
- // patch's old range (relative to the group).
550
- // - `spanChars` / `spanTokenCount`: the char length / token count of that
551
- // single span when `fitsOneSpan` is true; `0` otherwise (irrelevant then).
552
- // - `spanStart` / `spanEnd`: the group-relative `[start,end)` offsets of that
553
- // single span when `fitsOneSpan` is true; `0` otherwise (irrelevant then).
554
- //
555
- // When `tokenPatches` is `null` or empty, `fitsOneSpan` is false.
556
1007
  function resolveTokenPatchSentenceSpan(group, tokenPatches) {
557
1008
  const empty = {
558
1009
  fitsOneSpan: false,
@@ -611,74 +1062,99 @@ function resolveTokenPatchSentenceSpan(group, tokenPatches) {
611
1062
  // `tokenPatches === null`, every token-dependent predicate is false, so the
612
1063
  // only reachable results are `annotated_block`, `paragraph_rewrite` (via the
613
1064
  // line-count or sentence-terminator branch), and `fallback`.
1065
+
1066
+
1067
+
614
1068
  function classifyChangedGroup(group, tokenPatches, metrics) {
615
- const newGroupText = group.newText;
616
- const {
617
- maxNonEmptyLineCount,
618
- changedSpanChars,
619
- tokenPatchCount,
620
- totalTokenChangedChars,
621
- oldSentenceTerminatorCount,
622
- newSentenceTerminatorCount
623
- } = metrics;
624
-
625
- const isAnnotatedBlock = hasOriginalMarkerLine(newGroupText)
626
- && hasLaterRevisedMarkerLine(newGroupText)
627
- && maxNonEmptyLineCount >= 3;
628
- if (isAnnotatedBlock) {
1069
+ if (hasAnyAnnotatedMarker(group.oldText) || hasAnyAnnotatedMarker(group.newText)) {
629
1070
  return { type: 'annotated_block' };
630
1071
  }
631
1072
 
632
- const isDenseTokenRewrite = tokenPatches !== null
633
- && tokenPatchCount >= 6
634
- && changedSpanChars >= 160
635
- && tokenPatchCount / Math.max(1, maxNonEmptyLineCount) >= 2;
636
-
637
- const isParagraphRewrite = !isAnnotatedBlock
638
- && (
639
- maxNonEmptyLineCount >= 3
640
- || (oldSentenceTerminatorCount >= 2 && newSentenceTerminatorCount >= 2)
641
- || isDenseTokenRewrite
642
- );
643
- if (isParagraphRewrite) {
1073
+ const hasTokenPatches = Array.isArray(tokenPatches) && tokenPatches.length > 0;
1074
+ const maxTokenPatchChars = hasTokenPatches
1075
+ ? tokenPatches.reduce(
1076
+ (maximum, patch) => Math.max(
1077
+ maximum,
1078
+ patch.expected.length,
1079
+ patch.insert.length
1080
+ ),
1081
+ 0
1082
+ )
1083
+ : null;
1084
+ const editDensity = hasTokenPatches
1085
+ ? metrics.totalTokenChangedChars / Math.max(1, metrics.changedSpanChars)
1086
+ : null;
1087
+ const sentenceSpan = hasTokenPatches
1088
+ ? resolveTokenPatchSentenceSpan(group, tokenPatches)
1089
+ : null;
1090
+ const sentenceEditDensity = sentenceSpan?.fitsOneSpan
1091
+ ? metrics.totalTokenChangedChars / Math.max(1, sentenceSpan.spanChars)
1092
+ : null;
1093
+
1094
+ const isDenseTokenRewrite = hasTokenPatches
1095
+ && metrics.tokenPatchCount >= 6
1096
+ && metrics.changedSpanChars >= 160
1097
+ && editDensity >= 0.20
1098
+ && metrics.tokenPatchCount / Math.max(1, metrics.maxNonEmptyLineCount) >= 2;
1099
+ if (isDenseTokenRewrite) {
644
1100
  return { type: 'paragraph_rewrite' };
645
1101
  }
646
1102
 
647
- const sentenceSpan = resolveTokenPatchSentenceSpan(group, tokenPatches);
1103
+ // A dense rewrite wholly contained in one confident sentence stays a
1104
+ // sentence patch even when the surrounding line contains unchanged prose.
1105
+ if (
1106
+ hasTokenPatches
1107
+ && metrics.tokenPatchCount >= 3
1108
+ && sentenceSpan.fitsOneSpan
1109
+ && sentenceSpan.spanChars >= 80
1110
+ && sentenceEditDensity >= 0.20
1111
+ ) {
1112
+ return { type: 'sentence_rewrite' };
1113
+ }
648
1114
 
649
- const isSentenceRewrite = !isAnnotatedBlock
650
- && !isParagraphRewrite
651
- && tokenPatches !== null
652
- && tokenPatchCount >= 3
1115
+ // A collection of bounded, low-density edits stays local regardless of the
1116
+ // file size, line count, or number of repeated replacements.
1117
+ if (
1118
+ hasTokenPatches
1119
+ && maxTokenPatchChars <= 96
1120
+ && editDensity < 0.20
1121
+ ) {
1122
+ return { type: 'small_edit' };
1123
+ }
1124
+
1125
+ if (metrics.maxNonEmptyLineCount >= 3) {
1126
+ return { type: 'paragraph_rewrite' };
1127
+ }
1128
+ if (
1129
+ metrics.oldSentenceTerminatorCount >= 2
1130
+ && metrics.newSentenceTerminatorCount >= 2
1131
+ && !sentenceSpan?.fitsOneSpan
1132
+ ) {
1133
+ return { type: 'paragraph_rewrite' };
1134
+ }
1135
+
1136
+ if (
1137
+ hasTokenPatches
1138
+ && metrics.tokenPatchCount >= 3
653
1139
  && sentenceSpan.fitsOneSpan
654
- && (sentenceSpan.spanChars >= 80 || sentenceSpan.spanTokenCount >= 12);
655
- if (isSentenceRewrite) {
1140
+ && sentenceSpan.spanChars >= 80
1141
+ ) {
656
1142
  return { type: 'sentence_rewrite' };
657
1143
  }
658
1144
 
659
- const isSmallEdit = !isAnnotatedBlock
660
- && !isParagraphRewrite
661
- && !isSentenceRewrite
662
- && tokenPatches !== null
1145
+ if (
1146
+ hasTokenPatches
663
1147
  && (
664
- tokenPatchCount <= 2
665
- || (
666
- totalTokenChangedChars < 80
667
- && maxNonEmptyLineCount <= 2
668
- && !hasAnyAnnotatedMarker(newGroupText)
669
- )
670
- );
671
- if (isSmallEdit) {
1148
+ metrics.tokenPatchCount <= 2
1149
+ || metrics.totalTokenChangedChars < 80
1150
+ )
1151
+ ) {
672
1152
  return { type: 'small_edit' };
673
1153
  }
674
1154
 
675
1155
  return { type: 'fallback' };
676
1156
  }
677
1157
 
678
- // The single-patch fallback for a whole changed group (spec algorithm sketch).
679
- // Returns a one-element array so callers can treat every builder uniformly.
680
- // The patch's `from`/`to` are absolute offsets into the full original text:
681
- // `computeSingleTextPatch` adds `group.oldStart` to its segment-local offsets.
682
1158
  function singleGroupPatch(group) {
683
1159
  return [computeSingleTextPatch(group.oldText, group.newText, group.oldStart)];
684
1160
  }
@@ -811,89 +1287,61 @@ function isCoalesceFillerGap(gap) {
811
1287
  // `expected` the original slice and `insert` the merged inserts interleaved
812
1288
  // with the unchanged gap text. When nothing qualifies the token patches are
813
1289
  // returned unchanged. Absolute offsets are preserved throughout.
814
- function coalesceTokenPatches(group, tokenPatches) {
815
- if (tokenPatches === null || tokenPatches.length < 3) {
816
- return tokenPatches;
817
- }
818
-
819
- const spans = splitSentences(group.oldText);
820
- // Index of the sentence span that fully contains a patch's group-relative
821
- // range, or -1 when it is not cleanly inside any single span.
822
- const spanOf = patch => {
823
- const relativeFrom = patch.from - group.oldStart;
824
- const relativeTo = patch.to - group.oldStart;
825
- return spans.findIndex(span => (
826
- relativeFrom >= span.start && relativeTo <= span.end
827
- ));
828
- };
829
1290
 
830
- // Count patches per sentence span so the ">= 3 in the span" gate can be
831
- // checked before merging any run.
832
- const patchesPerSpan = new Map();
833
- for (const patch of tokenPatches) {
834
- const spanIndex = spanOf(patch);
835
- patchesPerSpan.set(spanIndex, (patchesPerSpan.get(spanIndex) || 0) + 1);
1291
+ function coalesceTokenPatches(group, tokenPatches) {
1292
+ if (!Array.isArray(tokenPatches) || tokenPatches.length <= 1) {
1293
+ return Array.isArray(tokenPatches) && tokenPatches.length > 0
1294
+ ? tokenPatches
1295
+ : singleGroupPatch(group);
836
1296
  }
837
1297
 
1298
+ const ordered = [...tokenPatches].sort(
1299
+ (left, right) => left.from - right.from || left.to - right.to
1300
+ );
838
1301
  const result = [];
839
- let run = [];
840
- let runSpanIndex = -1;
1302
+ let run = [ordered[0]];
841
1303
 
842
1304
  const flushRun = () => {
843
- if (run.length === 0) {
844
- return;
845
- }
846
- if (run.length === 1) {
847
- result.push(run[0]);
848
- } else {
1305
+ if (run.length >= 3 && computeNaturalPatchRunDensity(group, run) >= 0.35) {
849
1306
  result.push(mergeTokenPatchRun(group, run));
1307
+ } else {
1308
+ result.push(...run);
850
1309
  }
851
1310
  run = [];
852
1311
  };
853
1312
 
854
- for (const patch of tokenPatches) {
855
- const spanIndex = spanOf(patch);
856
- const eligibleSpan = spanIndex !== -1 && patchesPerSpan.get(spanIndex) >= 3;
857
-
858
- if (run.length === 0) {
859
- run = eligibleSpan ? [patch] : [];
860
- runSpanIndex = eligibleSpan ? spanIndex : -1;
861
- if (!eligibleSpan) {
862
- result.push(patch);
863
- }
864
- continue;
865
- }
866
-
867
- const prev = run[run.length - 1];
868
- const gap = group.oldText.slice(
869
- prev.to - group.oldStart,
870
- patch.from - group.oldStart
871
- );
872
- const mergeable = eligibleSpan
873
- && spanIndex === runSpanIndex
1313
+ for (let index = 1; index < ordered.length; index += 1) {
1314
+ const previous = run[run.length - 1];
1315
+ const current = ordered[index];
1316
+ const gapStart = Math.max(0, previous.to - group.oldStart);
1317
+ const gapEnd = Math.max(gapStart, current.from - group.oldStart);
1318
+ const gap = group.oldText.slice(gapStart, gapEnd);
1319
+ const canJoin = gap.length <= 40
1320
+ && !/\n\s*\n/.test(gap)
874
1321
  && isCoalesceFillerGap(gap);
875
1322
 
876
- if (mergeable) {
877
- run.push(patch);
878
- continue;
879
- }
880
-
881
- flushRun();
882
- run = eligibleSpan ? [patch] : [];
883
- runSpanIndex = eligibleSpan ? spanIndex : -1;
884
- if (!eligibleSpan) {
885
- result.push(patch);
1323
+ if (canJoin) {
1324
+ run.push(current);
1325
+ } else {
1326
+ flushRun();
1327
+ run = [current];
886
1328
  }
887
1329
  }
888
1330
  flushRun();
889
-
890
1331
  return result;
891
1332
  }
892
1333
 
893
- // Merges a run of >= 2 adjacent token patches into one patch spanning
894
- // `[firstFrom, lastTo)`. `expected` is the original-text slice (the patches'
895
- // expecteds interleaved with the unchanged gap text); `insert` is the patches'
896
- // inserts interleaved with the same gap text.
1334
+ function computeNaturalPatchRunDensity(group, run) {
1335
+ const first = run[0];
1336
+ const last = run[run.length - 1];
1337
+ const oldSpan = Math.max(1, last.to - first.from);
1338
+ const changedChars = run.reduce(
1339
+ (sum, patch) => sum + Math.max(patch.expected.length, patch.insert.length),
1340
+ 0
1341
+ );
1342
+ return Math.min(1, changedChars / oldSpan);
1343
+ }
1344
+
897
1345
  function mergeTokenPatchRun(group, run) {
898
1346
  const first = run[0];
899
1347
  const last = run[run.length - 1];