@khanhicetea/pi-better-tool 0.1.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +69 -45
- package/package.json +1 -1
- package/src/apply.ts +53 -18
- package/src/diagnostics.ts +239 -39
- package/src/index.ts +17 -7
- package/src/read-evidence.ts +174 -0
- package/src/similarity.ts +210 -156
- package/src/text.ts +36 -7
- package/src/tool.ts +265 -83
package/src/similarity.ts
CHANGED
|
@@ -1,173 +1,246 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
* oldText
|
|
4
|
-
* instead of forcing a full re-read.
|
|
2
|
+
* Bounded fuzzy line similarity used to locate the region of a file that a
|
|
3
|
+
* failed oldText almost matched.
|
|
5
4
|
*/
|
|
6
5
|
|
|
7
|
-
import {
|
|
6
|
+
import { normalizeForFuzzyMatch, normalizeToLF } from "./text.ts";
|
|
8
7
|
|
|
9
8
|
export interface AlignOp {
|
|
10
|
-
type: "equal" | "changed" | "file-only" | "old-only";
|
|
11
|
-
/** 1-based line number in the file (for equal/changed/file-only). */
|
|
9
|
+
type: "equal" | "similar" | "changed" | "file-only" | "old-only";
|
|
10
|
+
/** 1-based line number in the file (for equal/similar/changed/file-only). */
|
|
12
11
|
fileLine?: number;
|
|
12
|
+
/** Original LF-normalized file text, never fuzzy-normalized. */
|
|
13
13
|
fileText?: string;
|
|
14
14
|
/** 1-based line number in the model-provided oldText. */
|
|
15
15
|
oldLine?: number;
|
|
16
|
+
/** Original LF-normalized oldText, never fuzzy-normalized. */
|
|
16
17
|
oldText?: string;
|
|
18
|
+
/** Heuristic similarity for aligned pairs. */
|
|
19
|
+
similarity?: number;
|
|
17
20
|
}
|
|
18
21
|
|
|
19
22
|
export interface ClosestRegion {
|
|
20
23
|
startLine: number;
|
|
21
24
|
endLine: number;
|
|
22
|
-
/** 0..1
|
|
25
|
+
/** 0..1 order-preserving, one-to-one line similarity. */
|
|
23
26
|
score: number;
|
|
24
27
|
ops: AlignOp[];
|
|
28
|
+
/** Number of exactly equal original lines. */
|
|
25
29
|
equalCount: number;
|
|
26
30
|
totalOldLines: number;
|
|
27
31
|
/** oldText was truncated for comparison (very large oldText). */
|
|
28
32
|
truncated: boolean;
|
|
33
|
+
/** Best distinct, non-overlapping competitor, when retained by the bounded search. */
|
|
34
|
+
competitor?: { startLine: number; endLine: number; score: number };
|
|
35
|
+
/** A discarded window was too competitive to prove a safe score margin. */
|
|
36
|
+
competitionIncomplete: boolean;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
interface PreparedLine {
|
|
40
|
+
text: string;
|
|
41
|
+
grams: Map<string, number>;
|
|
42
|
+
gramCount: number;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function prepareLine(text: string): PreparedLine {
|
|
46
|
+
const grams = new Map<string, number>();
|
|
47
|
+
for (let i = 0; i + 1 < text.length; i++) {
|
|
48
|
+
const gram = text.slice(i, i + 2);
|
|
49
|
+
grams.set(gram, (grams.get(gram) ?? 0) + 1);
|
|
50
|
+
}
|
|
51
|
+
return { text, grams, gramCount: Math.max(0, text.length - 1) };
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
class WorkBudget {
|
|
55
|
+
private remaining: number;
|
|
56
|
+
constructor(remaining: number) {
|
|
57
|
+
this.remaining = remaining;
|
|
58
|
+
}
|
|
59
|
+
spend(amount: number): boolean {
|
|
60
|
+
this.remaining -= amount;
|
|
61
|
+
return this.remaining >= 0;
|
|
62
|
+
}
|
|
29
63
|
}
|
|
30
64
|
|
|
31
65
|
/** Dice coefficient over character bigrams; 1.0 for identical strings. */
|
|
32
66
|
export function lineSimilarity(a: string, b: string): number {
|
|
33
|
-
|
|
34
|
-
if (a.length < 2 || b.length < 2) return 0;
|
|
35
|
-
const gramsA = bigramCounts(a);
|
|
36
|
-
const gramsB = bigramCounts(b);
|
|
37
|
-
let overlap = 0;
|
|
38
|
-
let total = 0;
|
|
39
|
-
for (const count of gramsA.values()) total += count;
|
|
40
|
-
for (const count of gramsB.values()) total += count;
|
|
41
|
-
for (const [gram, countA] of gramsA) {
|
|
42
|
-
const countB = gramsB.get(gram);
|
|
43
|
-
if (countB) overlap += Math.min(countA, countB);
|
|
44
|
-
}
|
|
45
|
-
return (2 * overlap) / total;
|
|
67
|
+
return preparedSimilarity(prepareLine(a), prepareLine(b));
|
|
46
68
|
}
|
|
47
69
|
|
|
48
|
-
function
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
70
|
+
function preparedSimilarity(a: PreparedLine, b: PreparedLine, budget?: WorkBudget): number {
|
|
71
|
+
if (budget && !budget.spend(1)) throw new Error("similarity-work-budget-exhausted");
|
|
72
|
+
if (a.text === b.text) return 1;
|
|
73
|
+
if (a.gramCount === 0 || b.gramCount === 0) return 0;
|
|
74
|
+
const [small, large] = a.grams.size <= b.grams.size ? [a.grams, b.grams] : [b.grams, a.grams];
|
|
75
|
+
if (budget && !budget.spend(small.size)) throw new Error("similarity-work-budget-exhausted");
|
|
76
|
+
let overlap = 0;
|
|
77
|
+
for (const [gram, count] of small) {
|
|
78
|
+
const other = large.get(gram);
|
|
79
|
+
if (other) overlap += Math.min(count, other);
|
|
53
80
|
}
|
|
54
|
-
return
|
|
81
|
+
return (2 * overlap) / (a.gramCount + b.gramCount);
|
|
55
82
|
}
|
|
56
83
|
|
|
57
84
|
const MAX_COMPARE_LINES = 300;
|
|
58
|
-
const
|
|
85
|
+
const MAX_SEARCH_FILE_LINES = 10_000;
|
|
86
|
+
const MAX_CANDIDATE_WINDOWS = 48;
|
|
87
|
+
const MAX_QUERY_BYTES = 128 * 1024;
|
|
88
|
+
const MAX_LINE_CHARS = 16 * 1024;
|
|
89
|
+
const MAX_SIMILARITY_WORK = 20_000_000;
|
|
59
90
|
const MIN_SCORE = 0.3;
|
|
60
91
|
const MATCH_THRESHOLD = 0.75;
|
|
61
92
|
|
|
93
|
+
interface CandidateWindow {
|
|
94
|
+
start: number;
|
|
95
|
+
size: number;
|
|
96
|
+
positionalScore: number;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
interface ScoredWindow extends CandidateWindow {
|
|
100
|
+
score: number;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function originalLines(text: string): string[] {
|
|
104
|
+
const lines = normalizeToLF(text).split("\n");
|
|
105
|
+
// Remove only the synthetic segment after a terminal newline. Do not remove
|
|
106
|
+
// a real, unterminated whitespace-only line.
|
|
107
|
+
if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
|
|
108
|
+
return lines;
|
|
109
|
+
}
|
|
110
|
+
|
|
62
111
|
/**
|
|
63
|
-
* Find the window
|
|
64
|
-
*
|
|
65
|
-
* still lands the search on the right region.
|
|
112
|
+
* Find the best window and a distinct runner-up under explicit input/work
|
|
113
|
+
* budgets. Returning null is the safe fallback when a budget is exhausted.
|
|
66
114
|
*/
|
|
67
115
|
export function findClosestRegion(content: string, oldText: string): ClosestRegion | null {
|
|
68
|
-
|
|
69
|
-
const
|
|
70
|
-
|
|
116
|
+
if (Buffer.byteLength(oldText, "utf8") > MAX_QUERY_BYTES) return null;
|
|
117
|
+
const originalQ = originalLines(oldText);
|
|
118
|
+
const originalContent = originalLines(content);
|
|
119
|
+
if (originalQ.length === 0 || originalContent.length === 0 || originalContent.length > MAX_SEARCH_FILE_LINES) return null;
|
|
120
|
+
if ([...originalQ, ...originalContent].some((line) => line.length > MAX_LINE_CHARS)) return null;
|
|
71
121
|
|
|
122
|
+
const allQLines = originalQ.map(normalizeForFuzzyMatch);
|
|
123
|
+
const cLines = originalContent.map(normalizeForFuzzyMatch);
|
|
72
124
|
const truncated = allQLines.length > MAX_COMPARE_LINES;
|
|
73
125
|
const q = truncated ? allQLines.slice(0, MAX_COMPARE_LINES) : allQLines;
|
|
126
|
+
const qOriginal = truncated ? originalQ.slice(0, MAX_COMPARE_LINES) : originalQ;
|
|
74
127
|
const L = q.length;
|
|
128
|
+
const sizes = [...new Set([L - 1, L, L + 1].filter((size) => size >= 1 && size <= cLines.length))];
|
|
129
|
+
const preparedQ = q.map(prepareLine);
|
|
130
|
+
const preparedContent = cLines.map(prepareLine);
|
|
131
|
+
const candidates: CandidateWindow[] = [];
|
|
132
|
+
let discardedMaxPositionalScore = -1;
|
|
133
|
+
const budget = new WorkBudget(MAX_SIMILARITY_WORK);
|
|
75
134
|
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
const score = scoreWindow(cLines, q, s, w, L);
|
|
87
|
-
if (score > bestScore) {
|
|
88
|
-
bestScore = score;
|
|
89
|
-
bestStart = s;
|
|
90
|
-
bestSize = w;
|
|
135
|
+
try {
|
|
136
|
+
for (const size of sizes) {
|
|
137
|
+
for (let start = 0; start + size <= cLines.length; start++) {
|
|
138
|
+
const pairs = Math.min(L, size);
|
|
139
|
+
let sum = 0;
|
|
140
|
+
for (let i = 0; i < pairs; i++) {
|
|
141
|
+
sum += preparedSimilarity(preparedQ[i], preparedContent[start + i], budget);
|
|
142
|
+
}
|
|
143
|
+
const discarded = keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
|
|
144
|
+
if (discarded !== undefined) discardedMaxPositionalScore = Math.max(discardedMaxPositionalScore, discarded);
|
|
91
145
|
}
|
|
92
146
|
}
|
|
93
|
-
}
|
|
94
147
|
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
148
|
+
const scored: ScoredWindow[] = candidates.map((candidate) => ({
|
|
149
|
+
...candidate,
|
|
150
|
+
score: sequenceSimilarity(
|
|
151
|
+
preparedQ,
|
|
152
|
+
preparedContent.slice(candidate.start, candidate.start + candidate.size),
|
|
153
|
+
budget,
|
|
154
|
+
),
|
|
155
|
+
}));
|
|
156
|
+
scored.sort(
|
|
157
|
+
(a, b) => b.score - a.score || b.positionalScore - a.positionalScore || a.start - b.start || a.size - b.size,
|
|
158
|
+
);
|
|
159
|
+
const best = scored[0];
|
|
160
|
+
if (!best || best.score < MIN_SCORE) return null;
|
|
161
|
+
const bestEnd = best.start + best.size;
|
|
162
|
+
const runner = scored.find((candidate) => {
|
|
163
|
+
const candidateEnd = candidate.start + candidate.size;
|
|
164
|
+
return candidateEnd <= best.start || candidate.start >= bestEnd;
|
|
165
|
+
});
|
|
110
166
|
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
167
|
+
const windowLines = cLines.slice(best.start, bestEnd);
|
|
168
|
+
const windowOriginal = originalContent.slice(best.start, bestEnd);
|
|
169
|
+
const ops = alignLines(q, windowLines, qOriginal, windowOriginal, best.start, budget);
|
|
170
|
+
return {
|
|
171
|
+
startLine: best.start + 1,
|
|
172
|
+
endLine: bestEnd,
|
|
173
|
+
score: best.score,
|
|
174
|
+
ops,
|
|
175
|
+
equalCount: ops.filter((op) => op.type === "equal").length,
|
|
176
|
+
totalOldLines: L,
|
|
177
|
+
truncated,
|
|
178
|
+
competitor: runner
|
|
179
|
+
? { startLine: runner.start + 1, endLine: runner.start + runner.size, score: runner.score }
|
|
180
|
+
: undefined,
|
|
181
|
+
competitionIncomplete: discardedMaxPositionalScore >= best.positionalScore - MIN_DIRECT_POSITIONAL_GAP,
|
|
182
|
+
};
|
|
183
|
+
} catch (error) {
|
|
184
|
+
if (error instanceof Error && error.message === "similarity-work-budget-exhausted") return null;
|
|
185
|
+
throw error;
|
|
118
186
|
}
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
if (best === 1) break;
|
|
130
|
-
}
|
|
131
|
-
bagSum += best;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const MIN_DIRECT_POSITIONAL_GAP = 0.1;
|
|
190
|
+
|
|
191
|
+
/** Return the positional score of any candidate discarded by the bounded heap. */
|
|
192
|
+
function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): number | undefined {
|
|
193
|
+
if (candidates.length < MAX_CANDIDATE_WINDOWS) {
|
|
194
|
+
candidates.push(candidate);
|
|
195
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
196
|
+
return undefined;
|
|
132
197
|
}
|
|
133
|
-
|
|
198
|
+
const worst = candidates[0];
|
|
199
|
+
if (
|
|
200
|
+
candidate.positionalScore < worst.positionalScore ||
|
|
201
|
+
(candidate.positionalScore === worst.positionalScore && candidate.start >= worst.start)
|
|
202
|
+
) return candidate.positionalScore;
|
|
203
|
+
candidates[0] = candidate;
|
|
204
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
205
|
+
return worst.positionalScore;
|
|
134
206
|
}
|
|
135
207
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
[s + w - 1, mid],
|
|
145
|
-
[s + Math.floor(w / 2), mid],
|
|
146
|
-
];
|
|
147
|
-
for (const [ci, qi] of probes) {
|
|
148
|
-
if (ci >= 0 && ci < cLines.length && qi >= 0 && qi < L) {
|
|
149
|
-
if (lineSimilarity(cLines[ci], q[qi]) >= 0.45) return true;
|
|
208
|
+
/** Weighted LCS: lines are matched at most once and in source order. */
|
|
209
|
+
function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[], budget: WorkBudget): number {
|
|
210
|
+
let previous = new Float64Array(window.length + 1);
|
|
211
|
+
for (let i = 1; i <= query.length; i++) {
|
|
212
|
+
const current = new Float64Array(window.length + 1);
|
|
213
|
+
for (let j = 1; j <= window.length; j++) {
|
|
214
|
+
const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1], budget);
|
|
215
|
+
current[j] = Math.max(previous[j], current[j - 1], matched);
|
|
150
216
|
}
|
|
217
|
+
previous = current;
|
|
151
218
|
}
|
|
152
|
-
return
|
|
219
|
+
return previous[window.length] / Math.max(query.length, window.length);
|
|
153
220
|
}
|
|
154
221
|
|
|
155
|
-
/**
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
222
|
+
/** Align fuzzy lines while attaching original strings to every operation. */
|
|
223
|
+
function alignLines(
|
|
224
|
+
q: string[],
|
|
225
|
+
w: string[],
|
|
226
|
+
qOriginal: string[],
|
|
227
|
+
wOriginal: string[],
|
|
228
|
+
wOffset: number,
|
|
229
|
+
budget: WorkBudget,
|
|
230
|
+
): AlignOp[] {
|
|
161
231
|
const n = q.length;
|
|
162
232
|
const m = w.length;
|
|
233
|
+
const preparedQ = q.map(prepareLine);
|
|
234
|
+
const preparedW = w.map(prepareLine);
|
|
235
|
+
const similarities = Array.from({ length: n }, (_, i) =>
|
|
236
|
+
Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j], budget)),
|
|
237
|
+
);
|
|
163
238
|
const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
|
|
164
239
|
for (let i = n - 1; i >= 0; i--) {
|
|
165
240
|
for (let j = m - 1; j >= 0; j--) {
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
dp[i][j] = Math.max(dp[i + 1][j], dp[i][j + 1]);
|
|
170
|
-
}
|
|
241
|
+
dp[i][j] = similarities[i][j] >= MATCH_THRESHOLD
|
|
242
|
+
? dp[i + 1][j + 1] + 1
|
|
243
|
+
: Math.max(dp[i + 1][j], dp[i][j + 1]);
|
|
171
244
|
}
|
|
172
245
|
}
|
|
173
246
|
|
|
@@ -175,54 +248,40 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
|
|
|
175
248
|
let i = 0;
|
|
176
249
|
let j = 0;
|
|
177
250
|
while (i < n && j < m) {
|
|
178
|
-
if (
|
|
179
|
-
|
|
251
|
+
if (similarities[i][j] >= MATCH_THRESHOLD) {
|
|
252
|
+
const exact = qOriginal[i] === wOriginal[j];
|
|
253
|
+
raw.push({
|
|
254
|
+
type: exact ? "equal" : "similar",
|
|
255
|
+
fileLine: wOffset + j + 1,
|
|
256
|
+
fileText: wOriginal[j],
|
|
257
|
+
oldLine: i + 1,
|
|
258
|
+
oldText: qOriginal[i],
|
|
259
|
+
similarity: similarities[i][j],
|
|
260
|
+
});
|
|
180
261
|
i++;
|
|
181
262
|
j++;
|
|
182
263
|
} else if (dp[i + 1][j] > dp[i][j + 1]) {
|
|
183
|
-
|
|
184
|
-
// the file line so inserted file lines surface as file-only ops.
|
|
185
|
-
raw.push({ type: "old-only", oldLine: i + 1, oldText: q[i] });
|
|
186
|
-
i++;
|
|
264
|
+
raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
187
265
|
} else {
|
|
188
|
-
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText:
|
|
189
|
-
j++;
|
|
266
|
+
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
190
267
|
}
|
|
191
268
|
}
|
|
192
|
-
while (i < n) {
|
|
193
|
-
|
|
194
|
-
i++;
|
|
195
|
-
}
|
|
196
|
-
while (j < m) {
|
|
197
|
-
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: w[j] });
|
|
198
|
-
j++;
|
|
199
|
-
}
|
|
269
|
+
while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
270
|
+
while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
200
271
|
|
|
201
|
-
// Merge adjacent old-only/file-only runs into pairwise "changed" ops.
|
|
202
272
|
const merged: AlignOp[] = [];
|
|
203
273
|
for (let k = 0; k < raw.length; k++) {
|
|
204
274
|
const op = raw[k];
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
continue;
|
|
214
|
-
}
|
|
215
|
-
if (op.type === "file-only") {
|
|
216
|
-
const next = raw[k + 1];
|
|
217
|
-
if (next && next.type === "old-only") {
|
|
218
|
-
merged.push({ type: "changed", fileLine: op.fileLine, fileText: op.fileText, oldLine: next.oldLine, oldText: next.oldText });
|
|
219
|
-
k++;
|
|
220
|
-
continue;
|
|
221
|
-
}
|
|
275
|
+
const next = raw[k + 1];
|
|
276
|
+
if (op.type === "old-only" && next?.type === "file-only") {
|
|
277
|
+
merged.push({ type: "changed", fileLine: next.fileLine, fileText: next.fileText, oldLine: op.oldLine, oldText: op.oldText });
|
|
278
|
+
k++;
|
|
279
|
+
} else if (op.type === "file-only" && next?.type === "old-only") {
|
|
280
|
+
merged.push({ type: "changed", fileLine: op.fileLine, fileText: op.fileText, oldLine: next.oldLine, oldText: next.oldText });
|
|
281
|
+
k++;
|
|
282
|
+
} else {
|
|
222
283
|
merged.push(op);
|
|
223
|
-
continue;
|
|
224
284
|
}
|
|
225
|
-
merged.push(op);
|
|
226
285
|
}
|
|
227
286
|
return merged;
|
|
228
287
|
}
|
|
@@ -230,20 +289,15 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
|
|
|
230
289
|
/** Cheap probes for common causes of a failed match. */
|
|
231
290
|
export function probeMatchCauses(content: string, oldText: string): string[] {
|
|
232
291
|
const causes: string[] = [];
|
|
233
|
-
const stripWS = (
|
|
292
|
+
const stripWS = (value: string) => value.replace(/\s+/g, "");
|
|
234
293
|
if (stripWS(content).includes(stripWS(oldText))) {
|
|
235
|
-
causes.push(
|
|
236
|
-
"whitespace mismatch: the text matches when ALL whitespace is removed — check tabs vs spaces and indentation width",
|
|
237
|
-
);
|
|
294
|
+
causes.push("whitespace mismatch: the text matches when ALL whitespace is removed — check tabs vs spaces and indentation width");
|
|
238
295
|
} else if (content.toLowerCase().includes(oldText.toLowerCase())) {
|
|
239
296
|
causes.push("letter-case mismatch: the text matches case-insensitively");
|
|
240
297
|
}
|
|
241
298
|
const fileHasTabs = content.includes("\t");
|
|
242
299
|
const oldHasTabs = oldText.includes("\t");
|
|
243
|
-
if (fileHasTabs && !oldHasTabs)
|
|
244
|
-
|
|
245
|
-
} else if (!fileHasTabs && oldHasTabs) {
|
|
246
|
-
causes.push("your oldText contains tab characters but the file uses spaces");
|
|
247
|
-
}
|
|
300
|
+
if (fileHasTabs && !oldHasTabs) causes.push("the file contains tab characters but your oldText uses spaces");
|
|
301
|
+
else if (!fileHasTabs && oldHasTabs) causes.push("your oldText contains tab characters but the file uses spaces");
|
|
248
302
|
return causes;
|
|
249
303
|
}
|
package/src/text.ts
CHANGED
|
@@ -71,6 +71,17 @@ export function splitLinesWithEndings(content: string): string[] {
|
|
|
71
71
|
return content.match(/[^\n]*\n|[^\n]+/g) ?? [];
|
|
72
72
|
}
|
|
73
73
|
|
|
74
|
+
/**
|
|
75
|
+
* Split every logical line while retaining endings. Unlike
|
|
76
|
+
* splitLinesWithEndings(), this includes the zero-length logical line after a
|
|
77
|
+
* terminal newline. That distinction is needed when fuzzy normalization turns
|
|
78
|
+
* a whitespace-only final line into an empty line.
|
|
79
|
+
*/
|
|
80
|
+
export function splitLogicalLinesWithEndings(content: string): string[] {
|
|
81
|
+
const parts = content.split("\n");
|
|
82
|
+
return parts.map((part, index) => (index < parts.length - 1 ? `${part}\n` : part));
|
|
83
|
+
}
|
|
84
|
+
|
|
74
85
|
export function getLineSpans(content: string): LineSpan[] {
|
|
75
86
|
let offset = 0;
|
|
76
87
|
return splitLinesWithEndings(content).map((line) => {
|
|
@@ -80,6 +91,16 @@ export function getLineSpans(content: string): LineSpan[] {
|
|
|
80
91
|
});
|
|
81
92
|
}
|
|
82
93
|
|
|
94
|
+
/** Logical spans, including a possible zero-length final line. */
|
|
95
|
+
export function getLogicalLineSpans(content: string): LineSpan[] {
|
|
96
|
+
let offset = 0;
|
|
97
|
+
return splitLogicalLinesWithEndings(content).map((line) => {
|
|
98
|
+
const span = { start: offset, end: offset + line.length };
|
|
99
|
+
offset = span.end;
|
|
100
|
+
return span;
|
|
101
|
+
});
|
|
102
|
+
}
|
|
103
|
+
|
|
83
104
|
/** 0-based index of the line containing `offset` (clamped to the last line). */
|
|
84
105
|
export function lineAt(spans: LineSpan[], offset: number): number {
|
|
85
106
|
let lo = 0;
|
|
@@ -102,17 +123,25 @@ export function lineAt(spans: LineSpan[], offset: number): number {
|
|
|
102
123
|
* same normalization the matching engine uses for its uniqueness check.
|
|
103
124
|
* `fuzzyContent` must already be `normalizeForFuzzyMatch`-ed.
|
|
104
125
|
*/
|
|
105
|
-
export function countFuzzyOccurrences(fuzzyContent: string, needle: string): number {
|
|
106
|
-
|
|
107
|
-
|
|
126
|
+
export function countFuzzyOccurrences(fuzzyContent: string, needle: string, stopAfter = Number.POSITIVE_INFINITY): number {
|
|
127
|
+
const normalizedNeedle = normalizeForFuzzyMatch(needle);
|
|
128
|
+
if (!normalizedNeedle) return 0;
|
|
129
|
+
let count = 0;
|
|
130
|
+
let index = fuzzyContent.indexOf(normalizedNeedle);
|
|
131
|
+
while (index !== -1) {
|
|
132
|
+
count++;
|
|
133
|
+
if (count >= stopAfter) return count;
|
|
134
|
+
index = fuzzyContent.indexOf(normalizedNeedle, index + normalizedNeedle.length);
|
|
135
|
+
}
|
|
136
|
+
return count;
|
|
108
137
|
}
|
|
109
138
|
|
|
110
|
-
/** All start offsets of `needle
|
|
111
|
-
export function findAllOccurrences(haystack: string, needle: string): number[] {
|
|
139
|
+
/** All non-overlapping start offsets of `needle`, optionally bounded. */
|
|
140
|
+
export function findAllOccurrences(haystack: string, needle: string, limit = Number.POSITIVE_INFINITY): number[] {
|
|
112
141
|
const offsets: number[] = [];
|
|
113
|
-
if (!needle) return offsets;
|
|
142
|
+
if (!needle || limit <= 0) return offsets;
|
|
114
143
|
let idx = haystack.indexOf(needle);
|
|
115
|
-
while (idx !== -1) {
|
|
144
|
+
while (idx !== -1 && offsets.length < limit) {
|
|
116
145
|
offsets.push(idx);
|
|
117
146
|
idx = haystack.indexOf(needle, idx + needle.length);
|
|
118
147
|
}
|