@khanhicetea/pi-better-tool 0.1.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/similarity.ts CHANGED
@@ -1,173 +1,246 @@
1
1
  /**
2
- * Fuzzy line similarity used to locate the region of the file that a failed
3
- * oldText *almost* matched, so the tool can show the model the actual bytes
4
- * instead of forcing a full re-read.
2
+ * Bounded fuzzy line similarity used to locate the region of a file that a
3
+ * failed oldText almost matched.
5
4
  */
6
5
 
7
- import { toFuzzyLines } from "./text.ts";
6
+ import { normalizeForFuzzyMatch, normalizeToLF } from "./text.ts";
8
7
 
9
8
  export interface AlignOp {
10
- type: "equal" | "changed" | "file-only" | "old-only";
11
- /** 1-based line number in the file (for equal/changed/file-only). */
9
+ type: "equal" | "similar" | "changed" | "file-only" | "old-only";
10
+ /** 1-based line number in the file (for equal/similar/changed/file-only). */
12
11
  fileLine?: number;
12
+ /** Original LF-normalized file text, never fuzzy-normalized. */
13
13
  fileText?: string;
14
14
  /** 1-based line number in the model-provided oldText. */
15
15
  oldLine?: number;
16
+ /** Original LF-normalized oldText, never fuzzy-normalized. */
16
17
  oldText?: string;
18
+ /** Heuristic similarity for aligned pairs. */
19
+ similarity?: number;
17
20
  }
18
21
 
19
22
  export interface ClosestRegion {
20
23
  startLine: number;
21
24
  endLine: number;
22
- /** 0..1 mean line similarity. */
25
+ /** 0..1 order-preserving, one-to-one line similarity. */
23
26
  score: number;
24
27
  ops: AlignOp[];
28
+ /** Number of exactly equal original lines. */
25
29
  equalCount: number;
26
30
  totalOldLines: number;
27
31
  /** oldText was truncated for comparison (very large oldText). */
28
32
  truncated: boolean;
33
+ /** Best distinct, non-overlapping competitor, when retained by the bounded search. */
34
+ competitor?: { startLine: number; endLine: number; score: number };
35
+ /** A discarded window was too competitive to prove a safe score margin. */
36
+ competitionIncomplete: boolean;
37
+ }
38
+
39
+ interface PreparedLine {
40
+ text: string;
41
+ grams: Map<string, number>;
42
+ gramCount: number;
43
+ }
44
+
45
+ function prepareLine(text: string): PreparedLine {
46
+ const grams = new Map<string, number>();
47
+ for (let i = 0; i + 1 < text.length; i++) {
48
+ const gram = text.slice(i, i + 2);
49
+ grams.set(gram, (grams.get(gram) ?? 0) + 1);
50
+ }
51
+ return { text, grams, gramCount: Math.max(0, text.length - 1) };
52
+ }
53
+
54
+ class WorkBudget {
55
+ private remaining: number;
56
+ constructor(remaining: number) {
57
+ this.remaining = remaining;
58
+ }
59
+ spend(amount: number): boolean {
60
+ this.remaining -= amount;
61
+ return this.remaining >= 0;
62
+ }
29
63
  }
30
64
 
31
65
  /** Dice coefficient over character bigrams; 1.0 for identical strings. */
32
66
  export function lineSimilarity(a: string, b: string): number {
33
- if (a === b) return 1;
34
- if (a.length < 2 || b.length < 2) return 0;
35
- const gramsA = bigramCounts(a);
36
- const gramsB = bigramCounts(b);
37
- let overlap = 0;
38
- let total = 0;
39
- for (const count of gramsA.values()) total += count;
40
- for (const count of gramsB.values()) total += count;
41
- for (const [gram, countA] of gramsA) {
42
- const countB = gramsB.get(gram);
43
- if (countB) overlap += Math.min(countA, countB);
44
- }
45
- return (2 * overlap) / total;
67
+ return preparedSimilarity(prepareLine(a), prepareLine(b));
46
68
  }
47
69
 
48
- function bigramCounts(s: string): Map<string, number> {
49
- const counts = new Map<string, number>();
50
- for (let i = 0; i + 1 < s.length; i++) {
51
- const gram = s.slice(i, i + 2);
52
- counts.set(gram, (counts.get(gram) ?? 0) + 1);
70
+ function preparedSimilarity(a: PreparedLine, b: PreparedLine, budget?: WorkBudget): number {
71
+ if (budget && !budget.spend(1)) throw new Error("similarity-work-budget-exhausted");
72
+ if (a.text === b.text) return 1;
73
+ if (a.gramCount === 0 || b.gramCount === 0) return 0;
74
+ const [small, large] = a.grams.size <= b.grams.size ? [a.grams, b.grams] : [b.grams, a.grams];
75
+ if (budget && !budget.spend(small.size)) throw new Error("similarity-work-budget-exhausted");
76
+ let overlap = 0;
77
+ for (const [gram, count] of small) {
78
+ const other = large.get(gram);
79
+ if (other) overlap += Math.min(count, other);
53
80
  }
54
- return counts;
81
+ return (2 * overlap) / (a.gramCount + b.gramCount);
55
82
  }
56
83
 
57
84
  const MAX_COMPARE_LINES = 300;
58
- const PREFILTER_FILE_LINES = 2000;
85
+ const MAX_SEARCH_FILE_LINES = 10_000;
86
+ const MAX_CANDIDATE_WINDOWS = 48;
87
+ const MAX_QUERY_BYTES = 128 * 1024;
88
+ const MAX_LINE_CHARS = 16 * 1024;
89
+ const MAX_SIMILARITY_WORK = 20_000_000;
59
90
  const MIN_SCORE = 0.3;
60
91
  const MATCH_THRESHOLD = 0.75;
61
92
 
93
+ interface CandidateWindow {
94
+ start: number;
95
+ size: number;
96
+ positionalScore: number;
97
+ }
98
+
99
+ interface ScoredWindow extends CandidateWindow {
100
+ score: number;
101
+ }
102
+
103
+ function originalLines(text: string): string[] {
104
+ const lines = normalizeToLF(text).split("\n");
105
+ // Remove only the synthetic segment after a terminal newline. Do not remove
106
+ // a real, unterminated whitespace-only line.
107
+ if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
108
+ return lines;
109
+ }
110
+
62
111
  /**
63
- * Find the window of file lines most similar to oldText.
64
- * Compares window sizes of len-1, len, len+1 so a single extra/missing line
65
- * still lands the search on the right region.
112
+ * Find the best window and a distinct runner-up under explicit input/work
113
+ * budgets. Returning null is the safe fallback when a budget is exhausted.
66
114
  */
67
115
  export function findClosestRegion(content: string, oldText: string): ClosestRegion | null {
68
- const allQLines = toFuzzyLines(oldText);
69
- const cLines = toFuzzyLines(content);
70
- if (allQLines.length === 0 || cLines.length === 0) return null;
116
+ if (Buffer.byteLength(oldText, "utf8") > MAX_QUERY_BYTES) return null;
117
+ const originalQ = originalLines(oldText);
118
+ const originalContent = originalLines(content);
119
+ if (originalQ.length === 0 || originalContent.length === 0 || originalContent.length > MAX_SEARCH_FILE_LINES) return null;
120
+ if ([...originalQ, ...originalContent].some((line) => line.length > MAX_LINE_CHARS)) return null;
71
121
 
122
+ const allQLines = originalQ.map(normalizeForFuzzyMatch);
123
+ const cLines = originalContent.map(normalizeForFuzzyMatch);
72
124
  const truncated = allQLines.length > MAX_COMPARE_LINES;
73
125
  const q = truncated ? allQLines.slice(0, MAX_COMPARE_LINES) : allQLines;
126
+ const qOriginal = truncated ? originalQ.slice(0, MAX_COMPARE_LINES) : originalQ;
74
127
  const L = q.length;
128
+ const sizes = [...new Set([L - 1, L, L + 1].filter((size) => size >= 1 && size <= cLines.length))];
129
+ const preparedQ = q.map(prepareLine);
130
+ const preparedContent = cLines.map(prepareLine);
131
+ const candidates: CandidateWindow[] = [];
132
+ let discardedMaxPositionalScore = -1;
133
+ const budget = new WorkBudget(MAX_SIMILARITY_WORK);
75
134
 
76
- const sizes = [...new Set([L - 1, L, L + 1].filter((s) => s >= 1 && s <= cLines.length))];
77
- const bigFile = cLines.length > PREFILTER_FILE_LINES;
78
-
79
- let bestScore = -1;
80
- let bestStart = -1;
81
- let bestSize = L;
82
-
83
- for (const w of sizes) {
84
- for (let s = 0; s + w <= cLines.length; s++) {
85
- if (bigFile && !prefilter(cLines, q, s, w, L)) continue;
86
- const score = scoreWindow(cLines, q, s, w, L);
87
- if (score > bestScore) {
88
- bestScore = score;
89
- bestStart = s;
90
- bestSize = w;
135
+ try {
136
+ for (const size of sizes) {
137
+ for (let start = 0; start + size <= cLines.length; start++) {
138
+ const pairs = Math.min(L, size);
139
+ let sum = 0;
140
+ for (let i = 0; i < pairs; i++) {
141
+ sum += preparedSimilarity(preparedQ[i], preparedContent[start + i], budget);
142
+ }
143
+ const discarded = keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
144
+ if (discarded !== undefined) discardedMaxPositionalScore = Math.max(discardedMaxPositionalScore, discarded);
91
145
  }
92
146
  }
93
- }
94
147
 
95
- if (bestStart < 0 || bestScore < MIN_SCORE) return null;
96
-
97
- const windowLines = cLines.slice(bestStart, bestStart + bestSize);
98
- const ops = alignLines(q, windowLines, bestStart);
99
- const equalCount = ops.filter((op) => op.type === "equal").length;
100
- return {
101
- startLine: bestStart + 1,
102
- endLine: bestStart + bestSize,
103
- score: bestScore,
104
- ops,
105
- equalCount,
106
- totalOldLines: L,
107
- truncated,
108
- };
109
- }
148
+ const scored: ScoredWindow[] = candidates.map((candidate) => ({
149
+ ...candidate,
150
+ score: sequenceSimilarity(
151
+ preparedQ,
152
+ preparedContent.slice(candidate.start, candidate.start + candidate.size),
153
+ budget,
154
+ ),
155
+ }));
156
+ scored.sort(
157
+ (a, b) => b.score - a.score || b.positionalScore - a.positionalScore || a.start - b.start || a.size - b.size,
158
+ );
159
+ const best = scored[0];
160
+ if (!best || best.score < MIN_SCORE) return null;
161
+ const bestEnd = best.start + best.size;
162
+ const runner = scored.find((candidate) => {
163
+ const candidateEnd = candidate.start + candidate.size;
164
+ return candidateEnd <= best.start || candidate.start >= bestEnd;
165
+ });
110
166
 
111
- function scoreWindow(cLines: string[], q: string[], start: number, w: number, L: number): number {
112
- // Positional score: strict index-by-index pairing (handles indentation /
113
- // trailing-whitespace drift).
114
- const pairs = Math.min(L, w);
115
- let positionalSum = 0;
116
- for (let i = 0; i < pairs; i++) {
117
- positionalSum += lineSimilarity(q[i], cLines[start + i]);
167
+ const windowLines = cLines.slice(best.start, bestEnd);
168
+ const windowOriginal = originalContent.slice(best.start, bestEnd);
169
+ const ops = alignLines(q, windowLines, qOriginal, windowOriginal, best.start, budget);
170
+ return {
171
+ startLine: best.start + 1,
172
+ endLine: bestEnd,
173
+ score: best.score,
174
+ ops,
175
+ equalCount: ops.filter((op) => op.type === "equal").length,
176
+ totalOldLines: L,
177
+ truncated,
178
+ competitor: runner
179
+ ? { startLine: runner.start + 1, endLine: runner.start + runner.size, score: runner.score }
180
+ : undefined,
181
+ competitionIncomplete: discardedMaxPositionalScore >= best.positionalScore - MIN_DIRECT_POSITIONAL_GAP,
182
+ };
183
+ } catch (error) {
184
+ if (error instanceof Error && error.message === "similarity-work-budget-exhausted") return null;
185
+ throw error;
118
186
  }
119
- const positional = positionalSum / Math.max(L, w);
120
-
121
- // Bag score: best match per oldText line anywhere in the window (handles a
122
- // single inserted/removed line in the middle that breaks positional pairing).
123
- let bagSum = 0;
124
- for (let i = 0; i < L; i++) {
125
- let best = 0;
126
- for (let j = start; j < start + w; j++) {
127
- const sim = lineSimilarity(q[i], cLines[j]);
128
- if (sim > best) best = sim;
129
- if (best === 1) break;
130
- }
131
- bagSum += best;
187
+ }
188
+
189
+ const MIN_DIRECT_POSITIONAL_GAP = 0.1;
190
+
191
+ /** Return the positional score of any candidate discarded by the bounded heap. */
192
+ function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): number | undefined {
193
+ if (candidates.length < MAX_CANDIDATE_WINDOWS) {
194
+ candidates.push(candidate);
195
+ candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
196
+ return undefined;
132
197
  }
133
- return Math.max(positional, (bagSum / L) * 0.95);
198
+ const worst = candidates[0];
199
+ if (
200
+ candidate.positionalScore < worst.positionalScore ||
201
+ (candidate.positionalScore === worst.positionalScore && candidate.start >= worst.start)
202
+ ) return candidate.positionalScore;
203
+ candidates[0] = candidate;
204
+ candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
205
+ return worst.positionalScore;
134
206
  }
135
207
 
136
- function prefilter(cLines: string[], q: string[], s: number, w: number, L: number): boolean {
137
- const mid = Math.floor(L / 2);
138
- const probes: Array<[number, number]> = [
139
- [s, 0],
140
- [s, L - 1],
141
- [s, mid],
142
- [s + w - 1, L - 1],
143
- [s + w - 1, 0],
144
- [s + w - 1, mid],
145
- [s + Math.floor(w / 2), mid],
146
- ];
147
- for (const [ci, qi] of probes) {
148
- if (ci >= 0 && ci < cLines.length && qi >= 0 && qi < L) {
149
- if (lineSimilarity(cLines[ci], q[qi]) >= 0.45) return true;
208
+ /** Weighted LCS: lines are matched at most once and in source order. */
209
+ function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[], budget: WorkBudget): number {
210
+ let previous = new Float64Array(window.length + 1);
211
+ for (let i = 1; i <= query.length; i++) {
212
+ const current = new Float64Array(window.length + 1);
213
+ for (let j = 1; j <= window.length; j++) {
214
+ const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1], budget);
215
+ current[j] = Math.max(previous[j], current[j - 1], matched);
150
216
  }
217
+ previous = current;
151
218
  }
152
- return false;
219
+ return previous[window.length] / Math.max(query.length, window.length);
153
220
  }
154
221
 
155
- /**
156
- * LCS-style alignment between oldText lines and the best window.
157
- * Lines with similarity >= MATCH_THRESHOLD count as equal; adjacent
158
- * old-only/file-only runs are merged into "changed" pairs.
159
- */
160
- function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
222
+ /** Align fuzzy lines while attaching original strings to every operation. */
223
+ function alignLines(
224
+ q: string[],
225
+ w: string[],
226
+ qOriginal: string[],
227
+ wOriginal: string[],
228
+ wOffset: number,
229
+ budget: WorkBudget,
230
+ ): AlignOp[] {
161
231
  const n = q.length;
162
232
  const m = w.length;
233
+ const preparedQ = q.map(prepareLine);
234
+ const preparedW = w.map(prepareLine);
235
+ const similarities = Array.from({ length: n }, (_, i) =>
236
+ Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j], budget)),
237
+ );
163
238
  const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
164
239
  for (let i = n - 1; i >= 0; i--) {
165
240
  for (let j = m - 1; j >= 0; j--) {
166
- if (lineSimilarity(q[i], w[j]) >= MATCH_THRESHOLD) {
167
- dp[i][j] = dp[i + 1][j + 1] + 1;
168
- } else {
169
- dp[i][j] = Math.max(dp[i + 1][j], dp[i][j + 1]);
170
- }
241
+ dp[i][j] = similarities[i][j] >= MATCH_THRESHOLD
242
+ ? dp[i + 1][j + 1] + 1
243
+ : Math.max(dp[i + 1][j], dp[i][j + 1]);
171
244
  }
172
245
  }
173
246
 
@@ -175,54 +248,40 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
175
248
  let i = 0;
176
249
  let j = 0;
177
250
  while (i < n && j < m) {
178
- if (lineSimilarity(q[i], w[j]) >= MATCH_THRESHOLD) {
179
- raw.push({ type: "equal", fileLine: wOffset + j + 1, fileText: w[j], oldLine: i + 1, oldText: q[i] });
251
+ if (similarities[i][j] >= MATCH_THRESHOLD) {
252
+ const exact = qOriginal[i] === wOriginal[j];
253
+ raw.push({
254
+ type: exact ? "equal" : "similar",
255
+ fileLine: wOffset + j + 1,
256
+ fileText: wOriginal[j],
257
+ oldLine: i + 1,
258
+ oldText: qOriginal[i],
259
+ similarity: similarities[i][j],
260
+ });
180
261
  i++;
181
262
  j++;
182
263
  } else if (dp[i + 1][j] > dp[i][j + 1]) {
183
- // Strictly better to skip the oldText line; on ties prefer skipping
184
- // the file line so inserted file lines surface as file-only ops.
185
- raw.push({ type: "old-only", oldLine: i + 1, oldText: q[i] });
186
- i++;
264
+ raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
187
265
  } else {
188
- raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: w[j] });
189
- j++;
266
+ raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
190
267
  }
191
268
  }
192
- while (i < n) {
193
- raw.push({ type: "old-only", oldLine: i + 1, oldText: q[i] });
194
- i++;
195
- }
196
- while (j < m) {
197
- raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: w[j] });
198
- j++;
199
- }
269
+ while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
270
+ while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
200
271
 
201
- // Merge adjacent old-only/file-only runs into pairwise "changed" ops.
202
272
  const merged: AlignOp[] = [];
203
273
  for (let k = 0; k < raw.length; k++) {
204
274
  const op = raw[k];
205
- if (op.type === "old-only") {
206
- const next = raw[k + 1];
207
- if (next && next.type === "file-only") {
208
- merged.push({ type: "changed", fileLine: next.fileLine, fileText: next.fileText, oldLine: op.oldLine, oldText: op.oldText });
209
- k++;
210
- continue;
211
- }
212
- merged.push(op);
213
- continue;
214
- }
215
- if (op.type === "file-only") {
216
- const next = raw[k + 1];
217
- if (next && next.type === "old-only") {
218
- merged.push({ type: "changed", fileLine: op.fileLine, fileText: op.fileText, oldLine: next.oldLine, oldText: next.oldText });
219
- k++;
220
- continue;
221
- }
275
+ const next = raw[k + 1];
276
+ if (op.type === "old-only" && next?.type === "file-only") {
277
+ merged.push({ type: "changed", fileLine: next.fileLine, fileText: next.fileText, oldLine: op.oldLine, oldText: op.oldText });
278
+ k++;
279
+ } else if (op.type === "file-only" && next?.type === "old-only") {
280
+ merged.push({ type: "changed", fileLine: op.fileLine, fileText: op.fileText, oldLine: next.oldLine, oldText: next.oldText });
281
+ k++;
282
+ } else {
222
283
  merged.push(op);
223
- continue;
224
284
  }
225
- merged.push(op);
226
285
  }
227
286
  return merged;
228
287
  }
@@ -230,20 +289,15 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
230
289
  /** Cheap probes for common causes of a failed match. */
231
290
  export function probeMatchCauses(content: string, oldText: string): string[] {
232
291
  const causes: string[] = [];
233
- const stripWS = (s: string) => s.replace(/\s+/g, "");
292
+ const stripWS = (value: string) => value.replace(/\s+/g, "");
234
293
  if (stripWS(content).includes(stripWS(oldText))) {
235
- causes.push(
236
- "whitespace mismatch: the text matches when ALL whitespace is removed — check tabs vs spaces and indentation width",
237
- );
294
+ causes.push("whitespace mismatch: the text matches when ALL whitespace is removed — check tabs vs spaces and indentation width");
238
295
  } else if (content.toLowerCase().includes(oldText.toLowerCase())) {
239
296
  causes.push("letter-case mismatch: the text matches case-insensitively");
240
297
  }
241
298
  const fileHasTabs = content.includes("\t");
242
299
  const oldHasTabs = oldText.includes("\t");
243
- if (fileHasTabs && !oldHasTabs) {
244
- causes.push("the file contains tab characters but your oldText uses spaces");
245
- } else if (!fileHasTabs && oldHasTabs) {
246
- causes.push("your oldText contains tab characters but the file uses spaces");
247
- }
300
+ if (fileHasTabs && !oldHasTabs) causes.push("the file contains tab characters but your oldText uses spaces");
301
+ else if (!fileHasTabs && oldHasTabs) causes.push("your oldText contains tab characters but the file uses spaces");
248
302
  return causes;
249
303
  }
package/src/text.ts CHANGED
@@ -71,6 +71,17 @@ export function splitLinesWithEndings(content: string): string[] {
71
71
  return content.match(/[^\n]*\n|[^\n]+/g) ?? [];
72
72
  }
73
73
 
74
+ /**
75
+ * Split every logical line while retaining endings. Unlike
76
+ * splitLinesWithEndings(), this includes the zero-length logical line after a
77
+ * terminal newline. That distinction is needed when fuzzy normalization turns
78
+ * a whitespace-only final line into an empty line.
79
+ */
80
+ export function splitLogicalLinesWithEndings(content: string): string[] {
81
+ const parts = content.split("\n");
82
+ return parts.map((part, index) => (index < parts.length - 1 ? `${part}\n` : part));
83
+ }
84
+
74
85
  export function getLineSpans(content: string): LineSpan[] {
75
86
  let offset = 0;
76
87
  return splitLinesWithEndings(content).map((line) => {
@@ -80,6 +91,16 @@ export function getLineSpans(content: string): LineSpan[] {
80
91
  });
81
92
  }
82
93
 
94
+ /** Logical spans, including a possible zero-length final line. */
95
+ export function getLogicalLineSpans(content: string): LineSpan[] {
96
+ let offset = 0;
97
+ return splitLogicalLinesWithEndings(content).map((line) => {
98
+ const span = { start: offset, end: offset + line.length };
99
+ offset = span.end;
100
+ return span;
101
+ });
102
+ }
103
+
83
104
  /** 0-based index of the line containing `offset` (clamped to the last line). */
84
105
  export function lineAt(spans: LineSpan[], offset: number): number {
85
106
  let lo = 0;
@@ -102,17 +123,25 @@ export function lineAt(spans: LineSpan[], offset: number): number {
102
123
  * same normalization the matching engine uses for its uniqueness check.
103
124
  * `fuzzyContent` must already be `normalizeForFuzzyMatch`-ed.
104
125
  */
105
- export function countFuzzyOccurrences(fuzzyContent: string, needle: string): number {
106
- if (!needle) return 0;
107
- return fuzzyContent.split(normalizeForFuzzyMatch(needle)).length - 1;
126
+ export function countFuzzyOccurrences(fuzzyContent: string, needle: string, stopAfter = Number.POSITIVE_INFINITY): number {
127
+ const normalizedNeedle = normalizeForFuzzyMatch(needle);
128
+ if (!normalizedNeedle) return 0;
129
+ let count = 0;
130
+ let index = fuzzyContent.indexOf(normalizedNeedle);
131
+ while (index !== -1) {
132
+ count++;
133
+ if (count >= stopAfter) return count;
134
+ index = fuzzyContent.indexOf(normalizedNeedle, index + normalizedNeedle.length);
135
+ }
136
+ return count;
108
137
  }
109
138
 
110
- /** All start offsets of `needle` in `haystack` (literal string search). */
111
- export function findAllOccurrences(haystack: string, needle: string): number[] {
139
+ /** All non-overlapping start offsets of `needle`, optionally bounded. */
140
+ export function findAllOccurrences(haystack: string, needle: string, limit = Number.POSITIVE_INFINITY): number[] {
112
141
  const offsets: number[] = [];
113
- if (!needle) return offsets;
142
+ if (!needle || limit <= 0) return offsets;
114
143
  let idx = haystack.indexOf(needle);
115
- while (idx !== -1) {
144
+ while (idx !== -1 && offsets.length < limit) {
116
145
  offsets.push(idx);
117
146
  idx = haystack.indexOf(needle, idx + needle.length);
118
147
  }