@khanhicetea/pi-better-tool 0.2.1 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +69 -47
- package/package.json +1 -1
- package/src/apply.ts +22 -12
- package/src/diagnostics.ts +122 -19
- package/src/index.ts +6 -1
- package/src/read-evidence.ts +56 -32
- package/src/similarity.ts +140 -63
- package/src/text.ts +36 -7
- package/src/tool.ts +176 -78
package/src/read-evidence.ts
CHANGED
|
@@ -11,29 +11,44 @@ export interface ReadEvidence {
|
|
|
11
11
|
/** 0-based, end-exclusive offsets in LF-normalized, BOM-stripped content. */
|
|
12
12
|
startOffset: number;
|
|
13
13
|
endOffset: number;
|
|
14
|
-
/** 1-based inclusive range
|
|
14
|
+
/** 1-based inclusive range represented by the verified read output. */
|
|
15
15
|
startLine: number;
|
|
16
16
|
endLine: number;
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
interface ReadCall {
|
|
20
|
+
id: string;
|
|
20
21
|
path: string;
|
|
21
22
|
offset?: number;
|
|
22
23
|
limit?: number;
|
|
23
24
|
}
|
|
24
25
|
|
|
25
26
|
interface StoredToolResult {
|
|
27
|
+
role: "toolResult";
|
|
26
28
|
toolCallId: string;
|
|
27
29
|
toolName: string;
|
|
28
30
|
isError: boolean;
|
|
29
31
|
content: Array<{ type: string; text?: string }>;
|
|
30
32
|
}
|
|
31
33
|
|
|
34
|
+
interface StoredAssistant {
|
|
35
|
+
role: "assistant";
|
|
36
|
+
content: Array<{ type: string; id?: string; name?: string; arguments?: unknown }>;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
interface ContextEntryLike {
|
|
40
|
+
type: string;
|
|
41
|
+
message?: StoredAssistant | StoredToolResult | { role: string };
|
|
42
|
+
retainedTail?: Array<StoredAssistant | StoredToolResult | { role: string }>;
|
|
43
|
+
}
|
|
44
|
+
|
|
32
45
|
/**
|
|
33
|
-
* Find the newest
|
|
34
|
-
*
|
|
35
|
-
*
|
|
36
|
-
*
|
|
46
|
+
* Find the newest read call for targetPath in Pi's compaction-aware stored
|
|
47
|
+
* context. The call is accepted only when its successful result exactly
|
|
48
|
+
* matches the built-in read formatting for the current LF-normalized content.
|
|
49
|
+
*
|
|
50
|
+
* This establishes stored-context evidence, not proof of the final provider
|
|
51
|
+
* payload after other extensions' context/provider hooks.
|
|
37
52
|
*/
|
|
38
53
|
export async function findLatestReadEvidence(
|
|
39
54
|
sessionManager: Pick<ExtensionContext["sessionManager"], "buildContextEntries"> | undefined,
|
|
@@ -42,49 +57,58 @@ export async function findLatestReadEvidence(
|
|
|
42
57
|
resolvePath: (path: string) => Promise<string>,
|
|
43
58
|
): Promise<ReadEvidence | null> {
|
|
44
59
|
if (!sessionManager) return null;
|
|
45
|
-
const entries = sessionManager.buildContextEntries();
|
|
46
|
-
const
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
60
|
+
const entries = sessionManager.buildContextEntries() as ContextEntryLike[];
|
|
61
|
+
const messages = entries.flatMap((entry) => {
|
|
62
|
+
if (entry.type === "message" && entry.message) return [entry.message];
|
|
63
|
+
if (entry.type === "compaction" && Array.isArray(entry.retainedTail)) return entry.retainedTail;
|
|
64
|
+
return [];
|
|
65
|
+
});
|
|
66
|
+
const calls: ReadCall[] = [];
|
|
67
|
+
const results = new Map<string, StoredToolResult>();
|
|
68
|
+
|
|
69
|
+
for (const message of messages) {
|
|
70
|
+
if (message.role === "assistant") {
|
|
71
|
+
for (const item of (message as StoredAssistant).content) {
|
|
72
|
+
if (item.type !== "toolCall" || item.name !== "read" || typeof item.id !== "string") continue;
|
|
73
|
+
const args = item.arguments as Record<string, unknown> | undefined;
|
|
74
|
+
if (!args || typeof args.path !== "string") continue;
|
|
75
|
+
calls.push({
|
|
76
|
+
id: item.id,
|
|
77
|
+
path: args.path,
|
|
78
|
+
offset: typeof args.offset === "number" ? args.offset : undefined,
|
|
79
|
+
limit: typeof args.limit === "number" ? args.limit : undefined,
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
} else if (message.role === "toolResult") {
|
|
83
|
+
const result = message as StoredToolResult;
|
|
84
|
+
if (result.toolName === "read") results.set(result.toolCallId, result);
|
|
59
85
|
}
|
|
60
86
|
}
|
|
61
87
|
|
|
62
|
-
for (let
|
|
63
|
-
const
|
|
64
|
-
if (entry.type !== "message" || entry.message.role !== "toolResult") continue;
|
|
65
|
-
const result = entry.message as StoredToolResult;
|
|
66
|
-
if (result.toolName !== "read" || result.isError) continue;
|
|
67
|
-
const call = calls.get(result.toolCallId);
|
|
68
|
-
if (!call) continue;
|
|
69
|
-
|
|
88
|
+
for (let index = calls.length - 1; index >= 0; index--) {
|
|
89
|
+
const call = calls[index];
|
|
70
90
|
let readPath: string;
|
|
71
91
|
try {
|
|
72
92
|
readPath = await resolvePath(call.path);
|
|
73
93
|
} catch {
|
|
74
|
-
|
|
94
|
+
// Without a canonical identity we cannot prove that this newer read is
|
|
95
|
+
// unrelated, so fail closed instead of falling back to older intent.
|
|
96
|
+
return null;
|
|
75
97
|
}
|
|
76
98
|
if (readPath !== targetPath) continue;
|
|
77
99
|
|
|
78
|
-
//
|
|
79
|
-
//
|
|
100
|
+
// Never fall back to older intent when the newest same-file read is
|
|
101
|
+
// missing, failed, malformed, or stale.
|
|
102
|
+
const result = results.get(call.id);
|
|
103
|
+
if (!result || result.isError) return null;
|
|
80
104
|
return evidenceFromBuiltinRead(normalizedContent, call, result.content);
|
|
81
105
|
}
|
|
82
106
|
return null;
|
|
83
107
|
}
|
|
84
108
|
|
|
85
|
-
function evidenceFromBuiltinRead(
|
|
109
|
+
export function evidenceFromBuiltinRead(
|
|
86
110
|
content: string,
|
|
87
|
-
call: ReadCall,
|
|
111
|
+
call: Pick<ReadCall, "path" | "offset" | "limit">,
|
|
88
112
|
blocks: Array<{ type: string; text?: string }>,
|
|
89
113
|
): ReadEvidence | null {
|
|
90
114
|
if (blocks.length !== 1 || blocks[0].type !== "text" || typeof blocks[0].text !== "string") return null;
|
package/src/similarity.ts
CHANGED
|
@@ -3,16 +3,20 @@
|
|
|
3
3
|
* failed oldText almost matched.
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
|
-
import {
|
|
6
|
+
import { normalizeForFuzzyMatch, normalizeToLF } from "./text.ts";
|
|
7
7
|
|
|
8
8
|
export interface AlignOp {
|
|
9
|
-
type: "equal" | "changed" | "file-only" | "old-only";
|
|
10
|
-
/** 1-based line number in the file (for equal/changed/file-only). */
|
|
9
|
+
type: "equal" | "similar" | "changed" | "file-only" | "old-only";
|
|
10
|
+
/** 1-based line number in the file (for equal/similar/changed/file-only). */
|
|
11
11
|
fileLine?: number;
|
|
12
|
+
/** Original LF-normalized file text, never fuzzy-normalized. */
|
|
12
13
|
fileText?: string;
|
|
13
14
|
/** 1-based line number in the model-provided oldText. */
|
|
14
15
|
oldLine?: number;
|
|
16
|
+
/** Original LF-normalized oldText, never fuzzy-normalized. */
|
|
15
17
|
oldText?: string;
|
|
18
|
+
/** Heuristic similarity for aligned pairs. */
|
|
19
|
+
similarity?: number;
|
|
16
20
|
}
|
|
17
21
|
|
|
18
22
|
export interface ClosestRegion {
|
|
@@ -21,10 +25,15 @@ export interface ClosestRegion {
|
|
|
21
25
|
/** 0..1 order-preserving, one-to-one line similarity. */
|
|
22
26
|
score: number;
|
|
23
27
|
ops: AlignOp[];
|
|
28
|
+
/** Number of exactly equal original lines. */
|
|
24
29
|
equalCount: number;
|
|
25
30
|
totalOldLines: number;
|
|
26
31
|
/** oldText was truncated for comparison (very large oldText). */
|
|
27
32
|
truncated: boolean;
|
|
33
|
+
/** Best distinct, non-overlapping competitor, when retained by the bounded search. */
|
|
34
|
+
competitor?: { startLine: number; endLine: number; score: number };
|
|
35
|
+
/** A discarded window was too competitive to prove a safe score margin. */
|
|
36
|
+
competitionIncomplete: boolean;
|
|
28
37
|
}
|
|
29
38
|
|
|
30
39
|
interface PreparedLine {
|
|
@@ -42,15 +51,28 @@ function prepareLine(text: string): PreparedLine {
|
|
|
42
51
|
return { text, grams, gramCount: Math.max(0, text.length - 1) };
|
|
43
52
|
}
|
|
44
53
|
|
|
54
|
+
class WorkBudget {
|
|
55
|
+
private remaining: number;
|
|
56
|
+
constructor(remaining: number) {
|
|
57
|
+
this.remaining = remaining;
|
|
58
|
+
}
|
|
59
|
+
spend(amount: number): boolean {
|
|
60
|
+
this.remaining -= amount;
|
|
61
|
+
return this.remaining >= 0;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
45
65
|
/** Dice coefficient over character bigrams; 1.0 for identical strings. */
|
|
46
66
|
export function lineSimilarity(a: string, b: string): number {
|
|
47
67
|
return preparedSimilarity(prepareLine(a), prepareLine(b));
|
|
48
68
|
}
|
|
49
69
|
|
|
50
|
-
function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
|
|
70
|
+
function preparedSimilarity(a: PreparedLine, b: PreparedLine, budget?: WorkBudget): number {
|
|
71
|
+
if (budget && !budget.spend(1)) throw new Error("similarity-work-budget-exhausted");
|
|
51
72
|
if (a.text === b.text) return 1;
|
|
52
73
|
if (a.gramCount === 0 || b.gramCount === 0) return 0;
|
|
53
74
|
const [small, large] = a.grams.size <= b.grams.size ? [a.grams, b.grams] : [b.grams, a.grams];
|
|
75
|
+
if (budget && !budget.spend(small.size)) throw new Error("similarity-work-budget-exhausted");
|
|
54
76
|
let overlap = 0;
|
|
55
77
|
for (const [gram, count] of small) {
|
|
56
78
|
const other = large.get(gram);
|
|
@@ -61,7 +83,10 @@ function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
|
|
|
61
83
|
|
|
62
84
|
const MAX_COMPARE_LINES = 300;
|
|
63
85
|
const MAX_SEARCH_FILE_LINES = 10_000;
|
|
64
|
-
const MAX_CANDIDATE_WINDOWS =
|
|
86
|
+
const MAX_CANDIDATE_WINDOWS = 48;
|
|
87
|
+
const MAX_QUERY_BYTES = 128 * 1024;
|
|
88
|
+
const MAX_LINE_CHARS = 16 * 1024;
|
|
89
|
+
const MAX_SIMILARITY_WORK = 20_000_000;
|
|
65
90
|
const MIN_SCORE = 0.3;
|
|
66
91
|
const MATCH_THRESHOLD = 0.75;
|
|
67
92
|
|
|
@@ -71,81 +96,122 @@ interface CandidateWindow {
|
|
|
71
96
|
positionalScore: number;
|
|
72
97
|
}
|
|
73
98
|
|
|
99
|
+
interface ScoredWindow extends CandidateWindow {
|
|
100
|
+
score: number;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function originalLines(text: string): string[] {
|
|
104
|
+
const lines = normalizeToLF(text).split("\n");
|
|
105
|
+
// Remove only the synthetic segment after a terminal newline. Do not remove
|
|
106
|
+
// a real, unterminated whitespace-only line.
|
|
107
|
+
if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
|
|
108
|
+
return lines;
|
|
109
|
+
}
|
|
110
|
+
|
|
74
111
|
/**
|
|
75
|
-
* Find the window
|
|
76
|
-
*
|
|
77
|
-
* The first pass ranks every window using cheap positional similarity. Only a
|
|
78
|
-
* small bounded set of candidates receives the more expensive sequence score.
|
|
79
|
-
* The sequence score is order-preserving and one-to-one, so repeated query
|
|
80
|
-
* lines cannot all claim the same file line.
|
|
112
|
+
* Find the best window and a distinct runner-up under explicit input/work
|
|
113
|
+
* budgets. Returning null is the safe fallback when a budget is exhausted.
|
|
81
114
|
*/
|
|
82
115
|
export function findClosestRegion(content: string, oldText: string): ClosestRegion | null {
|
|
83
|
-
|
|
84
|
-
const
|
|
85
|
-
|
|
116
|
+
if (Buffer.byteLength(oldText, "utf8") > MAX_QUERY_BYTES) return null;
|
|
117
|
+
const originalQ = originalLines(oldText);
|
|
118
|
+
const originalContent = originalLines(content);
|
|
119
|
+
if (originalQ.length === 0 || originalContent.length === 0 || originalContent.length > MAX_SEARCH_FILE_LINES) return null;
|
|
120
|
+
if ([...originalQ, ...originalContent].some((line) => line.length > MAX_LINE_CHARS)) return null;
|
|
86
121
|
|
|
122
|
+
const allQLines = originalQ.map(normalizeForFuzzyMatch);
|
|
123
|
+
const cLines = originalContent.map(normalizeForFuzzyMatch);
|
|
87
124
|
const truncated = allQLines.length > MAX_COMPARE_LINES;
|
|
88
125
|
const q = truncated ? allQLines.slice(0, MAX_COMPARE_LINES) : allQLines;
|
|
126
|
+
const qOriginal = truncated ? originalQ.slice(0, MAX_COMPARE_LINES) : originalQ;
|
|
89
127
|
const L = q.length;
|
|
90
128
|
const sizes = [...new Set([L - 1, L, L + 1].filter((size) => size >= 1 && size <= cLines.length))];
|
|
91
129
|
const preparedQ = q.map(prepareLine);
|
|
92
130
|
const preparedContent = cLines.map(prepareLine);
|
|
93
131
|
const candidates: CandidateWindow[] = [];
|
|
132
|
+
let discardedMaxPositionalScore = -1;
|
|
133
|
+
const budget = new WorkBudget(MAX_SIMILARITY_WORK);
|
|
94
134
|
|
|
95
|
-
|
|
96
|
-
for (
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
135
|
+
try {
|
|
136
|
+
for (const size of sizes) {
|
|
137
|
+
for (let start = 0; start + size <= cLines.length; start++) {
|
|
138
|
+
const pairs = Math.min(L, size);
|
|
139
|
+
let sum = 0;
|
|
140
|
+
for (let i = 0; i < pairs; i++) {
|
|
141
|
+
sum += preparedSimilarity(preparedQ[i], preparedContent[start + i], budget);
|
|
142
|
+
}
|
|
143
|
+
const discarded = keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
|
|
144
|
+
if (discarded !== undefined) discardedMaxPositionalScore = Math.max(discardedMaxPositionalScore, discarded);
|
|
101
145
|
}
|
|
102
|
-
keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
|
|
103
146
|
}
|
|
104
|
-
}
|
|
105
147
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
148
|
+
const scored: ScoredWindow[] = candidates.map((candidate) => ({
|
|
149
|
+
...candidate,
|
|
150
|
+
score: sequenceSimilarity(
|
|
151
|
+
preparedQ,
|
|
152
|
+
preparedContent.slice(candidate.start, candidate.start + candidate.size),
|
|
153
|
+
budget,
|
|
154
|
+
),
|
|
155
|
+
}));
|
|
156
|
+
scored.sort(
|
|
157
|
+
(a, b) => b.score - a.score || b.positionalScore - a.positionalScore || a.start - b.start || a.size - b.size,
|
|
158
|
+
);
|
|
159
|
+
const best = scored[0];
|
|
160
|
+
if (!best || best.score < MIN_SCORE) return null;
|
|
161
|
+
const bestEnd = best.start + best.size;
|
|
162
|
+
const runner = scored.find((candidate) => {
|
|
163
|
+
const candidateEnd = candidate.start + candidate.size;
|
|
164
|
+
return candidateEnd <= best.start || candidate.start >= bestEnd;
|
|
165
|
+
});
|
|
166
|
+
|
|
167
|
+
const windowLines = cLines.slice(best.start, bestEnd);
|
|
168
|
+
const windowOriginal = originalContent.slice(best.start, bestEnd);
|
|
169
|
+
const ops = alignLines(q, windowLines, qOriginal, windowOriginal, best.start, budget);
|
|
170
|
+
return {
|
|
171
|
+
startLine: best.start + 1,
|
|
172
|
+
endLine: bestEnd,
|
|
173
|
+
score: best.score,
|
|
174
|
+
ops,
|
|
175
|
+
equalCount: ops.filter((op) => op.type === "equal").length,
|
|
176
|
+
totalOldLines: L,
|
|
177
|
+
truncated,
|
|
178
|
+
competitor: runner
|
|
179
|
+
? { startLine: runner.start + 1, endLine: runner.start + runner.size, score: runner.score }
|
|
180
|
+
: undefined,
|
|
181
|
+
competitionIncomplete: discardedMaxPositionalScore >= best.positionalScore - MIN_DIRECT_POSITIONAL_GAP,
|
|
182
|
+
};
|
|
183
|
+
} catch (error) {
|
|
184
|
+
if (error instanceof Error && error.message === "similarity-work-budget-exhausted") return null;
|
|
185
|
+
throw error;
|
|
115
186
|
}
|
|
116
|
-
if (!best || bestScore < MIN_SCORE) return null;
|
|
117
|
-
|
|
118
|
-
const windowLines = cLines.slice(best.start, best.start + best.size);
|
|
119
|
-
const ops = alignLines(q, windowLines, best.start);
|
|
120
|
-
return {
|
|
121
|
-
startLine: best.start + 1,
|
|
122
|
-
endLine: best.start + best.size,
|
|
123
|
-
score: bestScore,
|
|
124
|
-
ops,
|
|
125
|
-
equalCount: ops.filter((op) => op.type === "equal").length,
|
|
126
|
-
totalOldLines: L,
|
|
127
|
-
truncated,
|
|
128
|
-
};
|
|
129
187
|
}
|
|
130
188
|
|
|
131
|
-
|
|
189
|
+
const MIN_DIRECT_POSITIONAL_GAP = 0.1;
|
|
190
|
+
|
|
191
|
+
/** Return the positional score of any candidate discarded by the bounded heap. */
|
|
192
|
+
function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): number | undefined {
|
|
132
193
|
if (candidates.length < MAX_CANDIDATE_WINDOWS) {
|
|
133
194
|
candidates.push(candidate);
|
|
134
|
-
candidates.sort((a, b) => a.positionalScore - b.positionalScore);
|
|
135
|
-
return;
|
|
195
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
196
|
+
return undefined;
|
|
136
197
|
}
|
|
137
|
-
|
|
198
|
+
const worst = candidates[0];
|
|
199
|
+
if (
|
|
200
|
+
candidate.positionalScore < worst.positionalScore ||
|
|
201
|
+
(candidate.positionalScore === worst.positionalScore && candidate.start >= worst.start)
|
|
202
|
+
) return candidate.positionalScore;
|
|
138
203
|
candidates[0] = candidate;
|
|
139
|
-
candidates.sort((a, b) => a.positionalScore - b.positionalScore);
|
|
204
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
205
|
+
return worst.positionalScore;
|
|
140
206
|
}
|
|
141
207
|
|
|
142
208
|
/** Weighted LCS: lines are matched at most once and in source order. */
|
|
143
|
-
function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): number {
|
|
209
|
+
function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[], budget: WorkBudget): number {
|
|
144
210
|
let previous = new Float64Array(window.length + 1);
|
|
145
211
|
for (let i = 1; i <= query.length; i++) {
|
|
146
212
|
const current = new Float64Array(window.length + 1);
|
|
147
213
|
for (let j = 1; j <= window.length; j++) {
|
|
148
|
-
const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1]);
|
|
214
|
+
const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1], budget);
|
|
149
215
|
current[j] = Math.max(previous[j], current[j - 1], matched);
|
|
150
216
|
}
|
|
151
217
|
previous = current;
|
|
@@ -153,18 +219,21 @@ function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): numb
|
|
|
153
219
|
return previous[window.length] / Math.max(query.length, window.length);
|
|
154
220
|
}
|
|
155
221
|
|
|
156
|
-
/**
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
222
|
+
/** Align fuzzy lines while attaching original strings to every operation. */
|
|
223
|
+
function alignLines(
|
|
224
|
+
q: string[],
|
|
225
|
+
w: string[],
|
|
226
|
+
qOriginal: string[],
|
|
227
|
+
wOriginal: string[],
|
|
228
|
+
wOffset: number,
|
|
229
|
+
budget: WorkBudget,
|
|
230
|
+
): AlignOp[] {
|
|
162
231
|
const n = q.length;
|
|
163
232
|
const m = w.length;
|
|
164
233
|
const preparedQ = q.map(prepareLine);
|
|
165
234
|
const preparedW = w.map(prepareLine);
|
|
166
235
|
const similarities = Array.from({ length: n }, (_, i) =>
|
|
167
|
-
Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j])),
|
|
236
|
+
Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j], budget)),
|
|
168
237
|
);
|
|
169
238
|
const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
|
|
170
239
|
for (let i = n - 1; i >= 0; i--) {
|
|
@@ -180,17 +249,25 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
|
|
|
180
249
|
let j = 0;
|
|
181
250
|
while (i < n && j < m) {
|
|
182
251
|
if (similarities[i][j] >= MATCH_THRESHOLD) {
|
|
183
|
-
|
|
252
|
+
const exact = qOriginal[i] === wOriginal[j];
|
|
253
|
+
raw.push({
|
|
254
|
+
type: exact ? "equal" : "similar",
|
|
255
|
+
fileLine: wOffset + j + 1,
|
|
256
|
+
fileText: wOriginal[j],
|
|
257
|
+
oldLine: i + 1,
|
|
258
|
+
oldText: qOriginal[i],
|
|
259
|
+
similarity: similarities[i][j],
|
|
260
|
+
});
|
|
184
261
|
i++;
|
|
185
262
|
j++;
|
|
186
263
|
} else if (dp[i + 1][j] > dp[i][j + 1]) {
|
|
187
|
-
raw.push({ type: "old-only", oldLine: i + 1, oldText:
|
|
264
|
+
raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
188
265
|
} else {
|
|
189
|
-
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText:
|
|
266
|
+
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
190
267
|
}
|
|
191
268
|
}
|
|
192
|
-
while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText:
|
|
193
|
-
while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText:
|
|
269
|
+
while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
270
|
+
while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
194
271
|
|
|
195
272
|
const merged: AlignOp[] = [];
|
|
196
273
|
for (let k = 0; k < raw.length; k++) {
|
package/src/text.ts
CHANGED
|
@@ -71,6 +71,17 @@ export function splitLinesWithEndings(content: string): string[] {
|
|
|
71
71
|
return content.match(/[^\n]*\n|[^\n]+/g) ?? [];
|
|
72
72
|
}
|
|
73
73
|
|
|
74
|
+
/**
|
|
75
|
+
* Split every logical line while retaining endings. Unlike
|
|
76
|
+
* splitLinesWithEndings(), this includes the zero-length logical line after a
|
|
77
|
+
* terminal newline. That distinction is needed when fuzzy normalization turns
|
|
78
|
+
* a whitespace-only final line into an empty line.
|
|
79
|
+
*/
|
|
80
|
+
export function splitLogicalLinesWithEndings(content: string): string[] {
|
|
81
|
+
const parts = content.split("\n");
|
|
82
|
+
return parts.map((part, index) => (index < parts.length - 1 ? `${part}\n` : part));
|
|
83
|
+
}
|
|
84
|
+
|
|
74
85
|
export function getLineSpans(content: string): LineSpan[] {
|
|
75
86
|
let offset = 0;
|
|
76
87
|
return splitLinesWithEndings(content).map((line) => {
|
|
@@ -80,6 +91,16 @@ export function getLineSpans(content: string): LineSpan[] {
|
|
|
80
91
|
});
|
|
81
92
|
}
|
|
82
93
|
|
|
94
|
+
/** Logical spans, including a possible zero-length final line. */
|
|
95
|
+
export function getLogicalLineSpans(content: string): LineSpan[] {
|
|
96
|
+
let offset = 0;
|
|
97
|
+
return splitLogicalLinesWithEndings(content).map((line) => {
|
|
98
|
+
const span = { start: offset, end: offset + line.length };
|
|
99
|
+
offset = span.end;
|
|
100
|
+
return span;
|
|
101
|
+
});
|
|
102
|
+
}
|
|
103
|
+
|
|
83
104
|
/** 0-based index of the line containing `offset` (clamped to the last line). */
|
|
84
105
|
export function lineAt(spans: LineSpan[], offset: number): number {
|
|
85
106
|
let lo = 0;
|
|
@@ -102,17 +123,25 @@ export function lineAt(spans: LineSpan[], offset: number): number {
|
|
|
102
123
|
* same normalization the matching engine uses for its uniqueness check.
|
|
103
124
|
* `fuzzyContent` must already be `normalizeForFuzzyMatch`-ed.
|
|
104
125
|
*/
|
|
105
|
-
export function countFuzzyOccurrences(fuzzyContent: string, needle: string): number {
|
|
106
|
-
|
|
107
|
-
|
|
126
|
+
export function countFuzzyOccurrences(fuzzyContent: string, needle: string, stopAfter = Number.POSITIVE_INFINITY): number {
|
|
127
|
+
const normalizedNeedle = normalizeForFuzzyMatch(needle);
|
|
128
|
+
if (!normalizedNeedle) return 0;
|
|
129
|
+
let count = 0;
|
|
130
|
+
let index = fuzzyContent.indexOf(normalizedNeedle);
|
|
131
|
+
while (index !== -1) {
|
|
132
|
+
count++;
|
|
133
|
+
if (count >= stopAfter) return count;
|
|
134
|
+
index = fuzzyContent.indexOf(normalizedNeedle, index + normalizedNeedle.length);
|
|
135
|
+
}
|
|
136
|
+
return count;
|
|
108
137
|
}
|
|
109
138
|
|
|
110
|
-
/** All start offsets of `needle
|
|
111
|
-
export function findAllOccurrences(haystack: string, needle: string): number[] {
|
|
139
|
+
/** All non-overlapping start offsets of `needle`, optionally bounded. */
|
|
140
|
+
export function findAllOccurrences(haystack: string, needle: string, limit = Number.POSITIVE_INFINITY): number[] {
|
|
112
141
|
const offsets: number[] = [];
|
|
113
|
-
if (!needle) return offsets;
|
|
142
|
+
if (!needle || limit <= 0) return offsets;
|
|
114
143
|
let idx = haystack.indexOf(needle);
|
|
115
|
-
while (idx !== -1) {
|
|
144
|
+
while (idx !== -1 && offsets.length < limit) {
|
|
116
145
|
offsets.push(idx);
|
|
117
146
|
idx = haystack.indexOf(needle, idx + needle.length);
|
|
118
147
|
}
|