@khanhicetea/pi-better-tool 0.2.1 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +116 -69
- package/package.json +34 -5
- package/src/apply.ts +22 -12
- package/src/diagnostics.ts +159 -26
- package/src/index.ts +15 -5
- package/src/paths.ts +19 -0
- package/src/read-evidence.ts +74 -33
- package/src/read-symbol.ts +214 -0
- package/src/similarity.ts +140 -63
- package/src/source-file.ts +60 -0
- package/src/symbols.ts +228 -0
- package/src/text.ts +36 -7
- package/src/tool.ts +178 -106
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { Type, type Static } from "typebox";
|
|
4
|
+
import { boundCompleteOutput, fenceFor } from "./diagnostics.ts";
|
|
5
|
+
import { resolveToolPath } from "./paths.ts";
|
|
6
|
+
import { readSource, sourceReadError } from "./source-file.ts";
|
|
7
|
+
import { indexSymbols, languageForPath, SUPPORTED_LANGUAGES, type SourceSymbol } from "./symbols.ts";
|
|
8
|
+
import { getLineSpans } from "./text.ts";
|
|
9
|
+
|
|
10
|
+
export const readSymbolSchema = Type.Object({
|
|
11
|
+
path: Type.String({ minLength: 1, maxLength: 4096, description: "Local source file path (relative, absolute, @path, ~/path, or file URL)." }),
|
|
12
|
+
line: Type.Optional(Type.Integer({ minimum: 1, description: "Read the innermost symbol containing this 1-based line. Combine with symbol to disambiguate duplicate names." })),
|
|
13
|
+
column: Type.Optional(Type.Integer({ minimum: 1, description: "Optional 1-based UTF-16 column with line, to distinguish symbols on the same line." })),
|
|
14
|
+
symbol: Type.Optional(Type.String({ minLength: 1, maxLength: 256, description: "Exact symbol name or qualified name, for example Server.run. Omit both symbol and line to list a symbol outline." })),
|
|
15
|
+
parent: Type.Optional(Type.Integer({ minimum: 0, maximum: 20, description: "Move outward this many enclosing symbols after selection (default 0)." })),
|
|
16
|
+
context: Type.Optional(Type.Integer({ minimum: 0, maximum: 20, description: "Extra whole lines before and after the selected symbol (default 0)." })),
|
|
17
|
+
offset: Type.Optional(Type.Integer({ minimum: 1, description: "1-based position within the selected symbol/context or outline, not a file line. Use the exact continuation call from a partial result." })),
|
|
18
|
+
limit: Type.Optional(Type.Integer({ minimum: 1, maximum: 1800, description: "Maximum source lines (default 1000) or outline entries (default 50). Output also has a 48 KiB hard limit." })),
|
|
19
|
+
});
|
|
20
|
+
export type ReadSymbolInput = Static<typeof readSymbolSchema>;
|
|
21
|
+
|
|
22
|
+
export interface VisibleSource {
|
|
23
|
+
startLine: number;
|
|
24
|
+
endLine: number;
|
|
25
|
+
startOffset: number;
|
|
26
|
+
endOffset: number;
|
|
27
|
+
}
|
|
28
|
+
export interface ReadSymbolResult {
|
|
29
|
+
content: Array<{ type: "text"; text: string }>;
|
|
30
|
+
details: {
|
|
31
|
+
mode: "symbol" | "outline";
|
|
32
|
+
snapshot: string;
|
|
33
|
+
symbol?: SourceSymbol;
|
|
34
|
+
visible?: VisibleSource;
|
|
35
|
+
complete: boolean;
|
|
36
|
+
nextCall?: ReadSymbolInput;
|
|
37
|
+
};
|
|
38
|
+
}
|
|
39
|
+
const MAX_OUTPUT_BYTES = 48 * 1024;
|
|
40
|
+
const MAX_OUTPUT_LINES = 1950;
|
|
41
|
+
const label = (text: string) => JSON.stringify(text.length > 300 ? `${text.slice(0, 297)}…` : text);
|
|
42
|
+
const callText = (input: ReadSymbolInput) => `read_symbol ${JSON.stringify(input)}`;
|
|
43
|
+
|
|
44
|
+
export function validateReadSymbolInput(input: ReadSymbolInput): void {
|
|
45
|
+
if (!input || typeof input.path !== "string" || !input.path.length || input.path.length > 4096) throw new Error("read_symbol requires a non-empty path of at most 4096 characters.");
|
|
46
|
+
if (input.symbol !== undefined && (typeof input.symbol !== "string" || !input.symbol.trim() || input.symbol.length > 256)) throw new Error("read_symbol symbol must be a non-empty name of at most 256 characters.");
|
|
47
|
+
for (const [key, min, max] of [["line", 1, Number.MAX_SAFE_INTEGER], ["column", 1, Number.MAX_SAFE_INTEGER], ["parent", 0, 20], ["context", 0, 20], ["offset", 1, Number.MAX_SAFE_INTEGER], ["limit", 1, 1800]] as const) {
|
|
48
|
+
const value = input[key];
|
|
49
|
+
if (value !== undefined && (!Number.isSafeInteger(value) || value < min || value > max)) throw new Error(`read_symbol ${key} must be an integer from ${min} to ${max}.`);
|
|
50
|
+
}
|
|
51
|
+
if (input.column !== undefined && input.line === undefined) throw new Error("read_symbol column requires line.");
|
|
52
|
+
if ((input.parent || input.context) && input.line === undefined && input.symbol === undefined) throw new Error("read_symbol parent/context requires a line or symbol selector; omit them to get an outline.");
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function candidateLine(path: string, symbol: SourceSymbol): string {
|
|
56
|
+
return `- ${label(symbol.qualifiedName)} (${symbol.kind}), lines ${symbol.startLine}-${symbol.endLine}. ${callText({ path, line: symbol.startLine, column: symbol.startColumn, symbol: symbol.qualifiedName.length <= 256 ? symbol.qualifiedName : undefined })}`;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Non-copyable preview for failed selection; never presented as a complete symbol. */
|
|
60
|
+
function fallbackContext(input: ReadSymbolInput, content: string, reason: string, atLine = input.line ?? 1): Error {
|
|
61
|
+
const lines = content.split("\n");
|
|
62
|
+
const center = Math.min(Math.max(atLine, 1), lines.length);
|
|
63
|
+
const start = Math.max(1, center - 5);
|
|
64
|
+
const end = Math.min(lines.length, center + 10);
|
|
65
|
+
const preview = lines.slice(start - 1, end).map((line, index) => `${start + index}: ${line.length > 240 ? `${line.slice(0, 240)}… [line clipped]` : line}`).join("\n");
|
|
66
|
+
return new Error(`${reason}\nFile has ${lines.length} lines. Preview only, not a complete symbol or retryable edit snippet:\n${preview}\nNext: read ${JSON.stringify({ path: input.path, offset: start, limit: end - start + 1 })}`);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function inside(symbol: SourceSymbol, line: number, column?: number): boolean {
|
|
70
|
+
if (line < symbol.startLine || line > symbol.endLine) return false;
|
|
71
|
+
if (column !== undefined && ((line === symbol.startLine && column < symbol.startColumn) || (line === symbol.endLine && column >= symbol.endColumn))) return false;
|
|
72
|
+
return true;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function selectSymbol(input: ReadSymbolInput, symbols: SourceSymbol[]): SourceSymbol {
|
|
76
|
+
let matches = symbols.filter((symbol) =>
|
|
77
|
+
(input.symbol === undefined || symbol.name === input.symbol || symbol.qualifiedName === input.symbol) &&
|
|
78
|
+
(input.line === undefined || inside(symbol, input.line, input.column)),
|
|
79
|
+
);
|
|
80
|
+
if (input.symbol === undefined) {
|
|
81
|
+
// Keep innermost declarations, but never choose arbitrarily between siblings.
|
|
82
|
+
const matched = new Set(matches);
|
|
83
|
+
const ancestors = new Set<SourceSymbol>();
|
|
84
|
+
// Parents precede children in the index. Propagate once in reverse,
|
|
85
|
+
// rather than walking every ancestor chain in deeply nested source.
|
|
86
|
+
for (let index = symbols.length - 1; index >= 0; index--) {
|
|
87
|
+
const symbol = symbols[index];
|
|
88
|
+
if (symbol.parent !== undefined && (matched.has(symbol) || ancestors.has(symbol))) ancestors.add(symbols[symbol.parent]);
|
|
89
|
+
}
|
|
90
|
+
matches = matches.filter((symbol) => !ancestors.has(symbol));
|
|
91
|
+
}
|
|
92
|
+
if (matches.length !== 1) {
|
|
93
|
+
const candidates = matches.length ? matches : [...symbols].sort((a, b) => {
|
|
94
|
+
if (input.line !== undefined) return Math.abs(a.startLine - input.line) - Math.abs(b.startLine - input.line);
|
|
95
|
+
const query = input.symbol?.toLowerCase() ?? "";
|
|
96
|
+
return Number(b.qualifiedName.toLowerCase().includes(query)) - Number(a.qualifiedName.toLowerCase().includes(query));
|
|
97
|
+
});
|
|
98
|
+
throw new Error([
|
|
99
|
+
matches.length ? `Ambiguous symbol selection: ${matches.length} candidates. No symbol was selected.` : "No matching symbol. Names are exact and case-sensitive; no nearby symbol was selected automatically.",
|
|
100
|
+
...candidates.slice(0, 8).map((symbol) => candidateLine(input.path, symbol)),
|
|
101
|
+
...(candidates.length > 8 ? [`${candidates.length - 8} more candidates omitted.`] : []),
|
|
102
|
+
`Next: choose a candidate call above, or list the outline with ${callText({ path: input.path })}.`,
|
|
103
|
+
].join("\n"));
|
|
104
|
+
}
|
|
105
|
+
let selected = matches[0];
|
|
106
|
+
for (let depth = 0; depth < (input.parent ?? 0); depth++) {
|
|
107
|
+
if (selected.parent === undefined) throw new Error(`No enclosing symbol at parent=${input.parent}. Outermost available: ${candidateLine(input.path, selected)}\nRetry with a smaller parent value.`);
|
|
108
|
+
selected = symbols[selected.parent];
|
|
109
|
+
}
|
|
110
|
+
return selected;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function fits(text: string): boolean {
|
|
114
|
+
return Buffer.byteLength(text, "utf8") <= MAX_OUTPUT_BYTES && text.split("\n").length <= MAX_OUTPUT_LINES;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Pure snapshot-to-result path, also used to verify stored read evidence. */
|
|
118
|
+
export async function buildSymbolRead(input: ReadSymbolInput, content: string, signal?: AbortSignal): Promise<ReadSymbolResult> {
|
|
119
|
+
validateReadSymbolInput(input);
|
|
120
|
+
const lines = content.split("\n");
|
|
121
|
+
if (input.line !== undefined && input.line > lines.length) throw fallbackContext(input, content, `line=${input.line} is beyond EOF. Valid lines: 1-${lines.length}.`);
|
|
122
|
+
if (input.column !== undefined && input.column > lines[input.line! - 1].length + 1) throw fallbackContext(input, content, `column=${input.column} is outside line ${input.line}.`);
|
|
123
|
+
const parserPath = resolveToolPath(input.path, "/");
|
|
124
|
+
if (!languageForPath(parserPath)) throw fallbackContext(input, content, `Unsupported source type. read_symbol supports ${SUPPORTED_LANGUAGES}. Use read for this file.`);
|
|
125
|
+
let index: Awaited<ReturnType<typeof indexSymbols>>;
|
|
126
|
+
try { index = await indexSymbols(parserPath, content, signal); }
|
|
127
|
+
catch (error) {
|
|
128
|
+
signal?.throwIfAborted();
|
|
129
|
+
throw fallbackContext(input, content, `Symbol parser unavailable or analysis limit reached: ${error instanceof Error ? error.message.slice(0, 500) : "unknown parser error"}. Use read instead.`);
|
|
130
|
+
}
|
|
131
|
+
signal?.throwIfAborted();
|
|
132
|
+
if (index.errorLine !== undefined) throw fallbackContext(input, content, `Syntax error or incomplete syntax near line ${index.errorLine}; complete symbol boundaries are not reliable.`, input.line ?? index.errorLine);
|
|
133
|
+
const snapshot = createHash("sha256").update(content).digest("hex");
|
|
134
|
+
const header = `Source ${label(input.path)} (${lines.length} lines). Snapshot sha256:${snapshot}`;
|
|
135
|
+
const offset = input.offset ?? 1;
|
|
136
|
+
if (input.line === undefined && input.symbol === undefined) {
|
|
137
|
+
const total = index.symbols.length;
|
|
138
|
+
if (offset > Math.max(1, total)) throw new Error(`Outline offset=${offset} is beyond ${total} entries. Next: ${callText({ ...input, offset: Math.max(1, total - 49) })}`);
|
|
139
|
+
const entries: string[] = [];
|
|
140
|
+
let cursor = offset - 1;
|
|
141
|
+
for (; cursor < Math.min(total, offset - 1 + (input.limit ?? 50)); cursor++) {
|
|
142
|
+
const entry = candidateLine(input.path, index.symbols[cursor]);
|
|
143
|
+
if (!fits(`${header}\n${entries.join("\n")}\n${entry}\n${" ".repeat(6000)}`)) break;
|
|
144
|
+
entries.push(entry);
|
|
145
|
+
}
|
|
146
|
+
const nextCall = cursor < total ? { ...input, offset: cursor + 1 } : undefined;
|
|
147
|
+
const text = `${header}\nSymbol outline: ${total} declarations, ${entries.length} shown.${total === 0 ? " No symbols found; use read for source text." : ""}\n${entries.join("\n")}${nextCall ? `\nMore entries. Next: ${callText(nextCall)}` : ""}`;
|
|
148
|
+
return { content: [{ type: "text", text }], details: { mode: "outline", snapshot, complete: !nextCall, nextCall } };
|
|
149
|
+
}
|
|
150
|
+
if (!index.symbols.length) throw fallbackContext(input, content, "No symbol declarations found in this file.");
|
|
151
|
+
const selected = selectSymbol(input, index.symbols);
|
|
152
|
+
const start = Math.max(1, selected.startLine - (input.context ?? 0));
|
|
153
|
+
const end = Math.min(lines.length, selected.endLine + (input.context ?? 0));
|
|
154
|
+
const count = end - start + 1;
|
|
155
|
+
if (offset > count) throw new Error(`Symbol ${label(selected.qualifiedName)} spans lines ${selected.startLine}-${selected.endLine}; selected region has ${count} lines. offset=${offset} is outside it. Next: ${callText({ ...input, offset: 1 })}`);
|
|
156
|
+
const firstLine = start + offset - 1;
|
|
157
|
+
let lastLine = Math.min(end, firstLine + (input.limit ?? 1000) - 1);
|
|
158
|
+
const parents: string[] = [];
|
|
159
|
+
let parent = selected.parent;
|
|
160
|
+
while (parent !== undefined && parents.length < 20) { parents.unshift(label(index.symbols[parent].qualifiedName)); parent = index.symbols[parent].parent; }
|
|
161
|
+
const prefix = `${header}\nSymbol ${label(selected.qualifiedName)} (${selected.kind}), lines ${selected.startLine}-${selected.endLine}.${parents.length ? `\nEnclosing: ${parents.join(" > ")}` : ""}`;
|
|
162
|
+
let text = "";
|
|
163
|
+
let snippet = "";
|
|
164
|
+
let nextCall: ReadSymbolInput | undefined;
|
|
165
|
+
// Whole-line pages only. Account for fences, metadata, and continuation JSON.
|
|
166
|
+
while (lastLine >= firstLine) {
|
|
167
|
+
snippet = lines.slice(firstLine - 1, lastLine).join("\n");
|
|
168
|
+
const fence = fenceFor(snippet);
|
|
169
|
+
nextCall = lastLine < end ? { ...input, offset: lastLine - start + 2 } : undefined;
|
|
170
|
+
text = `${prefix}\nShowing file lines ${firstLine}-${lastLine} of selected lines ${start}-${end}. ${firstLine === start && lastLine === end ? "Complete selection." : "Partial selection; do not treat this page as the whole symbol."}\n\n${fence}\n${snippet}\n${fence}${nextCall ? `\n\nNext: ${callText(nextCall)}` : ""}`;
|
|
171
|
+
if (fits(text)) break;
|
|
172
|
+
// Remove a proportional chunk first; small pages shrink one line at a time.
|
|
173
|
+
lastLine -= Math.max(1, Math.floor((lastLine - firstLine + 1) / 4));
|
|
174
|
+
}
|
|
175
|
+
if (lastLine < firstLine) throw fallbackContext(input, content, `File line ${firstLine} cannot fit as a complete line in the 48 KiB output budget. No partial edit snippet was returned.`, firstLine);
|
|
176
|
+
const spans = getLineSpans(content);
|
|
177
|
+
const startOffset = spans[firstLine - 1]?.start ?? content.length;
|
|
178
|
+
return {
|
|
179
|
+
content: [{ type: "text", text }],
|
|
180
|
+
details: {
|
|
181
|
+
mode: "symbol", snapshot, symbol: selected,
|
|
182
|
+
visible: { startLine: firstLine, endLine: lastLine, startOffset, endOffset: startOffset + snippet.length },
|
|
183
|
+
complete: firstLine === start && lastLine === end, nextCall,
|
|
184
|
+
},
|
|
185
|
+
};
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
export async function executeReadSymbol(input: ReadSymbolInput, signal: AbortSignal | undefined, ctx: Pick<ExtensionContext, "cwd">): Promise<ReadSymbolResult> {
|
|
189
|
+
validateReadSymbolInput(input);
|
|
190
|
+
const path = resolveToolPath(input.path, ctx.cwd);
|
|
191
|
+
let content: string;
|
|
192
|
+
try { content = await readSource(path, signal); }
|
|
193
|
+
catch (error) { signal?.throwIfAborted(); throw new Error(boundCompleteOutput(await sourceReadError(path, error))); }
|
|
194
|
+
try { return await buildSymbolRead(input, content, signal); }
|
|
195
|
+
catch (error) {
|
|
196
|
+
signal?.throwIfAborted();
|
|
197
|
+
throw new Error(boundCompleteOutput(error instanceof Error ? error.message : String(error)));
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
export function registerReadSymbolTool(pi: ExtensionAPI): void {
|
|
202
|
+
pi.registerTool({
|
|
203
|
+
name: "read_symbol", label: "read_symbol",
|
|
204
|
+
description: `Read a complete function, method, class, or type from a local source file by containing line or exact symbol name. Supports ${SUPPORTED_LANGUAGES}. With path only, lists a symbol outline. Includes enclosing names, exact source ranges, and actionable failure context. Output is bounded to 48 KiB/1950 lines; partial results include an exact continuation call. Does not replace read for ordinary text or images.`,
|
|
205
|
+
promptSnippet: "Read whole source symbols by line/name, or list a file's symbol outline",
|
|
206
|
+
promptGuidelines: [
|
|
207
|
+
"Use read_symbol with path and line after grep/edit identifies a code location, instead of guessing successive read offsets to find the function boundary.",
|
|
208
|
+
"Use read_symbol with symbol for an exact name or qualified name; use path alone for an outline. Use parent to include an enclosing function or class.",
|
|
209
|
+
"For read_symbol partial output, use the supplied continuation arguments. offset is relative to the selection, not a file line. Only displayed source is evidence for edit.",
|
|
210
|
+
],
|
|
211
|
+
parameters: readSymbolSchema,
|
|
212
|
+
async execute(_id, input, signal, _onUpdate, ctx) { return executeReadSymbol(input, signal, ctx); },
|
|
213
|
+
});
|
|
214
|
+
}
|
package/src/similarity.ts
CHANGED
|
@@ -3,16 +3,20 @@
|
|
|
3
3
|
* failed oldText almost matched.
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
|
-
import {
|
|
6
|
+
import { normalizeForFuzzyMatch, normalizeToLF } from "./text.ts";
|
|
7
7
|
|
|
8
8
|
export interface AlignOp {
|
|
9
|
-
type: "equal" | "changed" | "file-only" | "old-only";
|
|
10
|
-
/** 1-based line number in the file (for equal/changed/file-only). */
|
|
9
|
+
type: "equal" | "similar" | "changed" | "file-only" | "old-only";
|
|
10
|
+
/** 1-based line number in the file (for equal/similar/changed/file-only). */
|
|
11
11
|
fileLine?: number;
|
|
12
|
+
/** Original LF-normalized file text, never fuzzy-normalized. */
|
|
12
13
|
fileText?: string;
|
|
13
14
|
/** 1-based line number in the model-provided oldText. */
|
|
14
15
|
oldLine?: number;
|
|
16
|
+
/** Original LF-normalized oldText, never fuzzy-normalized. */
|
|
15
17
|
oldText?: string;
|
|
18
|
+
/** Heuristic similarity for aligned pairs. */
|
|
19
|
+
similarity?: number;
|
|
16
20
|
}
|
|
17
21
|
|
|
18
22
|
export interface ClosestRegion {
|
|
@@ -21,10 +25,15 @@ export interface ClosestRegion {
|
|
|
21
25
|
/** 0..1 order-preserving, one-to-one line similarity. */
|
|
22
26
|
score: number;
|
|
23
27
|
ops: AlignOp[];
|
|
28
|
+
/** Number of exactly equal original lines. */
|
|
24
29
|
equalCount: number;
|
|
25
30
|
totalOldLines: number;
|
|
26
31
|
/** oldText was truncated for comparison (very large oldText). */
|
|
27
32
|
truncated: boolean;
|
|
33
|
+
/** Best distinct, non-overlapping competitor, when retained by the bounded search. */
|
|
34
|
+
competitor?: { startLine: number; endLine: number; score: number };
|
|
35
|
+
/** A discarded window was too competitive to prove a safe score margin. */
|
|
36
|
+
competitionIncomplete: boolean;
|
|
28
37
|
}
|
|
29
38
|
|
|
30
39
|
interface PreparedLine {
|
|
@@ -42,15 +51,28 @@ function prepareLine(text: string): PreparedLine {
|
|
|
42
51
|
return { text, grams, gramCount: Math.max(0, text.length - 1) };
|
|
43
52
|
}
|
|
44
53
|
|
|
54
|
+
class WorkBudget {
|
|
55
|
+
private remaining: number;
|
|
56
|
+
constructor(remaining: number) {
|
|
57
|
+
this.remaining = remaining;
|
|
58
|
+
}
|
|
59
|
+
spend(amount: number): boolean {
|
|
60
|
+
this.remaining -= amount;
|
|
61
|
+
return this.remaining >= 0;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
45
65
|
/** Dice coefficient over character bigrams; 1.0 for identical strings. */
|
|
46
66
|
export function lineSimilarity(a: string, b: string): number {
|
|
47
67
|
return preparedSimilarity(prepareLine(a), prepareLine(b));
|
|
48
68
|
}
|
|
49
69
|
|
|
50
|
-
function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
|
|
70
|
+
function preparedSimilarity(a: PreparedLine, b: PreparedLine, budget?: WorkBudget): number {
|
|
71
|
+
if (budget && !budget.spend(1)) throw new Error("similarity-work-budget-exhausted");
|
|
51
72
|
if (a.text === b.text) return 1;
|
|
52
73
|
if (a.gramCount === 0 || b.gramCount === 0) return 0;
|
|
53
74
|
const [small, large] = a.grams.size <= b.grams.size ? [a.grams, b.grams] : [b.grams, a.grams];
|
|
75
|
+
if (budget && !budget.spend(small.size)) throw new Error("similarity-work-budget-exhausted");
|
|
54
76
|
let overlap = 0;
|
|
55
77
|
for (const [gram, count] of small) {
|
|
56
78
|
const other = large.get(gram);
|
|
@@ -61,7 +83,10 @@ function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
|
|
|
61
83
|
|
|
62
84
|
const MAX_COMPARE_LINES = 300;
|
|
63
85
|
const MAX_SEARCH_FILE_LINES = 10_000;
|
|
64
|
-
const MAX_CANDIDATE_WINDOWS =
|
|
86
|
+
const MAX_CANDIDATE_WINDOWS = 48;
|
|
87
|
+
const MAX_QUERY_BYTES = 128 * 1024;
|
|
88
|
+
const MAX_LINE_CHARS = 16 * 1024;
|
|
89
|
+
const MAX_SIMILARITY_WORK = 20_000_000;
|
|
65
90
|
const MIN_SCORE = 0.3;
|
|
66
91
|
const MATCH_THRESHOLD = 0.75;
|
|
67
92
|
|
|
@@ -71,81 +96,122 @@ interface CandidateWindow {
|
|
|
71
96
|
positionalScore: number;
|
|
72
97
|
}
|
|
73
98
|
|
|
99
|
+
interface ScoredWindow extends CandidateWindow {
|
|
100
|
+
score: number;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function originalLines(text: string): string[] {
|
|
104
|
+
const lines = normalizeToLF(text).split("\n");
|
|
105
|
+
// Remove only the synthetic segment after a terminal newline. Do not remove
|
|
106
|
+
// a real, unterminated whitespace-only line.
|
|
107
|
+
if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
|
|
108
|
+
return lines;
|
|
109
|
+
}
|
|
110
|
+
|
|
74
111
|
/**
|
|
75
|
-
* Find the window
|
|
76
|
-
*
|
|
77
|
-
* The first pass ranks every window using cheap positional similarity. Only a
|
|
78
|
-
* small bounded set of candidates receives the more expensive sequence score.
|
|
79
|
-
* The sequence score is order-preserving and one-to-one, so repeated query
|
|
80
|
-
* lines cannot all claim the same file line.
|
|
112
|
+
* Find the best window and a distinct runner-up under explicit input/work
|
|
113
|
+
* budgets. Returning null is the safe fallback when a budget is exhausted.
|
|
81
114
|
*/
|
|
82
115
|
export function findClosestRegion(content: string, oldText: string): ClosestRegion | null {
|
|
83
|
-
|
|
84
|
-
const
|
|
85
|
-
|
|
116
|
+
if (Buffer.byteLength(oldText, "utf8") > MAX_QUERY_BYTES) return null;
|
|
117
|
+
const originalQ = originalLines(oldText);
|
|
118
|
+
const originalContent = originalLines(content);
|
|
119
|
+
if (originalQ.length === 0 || originalContent.length === 0 || originalContent.length > MAX_SEARCH_FILE_LINES) return null;
|
|
120
|
+
if ([...originalQ, ...originalContent].some((line) => line.length > MAX_LINE_CHARS)) return null;
|
|
86
121
|
|
|
122
|
+
const allQLines = originalQ.map(normalizeForFuzzyMatch);
|
|
123
|
+
const cLines = originalContent.map(normalizeForFuzzyMatch);
|
|
87
124
|
const truncated = allQLines.length > MAX_COMPARE_LINES;
|
|
88
125
|
const q = truncated ? allQLines.slice(0, MAX_COMPARE_LINES) : allQLines;
|
|
126
|
+
const qOriginal = truncated ? originalQ.slice(0, MAX_COMPARE_LINES) : originalQ;
|
|
89
127
|
const L = q.length;
|
|
90
128
|
const sizes = [...new Set([L - 1, L, L + 1].filter((size) => size >= 1 && size <= cLines.length))];
|
|
91
129
|
const preparedQ = q.map(prepareLine);
|
|
92
130
|
const preparedContent = cLines.map(prepareLine);
|
|
93
131
|
const candidates: CandidateWindow[] = [];
|
|
132
|
+
let discardedMaxPositionalScore = -1;
|
|
133
|
+
const budget = new WorkBudget(MAX_SIMILARITY_WORK);
|
|
94
134
|
|
|
95
|
-
|
|
96
|
-
for (
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
135
|
+
try {
|
|
136
|
+
for (const size of sizes) {
|
|
137
|
+
for (let start = 0; start + size <= cLines.length; start++) {
|
|
138
|
+
const pairs = Math.min(L, size);
|
|
139
|
+
let sum = 0;
|
|
140
|
+
for (let i = 0; i < pairs; i++) {
|
|
141
|
+
sum += preparedSimilarity(preparedQ[i], preparedContent[start + i], budget);
|
|
142
|
+
}
|
|
143
|
+
const discarded = keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
|
|
144
|
+
if (discarded !== undefined) discardedMaxPositionalScore = Math.max(discardedMaxPositionalScore, discarded);
|
|
101
145
|
}
|
|
102
|
-
keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
|
|
103
146
|
}
|
|
104
|
-
}
|
|
105
147
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
148
|
+
const scored: ScoredWindow[] = candidates.map((candidate) => ({
|
|
149
|
+
...candidate,
|
|
150
|
+
score: sequenceSimilarity(
|
|
151
|
+
preparedQ,
|
|
152
|
+
preparedContent.slice(candidate.start, candidate.start + candidate.size),
|
|
153
|
+
budget,
|
|
154
|
+
),
|
|
155
|
+
}));
|
|
156
|
+
scored.sort(
|
|
157
|
+
(a, b) => b.score - a.score || b.positionalScore - a.positionalScore || a.start - b.start || a.size - b.size,
|
|
158
|
+
);
|
|
159
|
+
const best = scored[0];
|
|
160
|
+
if (!best || best.score < MIN_SCORE) return null;
|
|
161
|
+
const bestEnd = best.start + best.size;
|
|
162
|
+
const runner = scored.find((candidate) => {
|
|
163
|
+
const candidateEnd = candidate.start + candidate.size;
|
|
164
|
+
return candidateEnd <= best.start || candidate.start >= bestEnd;
|
|
165
|
+
});
|
|
166
|
+
|
|
167
|
+
const windowLines = cLines.slice(best.start, bestEnd);
|
|
168
|
+
const windowOriginal = originalContent.slice(best.start, bestEnd);
|
|
169
|
+
const ops = alignLines(q, windowLines, qOriginal, windowOriginal, best.start, budget);
|
|
170
|
+
return {
|
|
171
|
+
startLine: best.start + 1,
|
|
172
|
+
endLine: bestEnd,
|
|
173
|
+
score: best.score,
|
|
174
|
+
ops,
|
|
175
|
+
equalCount: ops.filter((op) => op.type === "equal").length,
|
|
176
|
+
totalOldLines: L,
|
|
177
|
+
truncated,
|
|
178
|
+
competitor: runner
|
|
179
|
+
? { startLine: runner.start + 1, endLine: runner.start + runner.size, score: runner.score }
|
|
180
|
+
: undefined,
|
|
181
|
+
competitionIncomplete: discardedMaxPositionalScore >= best.positionalScore - MIN_DIRECT_POSITIONAL_GAP,
|
|
182
|
+
};
|
|
183
|
+
} catch (error) {
|
|
184
|
+
if (error instanceof Error && error.message === "similarity-work-budget-exhausted") return null;
|
|
185
|
+
throw error;
|
|
115
186
|
}
|
|
116
|
-
if (!best || bestScore < MIN_SCORE) return null;
|
|
117
|
-
|
|
118
|
-
const windowLines = cLines.slice(best.start, best.start + best.size);
|
|
119
|
-
const ops = alignLines(q, windowLines, best.start);
|
|
120
|
-
return {
|
|
121
|
-
startLine: best.start + 1,
|
|
122
|
-
endLine: best.start + best.size,
|
|
123
|
-
score: bestScore,
|
|
124
|
-
ops,
|
|
125
|
-
equalCount: ops.filter((op) => op.type === "equal").length,
|
|
126
|
-
totalOldLines: L,
|
|
127
|
-
truncated,
|
|
128
|
-
};
|
|
129
187
|
}
|
|
130
188
|
|
|
131
|
-
|
|
189
|
+
const MIN_DIRECT_POSITIONAL_GAP = 0.1;
|
|
190
|
+
|
|
191
|
+
/** Return the positional score of any candidate discarded by the bounded heap. */
|
|
192
|
+
function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): number | undefined {
|
|
132
193
|
if (candidates.length < MAX_CANDIDATE_WINDOWS) {
|
|
133
194
|
candidates.push(candidate);
|
|
134
|
-
candidates.sort((a, b) => a.positionalScore - b.positionalScore);
|
|
135
|
-
return;
|
|
195
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
196
|
+
return undefined;
|
|
136
197
|
}
|
|
137
|
-
|
|
198
|
+
const worst = candidates[0];
|
|
199
|
+
if (
|
|
200
|
+
candidate.positionalScore < worst.positionalScore ||
|
|
201
|
+
(candidate.positionalScore === worst.positionalScore && candidate.start >= worst.start)
|
|
202
|
+
) return candidate.positionalScore;
|
|
138
203
|
candidates[0] = candidate;
|
|
139
|
-
candidates.sort((a, b) => a.positionalScore - b.positionalScore);
|
|
204
|
+
candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
|
|
205
|
+
return worst.positionalScore;
|
|
140
206
|
}
|
|
141
207
|
|
|
142
208
|
/** Weighted LCS: lines are matched at most once and in source order. */
|
|
143
|
-
function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): number {
|
|
209
|
+
function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[], budget: WorkBudget): number {
|
|
144
210
|
let previous = new Float64Array(window.length + 1);
|
|
145
211
|
for (let i = 1; i <= query.length; i++) {
|
|
146
212
|
const current = new Float64Array(window.length + 1);
|
|
147
213
|
for (let j = 1; j <= window.length; j++) {
|
|
148
|
-
const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1]);
|
|
214
|
+
const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1], budget);
|
|
149
215
|
current[j] = Math.max(previous[j], current[j - 1], matched);
|
|
150
216
|
}
|
|
151
217
|
previous = current;
|
|
@@ -153,18 +219,21 @@ function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): numb
|
|
|
153
219
|
return previous[window.length] / Math.max(query.length, window.length);
|
|
154
220
|
}
|
|
155
221
|
|
|
156
|
-
/**
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
222
|
+
/** Align fuzzy lines while attaching original strings to every operation. */
|
|
223
|
+
function alignLines(
|
|
224
|
+
q: string[],
|
|
225
|
+
w: string[],
|
|
226
|
+
qOriginal: string[],
|
|
227
|
+
wOriginal: string[],
|
|
228
|
+
wOffset: number,
|
|
229
|
+
budget: WorkBudget,
|
|
230
|
+
): AlignOp[] {
|
|
162
231
|
const n = q.length;
|
|
163
232
|
const m = w.length;
|
|
164
233
|
const preparedQ = q.map(prepareLine);
|
|
165
234
|
const preparedW = w.map(prepareLine);
|
|
166
235
|
const similarities = Array.from({ length: n }, (_, i) =>
|
|
167
|
-
Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j])),
|
|
236
|
+
Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j], budget)),
|
|
168
237
|
);
|
|
169
238
|
const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
|
|
170
239
|
for (let i = n - 1; i >= 0; i--) {
|
|
@@ -180,17 +249,25 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
|
|
|
180
249
|
let j = 0;
|
|
181
250
|
while (i < n && j < m) {
|
|
182
251
|
if (similarities[i][j] >= MATCH_THRESHOLD) {
|
|
183
|
-
|
|
252
|
+
const exact = qOriginal[i] === wOriginal[j];
|
|
253
|
+
raw.push({
|
|
254
|
+
type: exact ? "equal" : "similar",
|
|
255
|
+
fileLine: wOffset + j + 1,
|
|
256
|
+
fileText: wOriginal[j],
|
|
257
|
+
oldLine: i + 1,
|
|
258
|
+
oldText: qOriginal[i],
|
|
259
|
+
similarity: similarities[i][j],
|
|
260
|
+
});
|
|
184
261
|
i++;
|
|
185
262
|
j++;
|
|
186
263
|
} else if (dp[i + 1][j] > dp[i][j + 1]) {
|
|
187
|
-
raw.push({ type: "old-only", oldLine: i + 1, oldText:
|
|
264
|
+
raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
188
265
|
} else {
|
|
189
|
-
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText:
|
|
266
|
+
raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
190
267
|
}
|
|
191
268
|
}
|
|
192
|
-
while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText:
|
|
193
|
-
while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText:
|
|
269
|
+
while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
|
|
270
|
+
while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
|
|
194
271
|
|
|
195
272
|
const merged: AlignOp[] = [];
|
|
196
273
|
for (let k = 0; k < raw.length; k++) {
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { constants } from "node:fs";
|
|
2
|
+
import { open, opendir } from "node:fs/promises";
|
|
3
|
+
import { basename, dirname, join } from "node:path";
|
|
4
|
+
import { MAX_SOURCE_BYTES } from "./symbols.ts";
|
|
5
|
+
import { normalizeToLF, splitBom } from "./text.ts";
|
|
6
|
+
|
|
7
|
+
/** Bounded local source read; no silent decoding loss and no special files. */
|
|
8
|
+
export async function readSource(path: string, signal?: AbortSignal): Promise<string> {
|
|
9
|
+
signal?.throwIfAborted();
|
|
10
|
+
const handle = await open(path, constants.O_RDONLY | constants.O_NONBLOCK);
|
|
11
|
+
try {
|
|
12
|
+
const stat = await handle.stat();
|
|
13
|
+
if (!stat.isFile()) throw new Error("The path is not a regular source file.");
|
|
14
|
+
if (stat.size > MAX_SOURCE_BYTES) throw new Error("Source exceeds the 2 MiB symbol analysis limit; use read with offset/limit.");
|
|
15
|
+
const buffer = Buffer.alloc(Math.min(stat.size + 1, MAX_SOURCE_BYTES + 1));
|
|
16
|
+
let length = 0;
|
|
17
|
+
while (length < buffer.length) {
|
|
18
|
+
signal?.throwIfAborted();
|
|
19
|
+
const { bytesRead } = await handle.read(buffer, length, buffer.length - length, null);
|
|
20
|
+
if (!bytesRead) break;
|
|
21
|
+
length += bytesRead;
|
|
22
|
+
}
|
|
23
|
+
const after = await handle.stat();
|
|
24
|
+
if (length !== stat.size || stat.size !== after.size || stat.mtimeMs !== after.mtimeMs || stat.ctimeMs !== after.ctimeMs) throw new Error("The source changed during the read; retry read_symbol.");
|
|
25
|
+
const bytes = buffer.subarray(0, length);
|
|
26
|
+
let content: string;
|
|
27
|
+
try { content = new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }).decode(bytes); }
|
|
28
|
+
catch { throw new Error("The source is not valid UTF-8; convert its encoding before symbol parsing."); }
|
|
29
|
+
if (bytes.includes(0)) throw new Error("The source contains NUL bytes; binary/UTF-16 input is not supported.");
|
|
30
|
+
signal?.throwIfAborted();
|
|
31
|
+
return splitBom(normalizeToLF(content)).text;
|
|
32
|
+
} finally {
|
|
33
|
+
await handle.close();
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** Only inspect a bounded number of siblings; never perform a hidden repo scan. */
|
|
38
|
+
export async function sourceReadError(path: string, error: unknown): Promise<string> {
|
|
39
|
+
const code = (error as NodeJS.ErrnoException)?.code;
|
|
40
|
+
const message = error instanceof Error ? error.message.slice(0, 1000) : String(error).slice(0, 1000);
|
|
41
|
+
const lines = [`Could not read source ${JSON.stringify(path)}: ${message}`];
|
|
42
|
+
if (code === "ENOENT" || code === "ENOTDIR") {
|
|
43
|
+
const names: string[] = [];
|
|
44
|
+
try {
|
|
45
|
+
const directory = await opendir(dirname(path));
|
|
46
|
+
let inspected = 0;
|
|
47
|
+
for await (const entry of directory) {
|
|
48
|
+
if (entry.isFile()) names.push(entry.name);
|
|
49
|
+
if (++inspected >= 100) break;
|
|
50
|
+
}
|
|
51
|
+
} catch { /* Parent may also be missing or inaccessible. */ }
|
|
52
|
+
const stem = basename(path).split(".")[0].toLowerCase();
|
|
53
|
+
names.sort((a, b) => Number(b.toLowerCase().includes(stem)) - Number(a.toLowerCase().includes(stem)) || a.localeCompare(b));
|
|
54
|
+
if (names.length) lines.push("Nearby file candidates (bounded listing, not automatic path corrections):", ...names.slice(0, 8).map((name) => ` ${JSON.stringify(join(dirname(path), name))}`));
|
|
55
|
+
lines.push("Verify the path and retry read_symbol; use find/ls if the parent path is wrong.");
|
|
56
|
+
} else if (code === "EACCES" || code === "EPERM") {
|
|
57
|
+
lines.push("Check file permissions. Do not retry the same call until access changes.");
|
|
58
|
+
} else lines.push("Use read for ordinary text, images, or bounded line ranges; no symbol boundaries were returned.");
|
|
59
|
+
return lines.join("\n");
|
|
60
|
+
}
|