@khanhicetea/pi-better-tool 0.2.1 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,214 @@
1
+ import { createHash } from "node:crypto";
2
+ import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
3
+ import { Type, type Static } from "typebox";
4
+ import { boundCompleteOutput, fenceFor } from "./diagnostics.ts";
5
+ import { resolveToolPath } from "./paths.ts";
6
+ import { readSource, sourceReadError } from "./source-file.ts";
7
+ import { indexSymbols, languageForPath, SUPPORTED_LANGUAGES, type SourceSymbol } from "./symbols.ts";
8
+ import { getLineSpans } from "./text.ts";
9
+
10
+ export const readSymbolSchema = Type.Object({
11
+ path: Type.String({ minLength: 1, maxLength: 4096, description: "Local source file path (relative, absolute, @path, ~/path, or file URL)." }),
12
+ line: Type.Optional(Type.Integer({ minimum: 1, description: "Read the innermost symbol containing this 1-based line. Combine with symbol to disambiguate duplicate names." })),
13
+ column: Type.Optional(Type.Integer({ minimum: 1, description: "Optional 1-based UTF-16 column with line, to distinguish symbols on the same line." })),
14
+ symbol: Type.Optional(Type.String({ minLength: 1, maxLength: 256, description: "Exact symbol name or qualified name, for example Server.run. Omit both symbol and line to list a symbol outline." })),
15
+ parent: Type.Optional(Type.Integer({ minimum: 0, maximum: 20, description: "Move outward this many enclosing symbols after selection (default 0)." })),
16
+ context: Type.Optional(Type.Integer({ minimum: 0, maximum: 20, description: "Extra whole lines before and after the selected symbol (default 0)." })),
17
+ offset: Type.Optional(Type.Integer({ minimum: 1, description: "1-based position within the selected symbol/context or outline, not a file line. Use the exact continuation call from a partial result." })),
18
+ limit: Type.Optional(Type.Integer({ minimum: 1, maximum: 1800, description: "Maximum source lines (default 1000) or outline entries (default 50). Output also has a 48 KiB hard limit." })),
19
+ });
20
+ export type ReadSymbolInput = Static<typeof readSymbolSchema>;
21
+
22
+ export interface VisibleSource {
23
+ startLine: number;
24
+ endLine: number;
25
+ startOffset: number;
26
+ endOffset: number;
27
+ }
28
+ export interface ReadSymbolResult {
29
+ content: Array<{ type: "text"; text: string }>;
30
+ details: {
31
+ mode: "symbol" | "outline";
32
+ snapshot: string;
33
+ symbol?: SourceSymbol;
34
+ visible?: VisibleSource;
35
+ complete: boolean;
36
+ nextCall?: ReadSymbolInput;
37
+ };
38
+ }
39
+ const MAX_OUTPUT_BYTES = 48 * 1024;
40
+ const MAX_OUTPUT_LINES = 1950;
41
+ const label = (text: string) => JSON.stringify(text.length > 300 ? `${text.slice(0, 297)}…` : text);
42
+ const callText = (input: ReadSymbolInput) => `read_symbol ${JSON.stringify(input)}`;
43
+
44
+ export function validateReadSymbolInput(input: ReadSymbolInput): void {
45
+ if (!input || typeof input.path !== "string" || !input.path.length || input.path.length > 4096) throw new Error("read_symbol requires a non-empty path of at most 4096 characters.");
46
+ if (input.symbol !== undefined && (typeof input.symbol !== "string" || !input.symbol.trim() || input.symbol.length > 256)) throw new Error("read_symbol symbol must be a non-empty name of at most 256 characters.");
47
+ for (const [key, min, max] of [["line", 1, Number.MAX_SAFE_INTEGER], ["column", 1, Number.MAX_SAFE_INTEGER], ["parent", 0, 20], ["context", 0, 20], ["offset", 1, Number.MAX_SAFE_INTEGER], ["limit", 1, 1800]] as const) {
48
+ const value = input[key];
49
+ if (value !== undefined && (!Number.isSafeInteger(value) || value < min || value > max)) throw new Error(`read_symbol ${key} must be an integer from ${min} to ${max}.`);
50
+ }
51
+ if (input.column !== undefined && input.line === undefined) throw new Error("read_symbol column requires line.");
52
+ if ((input.parent || input.context) && input.line === undefined && input.symbol === undefined) throw new Error("read_symbol parent/context requires a line or symbol selector; omit them to get an outline.");
53
+ }
54
+
55
+ function candidateLine(path: string, symbol: SourceSymbol): string {
56
+ return `- ${label(symbol.qualifiedName)} (${symbol.kind}), lines ${symbol.startLine}-${symbol.endLine}. ${callText({ path, line: symbol.startLine, column: symbol.startColumn, symbol: symbol.qualifiedName.length <= 256 ? symbol.qualifiedName : undefined })}`;
57
+ }
58
+
59
+ /** Non-copyable preview for failed selection; never presented as a complete symbol. */
60
+ function fallbackContext(input: ReadSymbolInput, content: string, reason: string, atLine = input.line ?? 1): Error {
61
+ const lines = content.split("\n");
62
+ const center = Math.min(Math.max(atLine, 1), lines.length);
63
+ const start = Math.max(1, center - 5);
64
+ const end = Math.min(lines.length, center + 10);
65
+ const preview = lines.slice(start - 1, end).map((line, index) => `${start + index}: ${line.length > 240 ? `${line.slice(0, 240)}… [line clipped]` : line}`).join("\n");
66
+ return new Error(`${reason}\nFile has ${lines.length} lines. Preview only, not a complete symbol or retryable edit snippet:\n${preview}\nNext: read ${JSON.stringify({ path: input.path, offset: start, limit: end - start + 1 })}`);
67
+ }
68
+
69
+ function inside(symbol: SourceSymbol, line: number, column?: number): boolean {
70
+ if (line < symbol.startLine || line > symbol.endLine) return false;
71
+ if (column !== undefined && ((line === symbol.startLine && column < symbol.startColumn) || (line === symbol.endLine && column >= symbol.endColumn))) return false;
72
+ return true;
73
+ }
74
+
75
+ function selectSymbol(input: ReadSymbolInput, symbols: SourceSymbol[]): SourceSymbol {
76
+ let matches = symbols.filter((symbol) =>
77
+ (input.symbol === undefined || symbol.name === input.symbol || symbol.qualifiedName === input.symbol) &&
78
+ (input.line === undefined || inside(symbol, input.line, input.column)),
79
+ );
80
+ if (input.symbol === undefined) {
81
+ // Keep innermost declarations, but never choose arbitrarily between siblings.
82
+ const matched = new Set(matches);
83
+ const ancestors = new Set<SourceSymbol>();
84
+ // Parents precede children in the index. Propagate once in reverse,
85
+ // rather than walking every ancestor chain in deeply nested source.
86
+ for (let index = symbols.length - 1; index >= 0; index--) {
87
+ const symbol = symbols[index];
88
+ if (symbol.parent !== undefined && (matched.has(symbol) || ancestors.has(symbol))) ancestors.add(symbols[symbol.parent]);
89
+ }
90
+ matches = matches.filter((symbol) => !ancestors.has(symbol));
91
+ }
92
+ if (matches.length !== 1) {
93
+ const candidates = matches.length ? matches : [...symbols].sort((a, b) => {
94
+ if (input.line !== undefined) return Math.abs(a.startLine - input.line) - Math.abs(b.startLine - input.line);
95
+ const query = input.symbol?.toLowerCase() ?? "";
96
+ return Number(b.qualifiedName.toLowerCase().includes(query)) - Number(a.qualifiedName.toLowerCase().includes(query));
97
+ });
98
+ throw new Error([
99
+ matches.length ? `Ambiguous symbol selection: ${matches.length} candidates. No symbol was selected.` : "No matching symbol. Names are exact and case-sensitive; no nearby symbol was selected automatically.",
100
+ ...candidates.slice(0, 8).map((symbol) => candidateLine(input.path, symbol)),
101
+ ...(candidates.length > 8 ? [`${candidates.length - 8} more candidates omitted.`] : []),
102
+ `Next: choose a candidate call above, or list the outline with ${callText({ path: input.path })}.`,
103
+ ].join("\n"));
104
+ }
105
+ let selected = matches[0];
106
+ for (let depth = 0; depth < (input.parent ?? 0); depth++) {
107
+ if (selected.parent === undefined) throw new Error(`No enclosing symbol at parent=${input.parent}. Outermost available: ${candidateLine(input.path, selected)}\nRetry with a smaller parent value.`);
108
+ selected = symbols[selected.parent];
109
+ }
110
+ return selected;
111
+ }
112
+
113
+ function fits(text: string): boolean {
114
+ return Buffer.byteLength(text, "utf8") <= MAX_OUTPUT_BYTES && text.split("\n").length <= MAX_OUTPUT_LINES;
115
+ }
116
+
117
+ /** Pure snapshot-to-result path, also used to verify stored read evidence. */
118
+ export async function buildSymbolRead(input: ReadSymbolInput, content: string, signal?: AbortSignal): Promise<ReadSymbolResult> {
119
+ validateReadSymbolInput(input);
120
+ const lines = content.split("\n");
121
+ if (input.line !== undefined && input.line > lines.length) throw fallbackContext(input, content, `line=${input.line} is beyond EOF. Valid lines: 1-${lines.length}.`);
122
+ if (input.column !== undefined && input.column > lines[input.line! - 1].length + 1) throw fallbackContext(input, content, `column=${input.column} is outside line ${input.line}.`);
123
+ const parserPath = resolveToolPath(input.path, "/");
124
+ if (!languageForPath(parserPath)) throw fallbackContext(input, content, `Unsupported source type. read_symbol supports ${SUPPORTED_LANGUAGES}. Use read for this file.`);
125
+ let index: Awaited<ReturnType<typeof indexSymbols>>;
126
+ try { index = await indexSymbols(parserPath, content, signal); }
127
+ catch (error) {
128
+ signal?.throwIfAborted();
129
+ throw fallbackContext(input, content, `Symbol parser unavailable or analysis limit reached: ${error instanceof Error ? error.message.slice(0, 500) : "unknown parser error"}. Use read instead.`);
130
+ }
131
+ signal?.throwIfAborted();
132
+ if (index.errorLine !== undefined) throw fallbackContext(input, content, `Syntax error or incomplete syntax near line ${index.errorLine}; complete symbol boundaries are not reliable.`, input.line ?? index.errorLine);
133
+ const snapshot = createHash("sha256").update(content).digest("hex");
134
+ const header = `Source ${label(input.path)} (${lines.length} lines). Snapshot sha256:${snapshot}`;
135
+ const offset = input.offset ?? 1;
136
+ if (input.line === undefined && input.symbol === undefined) {
137
+ const total = index.symbols.length;
138
+ if (offset > Math.max(1, total)) throw new Error(`Outline offset=${offset} is beyond ${total} entries. Next: ${callText({ ...input, offset: Math.max(1, total - 49) })}`);
139
+ const entries: string[] = [];
140
+ let cursor = offset - 1;
141
+ for (; cursor < Math.min(total, offset - 1 + (input.limit ?? 50)); cursor++) {
142
+ const entry = candidateLine(input.path, index.symbols[cursor]);
143
+ if (!fits(`${header}\n${entries.join("\n")}\n${entry}\n${" ".repeat(6000)}`)) break;
144
+ entries.push(entry);
145
+ }
146
+ const nextCall = cursor < total ? { ...input, offset: cursor + 1 } : undefined;
147
+ const text = `${header}\nSymbol outline: ${total} declarations, ${entries.length} shown.${total === 0 ? " No symbols found; use read for source text." : ""}\n${entries.join("\n")}${nextCall ? `\nMore entries. Next: ${callText(nextCall)}` : ""}`;
148
+ return { content: [{ type: "text", text }], details: { mode: "outline", snapshot, complete: !nextCall, nextCall } };
149
+ }
150
+ if (!index.symbols.length) throw fallbackContext(input, content, "No symbol declarations found in this file.");
151
+ const selected = selectSymbol(input, index.symbols);
152
+ const start = Math.max(1, selected.startLine - (input.context ?? 0));
153
+ const end = Math.min(lines.length, selected.endLine + (input.context ?? 0));
154
+ const count = end - start + 1;
155
+ if (offset > count) throw new Error(`Symbol ${label(selected.qualifiedName)} spans lines ${selected.startLine}-${selected.endLine}; selected region has ${count} lines. offset=${offset} is outside it. Next: ${callText({ ...input, offset: 1 })}`);
156
+ const firstLine = start + offset - 1;
157
+ let lastLine = Math.min(end, firstLine + (input.limit ?? 1000) - 1);
158
+ const parents: string[] = [];
159
+ let parent = selected.parent;
160
+ while (parent !== undefined && parents.length < 20) { parents.unshift(label(index.symbols[parent].qualifiedName)); parent = index.symbols[parent].parent; }
161
+ const prefix = `${header}\nSymbol ${label(selected.qualifiedName)} (${selected.kind}), lines ${selected.startLine}-${selected.endLine}.${parents.length ? `\nEnclosing: ${parents.join(" > ")}` : ""}`;
162
+ let text = "";
163
+ let snippet = "";
164
+ let nextCall: ReadSymbolInput | undefined;
165
+ // Whole-line pages only. Account for fences, metadata, and continuation JSON.
166
+ while (lastLine >= firstLine) {
167
+ snippet = lines.slice(firstLine - 1, lastLine).join("\n");
168
+ const fence = fenceFor(snippet);
169
+ nextCall = lastLine < end ? { ...input, offset: lastLine - start + 2 } : undefined;
170
+ text = `${prefix}\nShowing file lines ${firstLine}-${lastLine} of selected lines ${start}-${end}. ${firstLine === start && lastLine === end ? "Complete selection." : "Partial selection; do not treat this page as the whole symbol."}\n\n${fence}\n${snippet}\n${fence}${nextCall ? `\n\nNext: ${callText(nextCall)}` : ""}`;
171
+ if (fits(text)) break;
172
+ // Remove a proportional chunk first; small pages shrink one line at a time.
173
+ lastLine -= Math.max(1, Math.floor((lastLine - firstLine + 1) / 4));
174
+ }
175
+ if (lastLine < firstLine) throw fallbackContext(input, content, `File line ${firstLine} cannot fit as a complete line in the 48 KiB output budget. No partial edit snippet was returned.`, firstLine);
176
+ const spans = getLineSpans(content);
177
+ const startOffset = spans[firstLine - 1]?.start ?? content.length;
178
+ return {
179
+ content: [{ type: "text", text }],
180
+ details: {
181
+ mode: "symbol", snapshot, symbol: selected,
182
+ visible: { startLine: firstLine, endLine: lastLine, startOffset, endOffset: startOffset + snippet.length },
183
+ complete: firstLine === start && lastLine === end, nextCall,
184
+ },
185
+ };
186
+ }
187
+
188
+ export async function executeReadSymbol(input: ReadSymbolInput, signal: AbortSignal | undefined, ctx: Pick<ExtensionContext, "cwd">): Promise<ReadSymbolResult> {
189
+ validateReadSymbolInput(input);
190
+ const path = resolveToolPath(input.path, ctx.cwd);
191
+ let content: string;
192
+ try { content = await readSource(path, signal); }
193
+ catch (error) { signal?.throwIfAborted(); throw new Error(boundCompleteOutput(await sourceReadError(path, error))); }
194
+ try { return await buildSymbolRead(input, content, signal); }
195
+ catch (error) {
196
+ signal?.throwIfAborted();
197
+ throw new Error(boundCompleteOutput(error instanceof Error ? error.message : String(error)));
198
+ }
199
+ }
200
+
201
+ export function registerReadSymbolTool(pi: ExtensionAPI): void {
202
+ pi.registerTool({
203
+ name: "read_symbol", label: "read_symbol",
204
+ description: `Read a complete function, method, class, or type from a local source file by containing line or exact symbol name. Supports ${SUPPORTED_LANGUAGES}. With path only, lists a symbol outline. Includes enclosing names, exact source ranges, and actionable failure context. Output is bounded to 48 KiB/1950 lines; partial results include an exact continuation call. Does not replace read for ordinary text or images.`,
205
+ promptSnippet: "Read whole source symbols by line/name, or list a file's symbol outline",
206
+ promptGuidelines: [
207
+ "Use read_symbol with path and line after grep/edit identifies a code location, instead of guessing successive read offsets to find the function boundary.",
208
+ "Use read_symbol with symbol for an exact name or qualified name; use path alone for an outline. Use parent to include an enclosing function or class.",
209
+ "For read_symbol partial output, use the supplied continuation arguments. offset is relative to the selection, not a file line. Only displayed source is evidence for edit.",
210
+ ],
211
+ parameters: readSymbolSchema,
212
+ async execute(_id, input, signal, _onUpdate, ctx) { return executeReadSymbol(input, signal, ctx); },
213
+ });
214
+ }
package/src/similarity.ts CHANGED
@@ -3,16 +3,20 @@
3
3
  * failed oldText almost matched.
4
4
  */
5
5
 
6
- import { toFuzzyLines } from "./text.ts";
6
+ import { normalizeForFuzzyMatch, normalizeToLF } from "./text.ts";
7
7
 
8
8
  export interface AlignOp {
9
- type: "equal" | "changed" | "file-only" | "old-only";
10
- /** 1-based line number in the file (for equal/changed/file-only). */
9
+ type: "equal" | "similar" | "changed" | "file-only" | "old-only";
10
+ /** 1-based line number in the file (for equal/similar/changed/file-only). */
11
11
  fileLine?: number;
12
+ /** Original LF-normalized file text, never fuzzy-normalized. */
12
13
  fileText?: string;
13
14
  /** 1-based line number in the model-provided oldText. */
14
15
  oldLine?: number;
16
+ /** Original LF-normalized oldText, never fuzzy-normalized. */
15
17
  oldText?: string;
18
+ /** Heuristic similarity for aligned pairs. */
19
+ similarity?: number;
16
20
  }
17
21
 
18
22
  export interface ClosestRegion {
@@ -21,10 +25,15 @@ export interface ClosestRegion {
21
25
  /** 0..1 order-preserving, one-to-one line similarity. */
22
26
  score: number;
23
27
  ops: AlignOp[];
28
+ /** Number of exactly equal original lines. */
24
29
  equalCount: number;
25
30
  totalOldLines: number;
26
31
  /** oldText was truncated for comparison (very large oldText). */
27
32
  truncated: boolean;
33
+ /** Best distinct, non-overlapping competitor, when retained by the bounded search. */
34
+ competitor?: { startLine: number; endLine: number; score: number };
35
+ /** A discarded window was too competitive to prove a safe score margin. */
36
+ competitionIncomplete: boolean;
28
37
  }
29
38
 
30
39
  interface PreparedLine {
@@ -42,15 +51,28 @@ function prepareLine(text: string): PreparedLine {
42
51
  return { text, grams, gramCount: Math.max(0, text.length - 1) };
43
52
  }
44
53
 
54
+ class WorkBudget {
55
+ private remaining: number;
56
+ constructor(remaining: number) {
57
+ this.remaining = remaining;
58
+ }
59
+ spend(amount: number): boolean {
60
+ this.remaining -= amount;
61
+ return this.remaining >= 0;
62
+ }
63
+ }
64
+
45
65
  /** Dice coefficient over character bigrams; 1.0 for identical strings. */
46
66
  export function lineSimilarity(a: string, b: string): number {
47
67
  return preparedSimilarity(prepareLine(a), prepareLine(b));
48
68
  }
49
69
 
50
- function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
70
+ function preparedSimilarity(a: PreparedLine, b: PreparedLine, budget?: WorkBudget): number {
71
+ if (budget && !budget.spend(1)) throw new Error("similarity-work-budget-exhausted");
51
72
  if (a.text === b.text) return 1;
52
73
  if (a.gramCount === 0 || b.gramCount === 0) return 0;
53
74
  const [small, large] = a.grams.size <= b.grams.size ? [a.grams, b.grams] : [b.grams, a.grams];
75
+ if (budget && !budget.spend(small.size)) throw new Error("similarity-work-budget-exhausted");
54
76
  let overlap = 0;
55
77
  for (const [gram, count] of small) {
56
78
  const other = large.get(gram);
@@ -61,7 +83,10 @@ function preparedSimilarity(a: PreparedLine, b: PreparedLine): number {
61
83
 
62
84
  const MAX_COMPARE_LINES = 300;
63
85
  const MAX_SEARCH_FILE_LINES = 10_000;
64
- const MAX_CANDIDATE_WINDOWS = 24;
86
+ const MAX_CANDIDATE_WINDOWS = 48;
87
+ const MAX_QUERY_BYTES = 128 * 1024;
88
+ const MAX_LINE_CHARS = 16 * 1024;
89
+ const MAX_SIMILARITY_WORK = 20_000_000;
65
90
  const MIN_SCORE = 0.3;
66
91
  const MATCH_THRESHOLD = 0.75;
67
92
 
@@ -71,81 +96,122 @@ interface CandidateWindow {
71
96
  positionalScore: number;
72
97
  }
73
98
 
99
+ interface ScoredWindow extends CandidateWindow {
100
+ score: number;
101
+ }
102
+
103
+ function originalLines(text: string): string[] {
104
+ const lines = normalizeToLF(text).split("\n");
105
+ // Remove only the synthetic segment after a terminal newline. Do not remove
106
+ // a real, unterminated whitespace-only line.
107
+ if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
108
+ return lines;
109
+ }
110
+
74
111
  /**
75
- * Find the window of file lines most similar to oldText.
76
- *
77
- * The first pass ranks every window using cheap positional similarity. Only a
78
- * small bounded set of candidates receives the more expensive sequence score.
79
- * The sequence score is order-preserving and one-to-one, so repeated query
80
- * lines cannot all claim the same file line.
112
+ * Find the best window and a distinct runner-up under explicit input/work
113
+ * budgets. Returning null is the safe fallback when a budget is exhausted.
81
114
  */
82
115
  export function findClosestRegion(content: string, oldText: string): ClosestRegion | null {
83
- const allQLines = toFuzzyLines(oldText);
84
- const cLines = toFuzzyLines(content);
85
- if (allQLines.length === 0 || cLines.length === 0 || cLines.length > MAX_SEARCH_FILE_LINES) return null;
116
+ if (Buffer.byteLength(oldText, "utf8") > MAX_QUERY_BYTES) return null;
117
+ const originalQ = originalLines(oldText);
118
+ const originalContent = originalLines(content);
119
+ if (originalQ.length === 0 || originalContent.length === 0 || originalContent.length > MAX_SEARCH_FILE_LINES) return null;
120
+ if ([...originalQ, ...originalContent].some((line) => line.length > MAX_LINE_CHARS)) return null;
86
121
 
122
+ const allQLines = originalQ.map(normalizeForFuzzyMatch);
123
+ const cLines = originalContent.map(normalizeForFuzzyMatch);
87
124
  const truncated = allQLines.length > MAX_COMPARE_LINES;
88
125
  const q = truncated ? allQLines.slice(0, MAX_COMPARE_LINES) : allQLines;
126
+ const qOriginal = truncated ? originalQ.slice(0, MAX_COMPARE_LINES) : originalQ;
89
127
  const L = q.length;
90
128
  const sizes = [...new Set([L - 1, L, L + 1].filter((size) => size >= 1 && size <= cLines.length))];
91
129
  const preparedQ = q.map(prepareLine);
92
130
  const preparedContent = cLines.map(prepareLine);
93
131
  const candidates: CandidateWindow[] = [];
132
+ let discardedMaxPositionalScore = -1;
133
+ const budget = new WorkBudget(MAX_SIMILARITY_WORK);
94
134
 
95
- for (const size of sizes) {
96
- for (let start = 0; start + size <= cLines.length; start++) {
97
- const pairs = Math.min(L, size);
98
- let sum = 0;
99
- for (let i = 0; i < pairs; i++) {
100
- sum += preparedSimilarity(preparedQ[i], preparedContent[start + i]);
135
+ try {
136
+ for (const size of sizes) {
137
+ for (let start = 0; start + size <= cLines.length; start++) {
138
+ const pairs = Math.min(L, size);
139
+ let sum = 0;
140
+ for (let i = 0; i < pairs; i++) {
141
+ sum += preparedSimilarity(preparedQ[i], preparedContent[start + i], budget);
142
+ }
143
+ const discarded = keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
144
+ if (discarded !== undefined) discardedMaxPositionalScore = Math.max(discardedMaxPositionalScore, discarded);
101
145
  }
102
- keepCandidate(candidates, { start, size, positionalScore: sum / Math.max(L, size) });
103
146
  }
104
- }
105
147
 
106
- let bestScore = -1;
107
- let best: CandidateWindow | undefined;
108
- for (const candidate of candidates) {
109
- const window = preparedContent.slice(candidate.start, candidate.start + candidate.size);
110
- const score = sequenceSimilarity(preparedQ, window);
111
- if (score > bestScore || (score === bestScore && candidate.positionalScore > (best?.positionalScore ?? -1))) {
112
- bestScore = score;
113
- best = candidate;
114
- }
148
+ const scored: ScoredWindow[] = candidates.map((candidate) => ({
149
+ ...candidate,
150
+ score: sequenceSimilarity(
151
+ preparedQ,
152
+ preparedContent.slice(candidate.start, candidate.start + candidate.size),
153
+ budget,
154
+ ),
155
+ }));
156
+ scored.sort(
157
+ (a, b) => b.score - a.score || b.positionalScore - a.positionalScore || a.start - b.start || a.size - b.size,
158
+ );
159
+ const best = scored[0];
160
+ if (!best || best.score < MIN_SCORE) return null;
161
+ const bestEnd = best.start + best.size;
162
+ const runner = scored.find((candidate) => {
163
+ const candidateEnd = candidate.start + candidate.size;
164
+ return candidateEnd <= best.start || candidate.start >= bestEnd;
165
+ });
166
+
167
+ const windowLines = cLines.slice(best.start, bestEnd);
168
+ const windowOriginal = originalContent.slice(best.start, bestEnd);
169
+ const ops = alignLines(q, windowLines, qOriginal, windowOriginal, best.start, budget);
170
+ return {
171
+ startLine: best.start + 1,
172
+ endLine: bestEnd,
173
+ score: best.score,
174
+ ops,
175
+ equalCount: ops.filter((op) => op.type === "equal").length,
176
+ totalOldLines: L,
177
+ truncated,
178
+ competitor: runner
179
+ ? { startLine: runner.start + 1, endLine: runner.start + runner.size, score: runner.score }
180
+ : undefined,
181
+ competitionIncomplete: discardedMaxPositionalScore >= best.positionalScore - MIN_DIRECT_POSITIONAL_GAP,
182
+ };
183
+ } catch (error) {
184
+ if (error instanceof Error && error.message === "similarity-work-budget-exhausted") return null;
185
+ throw error;
115
186
  }
116
- if (!best || bestScore < MIN_SCORE) return null;
117
-
118
- const windowLines = cLines.slice(best.start, best.start + best.size);
119
- const ops = alignLines(q, windowLines, best.start);
120
- return {
121
- startLine: best.start + 1,
122
- endLine: best.start + best.size,
123
- score: bestScore,
124
- ops,
125
- equalCount: ops.filter((op) => op.type === "equal").length,
126
- totalOldLines: L,
127
- truncated,
128
- };
129
187
  }
130
188
 
131
- function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): void {
189
+ const MIN_DIRECT_POSITIONAL_GAP = 0.1;
190
+
191
+ /** Return the positional score of any candidate discarded by the bounded heap. */
192
+ function keepCandidate(candidates: CandidateWindow[], candidate: CandidateWindow): number | undefined {
132
193
  if (candidates.length < MAX_CANDIDATE_WINDOWS) {
133
194
  candidates.push(candidate);
134
- candidates.sort((a, b) => a.positionalScore - b.positionalScore);
135
- return;
195
+ candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
196
+ return undefined;
136
197
  }
137
- if (candidate.positionalScore <= candidates[0].positionalScore) return;
198
+ const worst = candidates[0];
199
+ if (
200
+ candidate.positionalScore < worst.positionalScore ||
201
+ (candidate.positionalScore === worst.positionalScore && candidate.start >= worst.start)
202
+ ) return candidate.positionalScore;
138
203
  candidates[0] = candidate;
139
- candidates.sort((a, b) => a.positionalScore - b.positionalScore);
204
+ candidates.sort((a, b) => a.positionalScore - b.positionalScore || b.start - a.start || b.size - a.size);
205
+ return worst.positionalScore;
140
206
  }
141
207
 
142
208
  /** Weighted LCS: lines are matched at most once and in source order. */
143
- function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): number {
209
+ function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[], budget: WorkBudget): number {
144
210
  let previous = new Float64Array(window.length + 1);
145
211
  for (let i = 1; i <= query.length; i++) {
146
212
  const current = new Float64Array(window.length + 1);
147
213
  for (let j = 1; j <= window.length; j++) {
148
- const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1]);
214
+ const matched = previous[j - 1] + preparedSimilarity(query[i - 1], window[j - 1], budget);
149
215
  current[j] = Math.max(previous[j], current[j - 1], matched);
150
216
  }
151
217
  previous = current;
@@ -153,18 +219,21 @@ function sequenceSimilarity(query: PreparedLine[], window: PreparedLine[]): numb
153
219
  return previous[window.length] / Math.max(query.length, window.length);
154
220
  }
155
221
 
156
- /**
157
- * LCS-style alignment between oldText lines and the best window. Lines with
158
- * similarity >= MATCH_THRESHOLD count as equal; adjacent insert/delete pairs
159
- * are merged into changed operations for compact diagnostics.
160
- */
161
- function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
222
+ /** Align fuzzy lines while attaching original strings to every operation. */
223
+ function alignLines(
224
+ q: string[],
225
+ w: string[],
226
+ qOriginal: string[],
227
+ wOriginal: string[],
228
+ wOffset: number,
229
+ budget: WorkBudget,
230
+ ): AlignOp[] {
162
231
  const n = q.length;
163
232
  const m = w.length;
164
233
  const preparedQ = q.map(prepareLine);
165
234
  const preparedW = w.map(prepareLine);
166
235
  const similarities = Array.from({ length: n }, (_, i) =>
167
- Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j])),
236
+ Array.from({ length: m }, (_, j) => preparedSimilarity(preparedQ[i], preparedW[j], budget)),
168
237
  );
169
238
  const dp: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
170
239
  for (let i = n - 1; i >= 0; i--) {
@@ -180,17 +249,25 @@ function alignLines(q: string[], w: string[], wOffset: number): AlignOp[] {
180
249
  let j = 0;
181
250
  while (i < n && j < m) {
182
251
  if (similarities[i][j] >= MATCH_THRESHOLD) {
183
- raw.push({ type: "equal", fileLine: wOffset + j + 1, fileText: w[j], oldLine: i + 1, oldText: q[i] });
252
+ const exact = qOriginal[i] === wOriginal[j];
253
+ raw.push({
254
+ type: exact ? "equal" : "similar",
255
+ fileLine: wOffset + j + 1,
256
+ fileText: wOriginal[j],
257
+ oldLine: i + 1,
258
+ oldText: qOriginal[i],
259
+ similarity: similarities[i][j],
260
+ });
184
261
  i++;
185
262
  j++;
186
263
  } else if (dp[i + 1][j] > dp[i][j + 1]) {
187
- raw.push({ type: "old-only", oldLine: i + 1, oldText: q[i++] });
264
+ raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
188
265
  } else {
189
- raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: w[j++] });
266
+ raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
190
267
  }
191
268
  }
192
- while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: q[i++] });
193
- while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: w[j++] });
269
+ while (i < n) raw.push({ type: "old-only", oldLine: i + 1, oldText: qOriginal[i++] });
270
+ while (j < m) raw.push({ type: "file-only", fileLine: wOffset + j + 1, fileText: wOriginal[j++] });
194
271
 
195
272
  const merged: AlignOp[] = [];
196
273
  for (let k = 0; k < raw.length; k++) {
@@ -0,0 +1,60 @@
1
+ import { constants } from "node:fs";
2
+ import { open, opendir } from "node:fs/promises";
3
+ import { basename, dirname, join } from "node:path";
4
+ import { MAX_SOURCE_BYTES } from "./symbols.ts";
5
+ import { normalizeToLF, splitBom } from "./text.ts";
6
+
7
+ /** Bounded local source read; no silent decoding loss and no special files. */
8
+ export async function readSource(path: string, signal?: AbortSignal): Promise<string> {
9
+ signal?.throwIfAborted();
10
+ const handle = await open(path, constants.O_RDONLY | constants.O_NONBLOCK);
11
+ try {
12
+ const stat = await handle.stat();
13
+ if (!stat.isFile()) throw new Error("The path is not a regular source file.");
14
+ if (stat.size > MAX_SOURCE_BYTES) throw new Error("Source exceeds the 2 MiB symbol analysis limit; use read with offset/limit.");
15
+ const buffer = Buffer.alloc(Math.min(stat.size + 1, MAX_SOURCE_BYTES + 1));
16
+ let length = 0;
17
+ while (length < buffer.length) {
18
+ signal?.throwIfAborted();
19
+ const { bytesRead } = await handle.read(buffer, length, buffer.length - length, null);
20
+ if (!bytesRead) break;
21
+ length += bytesRead;
22
+ }
23
+ const after = await handle.stat();
24
+ if (length !== stat.size || stat.size !== after.size || stat.mtimeMs !== after.mtimeMs || stat.ctimeMs !== after.ctimeMs) throw new Error("The source changed during the read; retry read_symbol.");
25
+ const bytes = buffer.subarray(0, length);
26
+ let content: string;
27
+ try { content = new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }).decode(bytes); }
28
+ catch { throw new Error("The source is not valid UTF-8; convert its encoding before symbol parsing."); }
29
+ if (bytes.includes(0)) throw new Error("The source contains NUL bytes; binary/UTF-16 input is not supported.");
30
+ signal?.throwIfAborted();
31
+ return splitBom(normalizeToLF(content)).text;
32
+ } finally {
33
+ await handle.close();
34
+ }
35
+ }
36
+
37
+ /** Only inspect a bounded number of siblings; never perform a hidden repo scan. */
38
+ export async function sourceReadError(path: string, error: unknown): Promise<string> {
39
+ const code = (error as NodeJS.ErrnoException)?.code;
40
+ const message = error instanceof Error ? error.message.slice(0, 1000) : String(error).slice(0, 1000);
41
+ const lines = [`Could not read source ${JSON.stringify(path)}: ${message}`];
42
+ if (code === "ENOENT" || code === "ENOTDIR") {
43
+ const names: string[] = [];
44
+ try {
45
+ const directory = await opendir(dirname(path));
46
+ let inspected = 0;
47
+ for await (const entry of directory) {
48
+ if (entry.isFile()) names.push(entry.name);
49
+ if (++inspected >= 100) break;
50
+ }
51
+ } catch { /* Parent may also be missing or inaccessible. */ }
52
+ const stem = basename(path).split(".")[0].toLowerCase();
53
+ names.sort((a, b) => Number(b.toLowerCase().includes(stem)) - Number(a.toLowerCase().includes(stem)) || a.localeCompare(b));
54
+ if (names.length) lines.push("Nearby file candidates (bounded listing, not automatic path corrections):", ...names.slice(0, 8).map((name) => ` ${JSON.stringify(join(dirname(path), name))}`));
55
+ lines.push("Verify the path and retry read_symbol; use find/ls if the parent path is wrong.");
56
+ } else if (code === "EACCES" || code === "EPERM") {
57
+ lines.push("Check file permissions. Do not retry the same call until access changes.");
58
+ } else lines.push("Use read for ordinary text, images, or bounded line ranges; no symbol boundaries were returned.");
59
+ return lines.join("\n");
60
+ }