pi-hashline-edit-pro 2.7.1 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -34
- package/index.ts +11 -9
- package/package.json +2 -2
- package/prompts/grep-guidelines.md +3 -6
- package/prompts/grep-snippet.md +1 -1
- package/prompts/grep.md +1 -1
- package/prompts/insert-guidelines.md +3 -6
- package/prompts/insert-snippet.md +1 -1
- package/prompts/insert.md +1 -1
- package/prompts/read-guidelines.md +2 -2
- package/prompts/read-snippet.md +1 -1
- package/prompts/read.md +1 -1
- package/prompts/replace-guidelines.md +5 -8
- package/prompts/replace-snippet.md +1 -1
- package/prompts/replace.md +1 -1
- package/prompts/undo-last-change-guidelines.md +2 -3
- package/prompts/undo-last-change-snippet.md +1 -1
- package/prompts/undo-last-change.md +1 -1
- package/src/constants.ts +4 -2
- package/src/file-kind.ts +8 -6
- package/src/file-reader.ts +25 -11
- package/src/fs-write.ts +28 -2
- package/src/grep.ts +137 -43
- package/src/hash-store.ts +57 -17
- package/src/hashline/hash.ts +16 -9
- package/src/hashline/index.ts +1 -0
- package/src/hashline/parse.ts +17 -3
- package/src/hashline/resolve.ts +9 -8
- package/src/insert.ts +11 -16
- package/src/payload-contract.ts +102 -0
- package/src/read.ts +32 -11
- package/src/replace-diff.ts +108 -19
- package/src/replace-render.ts +47 -43
- package/src/replace-response.ts +4 -1
- package/src/replace-undo.ts +17 -12
- package/src/replace.ts +44 -106
- package/src/served.ts +26 -5
- package/src/utils.ts +71 -0
- package/src/write-hook.ts +59 -0
- package/src/replace-normalize.ts +0 -13
package/src/fs-write.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { randomUUID } from "crypto";
|
|
2
2
|
import {
|
|
3
|
+
chmod,
|
|
3
4
|
lstat,
|
|
4
5
|
mkdir,
|
|
5
6
|
open,
|
|
@@ -11,6 +12,7 @@ import {
|
|
|
11
12
|
writeFile,
|
|
12
13
|
} from "fs/promises";
|
|
13
14
|
import { dirname, join, parse, resolve, sep } from "path";
|
|
15
|
+
import { toCwd } from "./paths";
|
|
14
16
|
import { errCode } from "./utils";
|
|
15
17
|
|
|
16
18
|
export async function resolveTarget(path: string): Promise<string> {
|
|
@@ -25,7 +27,15 @@ export async function resolveTarget(path: string): Promise<string> {
|
|
|
25
27
|
async function resParts(
|
|
26
28
|
currentPath: string,
|
|
27
29
|
remainingParts: string[],
|
|
30
|
+
symlinkDepth = 0,
|
|
28
31
|
): Promise<string> {
|
|
32
|
+
if (symlinkDepth > 40) {
|
|
33
|
+
const error = new Error(
|
|
34
|
+
`Too many symbolic links while resolving ${path}`,
|
|
35
|
+
) as NodeJS.ErrnoException;
|
|
36
|
+
error.code = "ELOOP";
|
|
37
|
+
throw error;
|
|
38
|
+
}
|
|
29
39
|
if (remainingParts.length === 0) {
|
|
30
40
|
return currentPath;
|
|
31
41
|
}
|
|
@@ -36,7 +46,7 @@ export async function resolveTarget(path: string): Promise<string> {
|
|
|
36
46
|
try {
|
|
37
47
|
const candidateStats = await lstat(candidatePath);
|
|
38
48
|
if (!candidateStats.isSymbolicLink()) {
|
|
39
|
-
return resParts(candidatePath, tail);
|
|
49
|
+
return resParts(candidatePath, tail, symlinkDepth);
|
|
40
50
|
}
|
|
41
51
|
|
|
42
52
|
if (visitedSymlinks.has(candidatePath)) {
|
|
@@ -59,7 +69,7 @@ export async function resolveTarget(path: string): Promise<string> {
|
|
|
59
69
|
return resParts(parse(linkTargetPath).root, [
|
|
60
70
|
...targetParts,
|
|
61
71
|
...tail,
|
|
62
|
-
]);
|
|
72
|
+
], symlinkDepth + 1);
|
|
63
73
|
} catch (error: unknown) {
|
|
64
74
|
if (errCode(error) === "ENOENT") {
|
|
65
75
|
return join(candidatePath, ...tail);
|
|
@@ -110,6 +120,11 @@ async function syncDir(dir: string): Promise<void> {
|
|
|
110
120
|
}
|
|
111
121
|
}
|
|
112
122
|
|
|
123
|
+
export async function resolveInCwd(path: string, cwd: string): Promise<{ absolute: string; resolved: string }> {
|
|
124
|
+
const absolute = toCwd(path, cwd);
|
|
125
|
+
const resolved = await resolveTarget(absolute);
|
|
126
|
+
return { absolute, resolved };
|
|
127
|
+
}
|
|
113
128
|
export async function writeAtomic(
|
|
114
129
|
path: string,
|
|
115
130
|
content: string,
|
|
@@ -127,6 +142,17 @@ export async function writeAtomic(
|
|
|
127
142
|
|
|
128
143
|
if (existingStats && existingStats.nlink > 1) {
|
|
129
144
|
await writeFile(targetPath, content, "utf-8");
|
|
145
|
+
try {
|
|
146
|
+
await chmod(targetPath, existingStats.mode & 0o7777);
|
|
147
|
+
} catch {}
|
|
148
|
+
try {
|
|
149
|
+
const handle = await open(targetPath, "r");
|
|
150
|
+
try {
|
|
151
|
+
await handle.sync();
|
|
152
|
+
} finally {
|
|
153
|
+
await handle.close();
|
|
154
|
+
}
|
|
155
|
+
} catch {}
|
|
130
156
|
return;
|
|
131
157
|
}
|
|
132
158
|
|
package/src/grep.ts
CHANGED
|
@@ -1,20 +1,24 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
import { formatSize, DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, type TruncationResult } from "@earendil-works/pi-coding-agent";
|
|
2
3
|
import { Type } from "typebox";
|
|
3
4
|
import { readdir, stat } from "fs/promises";
|
|
4
5
|
import { dirname, join, relative } from "path";
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
6
|
+
import { tryReadNormFile } from "./file-reader";
|
|
7
|
+
import { MAX_HASH_LINES, fmtRow, HASH_LEN, HASH_SEP } from "./hashline";
|
|
8
|
+
import { MAX_GREP_LINE_BYTES } from "./constants";
|
|
8
9
|
import { toCwd } from "./paths";
|
|
9
10
|
import { loadP, loadGuide } from "./prompts";
|
|
10
|
-
import { normReq } from "./
|
|
11
|
+
import { normReq } from "./payload-contract";
|
|
11
12
|
import { recordServedSafe } from "./served";
|
|
12
|
-
import { abortIf, errCode, isRec, makePrepareArguments, rejectUnknownFields, visLines } from "./utils";
|
|
13
|
+
import { abortIf, errCode, isRec, makePrepareArguments, rejectUnknownFields, truncateToBytes, visLines } from "./utils";
|
|
13
14
|
|
|
14
15
|
const GREP_KS = new Set(["pattern", "path", "glob", "context", "ignoreCase", "literal", "limit"]);
|
|
15
16
|
const SKIP_DIRS = new Set(["node_modules", ".git", ".tmp", "coverage"]);
|
|
16
17
|
const MAX_SCAN_FILES = 4000;
|
|
17
|
-
|
|
18
|
+
|
|
19
|
+
function cmp(a: string, b: string): number {
|
|
20
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
21
|
+
}
|
|
18
22
|
|
|
19
23
|
export interface GrepReq {
|
|
20
24
|
pattern: string;
|
|
@@ -52,6 +56,7 @@ function buildRegex(pattern: string, literal: boolean, ignoreCase: boolean): Reg
|
|
|
52
56
|
}
|
|
53
57
|
|
|
54
58
|
function globToRegex(glob: string): RegExp {
|
|
59
|
+
if (glob.startsWith("/")) glob = glob.slice(1);
|
|
55
60
|
let source = "";
|
|
56
61
|
let i = 0;
|
|
57
62
|
while (i < glob.length) {
|
|
@@ -78,12 +83,6 @@ function globToRegex(glob: string): RegExp {
|
|
|
78
83
|
return new RegExp(`^${source}$`);
|
|
79
84
|
}
|
|
80
85
|
|
|
81
|
-
function isSkipableLoadError(error: unknown): boolean {
|
|
82
|
-
const code = errCode(error);
|
|
83
|
-
if (code === "EACCES" || code === "EPERM" || code === "ENOENT" || code === "ELOOP") return true;
|
|
84
|
-
return error instanceof Error && error.message.startsWith("[E_FILE_TOO_LARGE]");
|
|
85
|
-
}
|
|
86
|
-
|
|
87
86
|
interface FileHit {
|
|
88
87
|
path: string;
|
|
89
88
|
displayPath: string;
|
|
@@ -92,6 +91,42 @@ interface FileHit {
|
|
|
92
91
|
hashes: string[];
|
|
93
92
|
matchCount: number;
|
|
94
93
|
totalMatchCount: number;
|
|
94
|
+
fragmented: boolean[];
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const GREP_ROW_OVERHEAD_BYTES = HASH_LEN + Buffer.byteLength(HASH_SEP, "utf-8");
|
|
98
|
+
const GREP_ROW_CONTENT_BYTES = MAX_GREP_LINE_BYTES - GREP_ROW_OVERHEAD_BYTES;
|
|
99
|
+
|
|
100
|
+
function snapCharBoundaries(line: string, start: number, end: number): [number, number] {
|
|
101
|
+
let s = start;
|
|
102
|
+
let e = end;
|
|
103
|
+
if (s > 0 && s < line.length) {
|
|
104
|
+
const c = line.charCodeAt(s);
|
|
105
|
+
if (c >= 0xdc00 && c <= 0xdfff && line.charCodeAt(s - 1) >= 0xd800 && line.charCodeAt(s - 1) <= 0xdbff) s -= 1;
|
|
106
|
+
}
|
|
107
|
+
if (e > 0 && e < line.length) {
|
|
108
|
+
const c = line.charCodeAt(e - 1);
|
|
109
|
+
if (c >= 0xd800 && c <= 0xdbff && line.charCodeAt(e) >= 0xdc00 && line.charCodeAt(e) <= 0xdfff) e += 1;
|
|
110
|
+
}
|
|
111
|
+
return [s, e];
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
function grepMatchFragment(line: string, regex: RegExp): string {
|
|
115
|
+
const m = regex.exec(line);
|
|
116
|
+
const matchStart = m?.index ?? 0;
|
|
117
|
+
const matchLen = m?.[0].length ?? 0;
|
|
118
|
+
const budget = GREP_ROW_CONTENT_BYTES - 6;
|
|
119
|
+
const half = Math.floor((budget - Math.min(matchLen, budget)) / 2);
|
|
120
|
+
const [start, end] = snapCharBoundaries(line, Math.max(0, matchStart - half), Math.min(line.length, matchStart + matchLen + half));
|
|
121
|
+
const content = truncateToBytes(line.slice(start, end), budget);
|
|
122
|
+
const lead = start > 0 ? "..." : "";
|
|
123
|
+
const tail = end < line.length ? "..." : "";
|
|
124
|
+
return truncateToBytes(`${lead}${content}${tail}`, GREP_ROW_CONTENT_BYTES);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function grepHeadFragment(line: string): string {
|
|
128
|
+
const head = truncateToBytes(line, GREP_ROW_CONTENT_BYTES - 3);
|
|
129
|
+
return head.length < line.length ? `${head}...` : head;
|
|
95
130
|
}
|
|
96
131
|
|
|
97
132
|
interface ScanState {
|
|
@@ -105,14 +140,16 @@ async function walkFiles(
|
|
|
105
140
|
onFile: (absPath: string) => Promise<void>,
|
|
106
141
|
): Promise<void> {
|
|
107
142
|
const queue: string[] = [root];
|
|
108
|
-
|
|
109
|
-
|
|
143
|
+
let head = 0;
|
|
144
|
+
while (head < queue.length && !state.stopped) {
|
|
145
|
+
const dir = queue[head++]!;
|
|
110
146
|
let entries;
|
|
111
147
|
try {
|
|
112
148
|
entries = await readdir(dir, { withFileTypes: true });
|
|
113
149
|
} catch {
|
|
114
150
|
continue;
|
|
115
151
|
}
|
|
152
|
+
entries.sort((a, b) => cmp(a.name, b.name));
|
|
116
153
|
for (const entry of entries) {
|
|
117
154
|
if (state.stopped) break;
|
|
118
155
|
const full = join(dir, entry.name);
|
|
@@ -143,23 +180,10 @@ async function searchFile(
|
|
|
143
180
|
const displayPath = relative(cwd, absPath).replace(/\\/g, "/");
|
|
144
181
|
if (globRegex) {
|
|
145
182
|
const globPath = relative(globRoot, absPath).replace(/\\/g, "/");
|
|
146
|
-
if (!globRegex.test(globPath)) return undefined;
|
|
147
|
-
}
|
|
148
|
-
let file;
|
|
149
|
-
try {
|
|
150
|
-
file = await loadFileKindAndText(absPath, { maxLines: MAX_HASH_LINES, displayPath });
|
|
151
|
-
} catch (error) {
|
|
152
|
-
if (isSkipableLoadError(error)) return undefined;
|
|
153
|
-
throw error;
|
|
154
|
-
}
|
|
155
|
-
if (file.kind !== "text") return undefined;
|
|
156
|
-
let norm;
|
|
157
|
-
try {
|
|
158
|
-
norm = await readNormFile(absPath, cwd, { maxLines: MAX_HASH_LINES, preloadedFile: file, noPersist: true });
|
|
159
|
-
} catch (error) {
|
|
160
|
-
if (isSkipableLoadError(error)) return undefined;
|
|
161
|
-
throw error;
|
|
183
|
+
if (!globRegex.test(globPath) && !globRegex.test(displayPath)) return undefined;
|
|
162
184
|
}
|
|
185
|
+
const norm = await tryReadNormFile(absPath, cwd, { maxLines: MAX_HASH_LINES, noPersist: true });
|
|
186
|
+
if (!norm) return undefined;
|
|
163
187
|
const lines = visLines(norm.normalized);
|
|
164
188
|
const matchLines: number[] = [];
|
|
165
189
|
for (let i = 0; i < lines.length; i++) {
|
|
@@ -172,11 +196,24 @@ async function searchFile(
|
|
|
172
196
|
for (let j = Math.max(0, i - context); j <= Math.min(lines.length - 1, i + context); j++) shown.add(j);
|
|
173
197
|
}
|
|
174
198
|
const sorted = [...shown].sort((a, b) => a - b);
|
|
199
|
+
const matchSet = new Set(matchLines);
|
|
175
200
|
const rows: string[] = [];
|
|
176
201
|
const hashes: string[] = [];
|
|
202
|
+
const fragmented: boolean[] = [];
|
|
177
203
|
for (const idx of sorted) {
|
|
178
|
-
|
|
179
|
-
|
|
204
|
+
const hash = norm.fileHashes[idx]!;
|
|
205
|
+
const line = lines[idx]!;
|
|
206
|
+
const row = fmtRow(hash, line);
|
|
207
|
+
if (Buffer.byteLength(row, "utf-8") > MAX_GREP_LINE_BYTES) {
|
|
208
|
+
const content = matchSet.has(idx) ? grepMatchFragment(line, regex) : grepHeadFragment(line);
|
|
209
|
+
rows.push(fmtRow(hash, content));
|
|
210
|
+
hashes.push(hash);
|
|
211
|
+
fragmented.push(true);
|
|
212
|
+
} else {
|
|
213
|
+
rows.push(row);
|
|
214
|
+
hashes.push(hash);
|
|
215
|
+
fragmented.push(false);
|
|
216
|
+
}
|
|
180
217
|
}
|
|
181
218
|
return {
|
|
182
219
|
path: norm.absolutePath,
|
|
@@ -186,6 +223,7 @@ async function searchFile(
|
|
|
186
223
|
hashes,
|
|
187
224
|
matchCount: keptMatches.length,
|
|
188
225
|
totalMatchCount: matchLines.length,
|
|
226
|
+
fragmented,
|
|
189
227
|
};
|
|
190
228
|
}
|
|
191
229
|
|
|
@@ -201,7 +239,7 @@ const grepToolSchema = Type.Object(
|
|
|
201
239
|
),
|
|
202
240
|
glob: Type.Optional(
|
|
203
241
|
Type.String({
|
|
204
|
-
description: "Filter files by glob pattern; * matches across directories, e.g. '*.ts' or '**/*.spec.ts'",
|
|
242
|
+
description: "Filter files by glob pattern; * matches across directories, e.g. '*.ts' or '**/*.spec.ts'. A leading / is ignored; the pattern may be relative to the search root or to the current directory.",
|
|
205
243
|
}),
|
|
206
244
|
),
|
|
207
245
|
ignoreCase: Type.Optional(
|
|
@@ -239,6 +277,7 @@ export function regGrep(pi: ExtensionAPI): void {
|
|
|
239
277
|
promptGuidelines: loadGuide("../prompts/grep-guidelines.md"),
|
|
240
278
|
prepareArguments: makePrepareArguments(),
|
|
241
279
|
parameters: grepToolSchema,
|
|
280
|
+
executionMode: "sequential",
|
|
242
281
|
|
|
243
282
|
async execute(_toolCallId, params, signal, _onUpdate, ctx) {
|
|
244
283
|
const canonical = normReq(params);
|
|
@@ -268,14 +307,36 @@ export function regGrep(pi: ExtensionAPI): void {
|
|
|
268
307
|
await walkFiles(base, state, async (absPath) => {
|
|
269
308
|
files.push(absPath);
|
|
270
309
|
});
|
|
310
|
+
files.sort(cmp);
|
|
271
311
|
}
|
|
272
312
|
const hits: FileHit[] = [];
|
|
273
313
|
let matches = 0;
|
|
274
314
|
let limitTruncated = false;
|
|
275
315
|
let rowTruncated = false;
|
|
276
316
|
let rowCount = 0;
|
|
277
|
-
|
|
317
|
+
let byteCount = 0;
|
|
318
|
+
let totalRows = 0;
|
|
319
|
+
let totalBytes = 0;
|
|
320
|
+
let truncatedBy: "lines" | "bytes" | null = null;
|
|
321
|
+
let linesReplaced = 0;
|
|
322
|
+
let countOnly = false;
|
|
323
|
+
for (let f = 0; f < files.length; f++) {
|
|
278
324
|
abortIf(signal);
|
|
325
|
+
const absPath = files[f]!;
|
|
326
|
+
if (countOnly) {
|
|
327
|
+
const hit = await searchFile(absPath, globRoot, ctx.cwd, regex, globRegex, context, Number.MAX_SAFE_INTEGER);
|
|
328
|
+
if (!hit) continue;
|
|
329
|
+
totalRows += hit.rows.length;
|
|
330
|
+
for (const row of hit.rows) totalBytes += Buffer.byteLength(row, "utf-8") + 1;
|
|
331
|
+
const remaining = limit - matches;
|
|
332
|
+
if (remaining > 0) {
|
|
333
|
+
matches += Math.min(hit.matchCount, remaining);
|
|
334
|
+
if (hit.matchCount > remaining) limitTruncated = true;
|
|
335
|
+
} else {
|
|
336
|
+
limitTruncated = true;
|
|
337
|
+
}
|
|
338
|
+
continue;
|
|
339
|
+
}
|
|
279
340
|
const remaining = limit - matches;
|
|
280
341
|
if (remaining <= 0) {
|
|
281
342
|
limitTruncated = true;
|
|
@@ -283,19 +344,34 @@ export function regGrep(pi: ExtensionAPI): void {
|
|
|
283
344
|
}
|
|
284
345
|
const hit = await searchFile(absPath, globRoot, ctx.cwd, regex, globRegex, context, remaining);
|
|
285
346
|
if (!hit) continue;
|
|
286
|
-
const
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
347
|
+
const keptRows: string[] = [];
|
|
348
|
+
const keptHashes: string[] = [];
|
|
349
|
+
for (let i = 0; i < hit.rows.length; i++) {
|
|
350
|
+
const row = hit.rows[i]!;
|
|
351
|
+
const rowBytes = Buffer.byteLength(row, "utf-8") + 1;
|
|
352
|
+
if (rowCount >= DEFAULT_MAX_LINES || byteCount + rowBytes > DEFAULT_MAX_BYTES) {
|
|
353
|
+
rowTruncated = true;
|
|
354
|
+
if (truncatedBy === null) truncatedBy = byteCount + rowBytes > DEFAULT_MAX_BYTES ? "bytes" : "lines";
|
|
355
|
+
for (let j = i; j < hit.rows.length; j++) {
|
|
356
|
+
totalRows += 1;
|
|
357
|
+
totalBytes += Buffer.byteLength(hit.rows[j]!, "utf-8") + 1;
|
|
358
|
+
}
|
|
359
|
+
break;
|
|
360
|
+
}
|
|
361
|
+
keptRows.push(row);
|
|
362
|
+
keptHashes.push(hit.hashes[i]);
|
|
363
|
+
if (hit.fragmented[i]) linesReplaced += 1;
|
|
364
|
+
rowCount += 1;
|
|
365
|
+
byteCount += rowBytes;
|
|
366
|
+
totalRows += 1;
|
|
367
|
+
totalBytes += rowBytes;
|
|
290
368
|
}
|
|
291
|
-
const keptRows = hit.rows.slice(0, rowBudget);
|
|
292
|
-
const keptHashes = hit.hashes.slice(0, rowBudget);
|
|
293
|
-
rowCount += keptRows.length;
|
|
294
369
|
if (hit.totalMatchCount > hit.matchCount) limitTruncated = true;
|
|
295
|
-
if (keptRows.length < hit.rows.length) rowTruncated = true;
|
|
296
370
|
matches += hit.matchCount;
|
|
297
371
|
hits.push({ ...hit, rows: keptRows, hashes: keptHashes });
|
|
372
|
+
if (rowTruncated) countOnly = true;
|
|
298
373
|
}
|
|
374
|
+
hits.sort((a, b) => cmp(a.displayPath, b.displayPath));
|
|
299
375
|
for (const hit of hits) {
|
|
300
376
|
await recordServedSafe(hit.path, hit.hashes, "grep", new Set(hit.fileHashes));
|
|
301
377
|
}
|
|
@@ -303,14 +379,32 @@ export function regGrep(pi: ExtensionAPI): void {
|
|
|
303
379
|
.map((hit) => `=== ${hit.displayPath} ===\n${hit.rows.join("\n")}`)
|
|
304
380
|
.join("\n");
|
|
305
381
|
const notes: string[] = [];
|
|
306
|
-
if (rowTruncated) notes.push(`[grep: output truncated at ${
|
|
382
|
+
if (rowTruncated) notes.push(`[grep: output truncated at ${DEFAULT_MAX_LINES} rows or ${formatSize(DEFAULT_MAX_BYTES)}; refine the pattern to see more.]`);
|
|
307
383
|
if (limitTruncated) notes.push(`[grep: showing first ${limit} matches; increase limit to see more.]`);
|
|
308
384
|
if (state.stopped) notes.push(`[grep: scan cap of ${MAX_SCAN_FILES} files reached; results may be incomplete.]`);
|
|
385
|
+
if (linesReplaced > 0) notes.push(`[grep: ${linesReplaced} line(s) exceed ${formatSize(MAX_GREP_LINE_BYTES)} and are shown as truncated fragments; use read to see the full lines.]`);
|
|
309
386
|
const truncated = limitTruncated || rowTruncated;
|
|
387
|
+
const truncation: TruncationResult | undefined = rowTruncated
|
|
388
|
+
? {
|
|
389
|
+
content: blocks,
|
|
390
|
+
truncated: true,
|
|
391
|
+
truncatedBy,
|
|
392
|
+
totalLines: totalRows,
|
|
393
|
+
totalBytes,
|
|
394
|
+
outputLines: rowCount,
|
|
395
|
+
outputBytes: byteCount,
|
|
396
|
+
lastLinePartial: false,
|
|
397
|
+
firstLineExceedsLimit: false,
|
|
398
|
+
maxLines: DEFAULT_MAX_LINES,
|
|
399
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
400
|
+
}
|
|
401
|
+
: undefined;
|
|
310
402
|
const text = blocks.length > 0 ? `${blocks}${notes.length > 0 ? `\n${notes.join("\n")}` : ""}` : "No matches found.";
|
|
311
403
|
return {
|
|
312
404
|
content: [{ type: "text", text }],
|
|
313
405
|
details: {
|
|
406
|
+
...(truncation ? { truncation } : {}),
|
|
407
|
+
...(linesReplaced > 0 ? { linesTruncated: true as const } : {}),
|
|
314
408
|
metrics: {
|
|
315
409
|
matches,
|
|
316
410
|
files: hits.length,
|
package/src/hash-store.ts
CHANGED
|
@@ -6,6 +6,8 @@ import { initHasher, contentChecksum } from "./hashline/hasher";
|
|
|
6
6
|
import { HASH_RE } from "./hashline/alphabet";
|
|
7
7
|
import { HASH_STORE_VERSION, HASH_STORE_BUSY_TIMEOUT } from "./constants";
|
|
8
8
|
|
|
9
|
+
export const STORE_NOT_OPEN_MESSAGE = "Hash store is not open; transactional update aborted";
|
|
10
|
+
|
|
9
11
|
type SqlParams = (string | number)[];
|
|
10
12
|
|
|
11
13
|
interface RawStatement {
|
|
@@ -104,21 +106,22 @@ export function isValidHashList(value: unknown): value is string[] {
|
|
|
104
106
|
return true;
|
|
105
107
|
}
|
|
106
108
|
|
|
107
|
-
export function parseHashList(raw: string, onInvalid: () => void): string[] | undefined {
|
|
109
|
+
export function parseHashList(raw: string, onInvalid: () => void, context?: string): string[] | undefined {
|
|
108
110
|
let parsed: unknown;
|
|
109
111
|
try {
|
|
110
112
|
parsed = JSON.parse(raw);
|
|
111
|
-
} catch {
|
|
113
|
+
} catch (error) {
|
|
114
|
+
console.error(`[parseHashList]${context ? ` ${context}:` : ""} failed to parse stored hashes JSON:`, error);
|
|
112
115
|
onInvalid();
|
|
113
116
|
return undefined;
|
|
114
117
|
}
|
|
115
118
|
if (!isValidHashList(parsed)) {
|
|
119
|
+
console.error(`[parseHashList]${context ? ` ${context}:` : ""} stored hashes did not pass validation:`, Array.isArray(parsed) ? `length=${parsed.length} sample=${JSON.stringify(parsed.slice(0, 3))}` : (() => { try { return JSON.stringify(parsed)?.slice(0, 500) ?? String(parsed).slice(0, 500); } catch { return String(parsed).slice(0, 500); } })());
|
|
116
120
|
onInvalid();
|
|
117
121
|
return undefined;
|
|
118
122
|
}
|
|
119
123
|
return parsed;
|
|
120
124
|
}
|
|
121
|
-
|
|
122
125
|
export function parseStoredHashes(
|
|
123
126
|
row: Record<string, unknown> | undefined,
|
|
124
127
|
onInvalid: () => void,
|
|
@@ -157,13 +160,14 @@ function isBusyError(error: unknown): boolean {
|
|
|
157
160
|
return error instanceof Error && /busy|locked/i.test(error.message);
|
|
158
161
|
}
|
|
159
162
|
|
|
163
|
+
const sleepSab = new Int32Array(new SharedArrayBuffer(4));
|
|
164
|
+
|
|
160
165
|
function sleepSync(ms: number): void {
|
|
161
|
-
|
|
162
|
-
Atomics.wait(sab, 0, 0, ms);
|
|
166
|
+
Atomics.wait(sleepSab, 0, 0, ms);
|
|
163
167
|
}
|
|
164
168
|
|
|
165
169
|
const BUSY_RETRIES = 3;
|
|
166
|
-
const BUSY_RETRY_DELAY_MS =
|
|
170
|
+
const BUSY_RETRY_DELAY_MS = 50;
|
|
167
171
|
|
|
168
172
|
function withBusyRetry<T>(fn: () => T): T {
|
|
169
173
|
let lastError: unknown;
|
|
@@ -173,14 +177,28 @@ function withBusyRetry<T>(fn: () => T): T {
|
|
|
173
177
|
} catch (error) {
|
|
174
178
|
lastError = error;
|
|
175
179
|
if (!isBusyError(error) || attempt === BUSY_RETRIES) throw error;
|
|
176
|
-
sleepSync(BUSY_RETRY_DELAY_MS);
|
|
180
|
+
sleepSync(BUSY_RETRY_DELAY_MS * (1 << attempt));
|
|
177
181
|
}
|
|
178
182
|
}
|
|
179
183
|
throw lastError;
|
|
180
184
|
}
|
|
181
185
|
|
|
182
|
-
function
|
|
183
|
-
|
|
186
|
+
async function withBusyRetryAsync<T>(fn: () => T): Promise<T> {
|
|
187
|
+
let lastError: unknown;
|
|
188
|
+
for (let attempt = 0; attempt <= BUSY_RETRIES; attempt++) {
|
|
189
|
+
try {
|
|
190
|
+
return fn();
|
|
191
|
+
} catch (error) {
|
|
192
|
+
lastError = error;
|
|
193
|
+
if (!isBusyError(error) || attempt === BUSY_RETRIES) throw error;
|
|
194
|
+
await new Promise<void>((r) => setTimeout(r, BUSY_RETRY_DELAY_MS * (1 << attempt)));
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
throw lastError;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
async function openDbWithBusyRetryAsync(storePath: string): Promise<{ db: RawDb; stmts: Prepared }> {
|
|
201
|
+
return withBusyRetryAsync(() => openDb(storePath));
|
|
184
202
|
}
|
|
185
203
|
|
|
186
204
|
function retriedWrite(
|
|
@@ -341,19 +359,19 @@ async function openStore(storePath: string): Promise<HashStore> {
|
|
|
341
359
|
let existed = existsSync(storePath);
|
|
342
360
|
let opened: { db: RawDb; stmts: Prepared };
|
|
343
361
|
try {
|
|
344
|
-
opened =
|
|
362
|
+
opened = await openDbWithBusyRetryAsync(storePath);
|
|
345
363
|
} catch (error) {
|
|
346
364
|
if (!isCorruptionError(error)) throw error;
|
|
347
365
|
console.error("Hash store failed to open, rebuilding:", error);
|
|
348
366
|
await quarantineStore(storePath);
|
|
349
367
|
existed = false;
|
|
350
|
-
opened =
|
|
368
|
+
opened = await openDbWithBusyRetryAsync(storePath);
|
|
351
369
|
}
|
|
352
370
|
if (!isHealthy(opened.db)) {
|
|
353
371
|
shutdownDb(opened.db);
|
|
354
372
|
await quarantineStore(storePath);
|
|
355
373
|
existed = false;
|
|
356
|
-
opened =
|
|
374
|
+
opened = await openDbWithBusyRetryAsync(storePath);
|
|
357
375
|
}
|
|
358
376
|
const { db, stmts } = opened;
|
|
359
377
|
|
|
@@ -403,9 +421,9 @@ export function shutdownHashStore(): void {
|
|
|
403
421
|
snapshotCache.clear();
|
|
404
422
|
}
|
|
405
423
|
|
|
406
|
-
function withStore(fn: () => void): void {
|
|
424
|
+
export function withStore(fn: () => void): void {
|
|
407
425
|
if (!cachedDb) {
|
|
408
|
-
throw new Error(
|
|
426
|
+
throw new Error(STORE_NOT_OPEN_MESSAGE);
|
|
409
427
|
}
|
|
410
428
|
withBusyRetry(() => {
|
|
411
429
|
cachedDb!.db.exec("BEGIN IMMEDIATE");
|
|
@@ -528,6 +546,14 @@ export function upsertSnapshot(
|
|
|
528
546
|
store.stmts.upsert(path, checksum, lineCount, JSON.stringify(hashes), Date.now());
|
|
529
547
|
cacheSnapshot(path, checksum, lineCount, hashes);
|
|
530
548
|
}
|
|
549
|
+
export function persistSnapshot(
|
|
550
|
+
store: HashStore,
|
|
551
|
+
path: string,
|
|
552
|
+
content: string,
|
|
553
|
+
hashes: string[],
|
|
554
|
+
): void {
|
|
555
|
+
upsertSnapshot(store, path, contentChecksum(content), splitLines(content).length, hashes);
|
|
556
|
+
}
|
|
531
557
|
|
|
532
558
|
export function upsertUndo(store: HashStore, path: string, entry: UndoRecord): void {
|
|
533
559
|
store.stmts.undoUpsert(
|
|
@@ -570,7 +596,11 @@ async function statMissing(rows: { path: string }[]): Promise<string[]> {
|
|
|
570
596
|
try {
|
|
571
597
|
await stat(row.path);
|
|
572
598
|
return undefined;
|
|
573
|
-
} catch {
|
|
599
|
+
} catch (error: unknown) {
|
|
600
|
+
if (errCode(error) !== "ENOENT") {
|
|
601
|
+
console.error("Failed to stat hash store path:", row.path, error);
|
|
602
|
+
return undefined;
|
|
603
|
+
}
|
|
574
604
|
return row.path;
|
|
575
605
|
}
|
|
576
606
|
}),
|
|
@@ -589,22 +619,32 @@ export async function pruneMissing(store: HashStore): Promise<void> {
|
|
|
589
619
|
withStore(() => {
|
|
590
620
|
for (const path of missing) {
|
|
591
621
|
store.stmts.deleteOne(path);
|
|
592
|
-
snapshotCache.delete(path);
|
|
593
622
|
store.stmts.servedDelete(path);
|
|
594
623
|
}
|
|
595
624
|
});
|
|
625
|
+
for (const path of missing) snapshotCache.delete(path);
|
|
596
626
|
}
|
|
597
627
|
|
|
598
628
|
function matchPathsByHashes(
|
|
599
629
|
rows: { path: string; hashes: string }[],
|
|
600
630
|
hashes: string[],
|
|
601
631
|
): string[] {
|
|
632
|
+
const needed = new Set(hashes);
|
|
633
|
+
if (needed.size === 0) return [];
|
|
602
634
|
const matches: string[] = [];
|
|
603
635
|
for (const row of rows) {
|
|
604
636
|
try {
|
|
605
637
|
const parsed = JSON.parse(row.hashes) as unknown;
|
|
606
638
|
if (!isValidHashList(parsed)) continue;
|
|
607
|
-
|
|
639
|
+
const parsedSet = new Set(parsed);
|
|
640
|
+
let ok = true;
|
|
641
|
+
for (const h of needed) {
|
|
642
|
+
if (!parsedSet.has(h)) {
|
|
643
|
+
ok = false;
|
|
644
|
+
break;
|
|
645
|
+
}
|
|
646
|
+
}
|
|
647
|
+
if (ok) matches.push(row.path);
|
|
608
648
|
} catch {
|
|
609
649
|
continue;
|
|
610
650
|
}
|
package/src/hashline/hash.ts
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
|
-
import { splitLines } from "../utils";
|
|
1
|
+
import { splitLines, truncateToBytes, getCached } from "../utils";
|
|
2
|
+
import { MAX_HASH_SOURCE_BYTES } from "../constants";
|
|
2
3
|
import {
|
|
3
4
|
loadHashStore,
|
|
4
5
|
type HashStore,
|
|
5
6
|
getSnapshot,
|
|
6
|
-
|
|
7
|
+
persistSnapshot,
|
|
7
8
|
} from "../hash-store";
|
|
8
|
-
import { xxh32,
|
|
9
|
+
import { xxh32, initHasher } from "./hasher";
|
|
9
10
|
import { HASH_LEN, ALPH, ALPH_RE, HASH_CLASS, HASH_RUN } from "./alphabet";
|
|
10
11
|
export { initHasher, HASH_LEN, ALPH_RE, HASH_CLASS, HASH_RUN };
|
|
11
12
|
|
|
@@ -75,6 +76,10 @@ export function canon(line: string): string {
|
|
|
75
76
|
return line.replace(/\r/g, "").trimEnd();
|
|
76
77
|
}
|
|
77
78
|
|
|
79
|
+
export function hashSource(line: string): string {
|
|
80
|
+
return truncateToBytes(canon(line), MAX_HASH_SOURCE_BYTES);
|
|
81
|
+
}
|
|
82
|
+
|
|
78
83
|
const BITSET_WORDS = Math.ceil(HASH_SPACE / 32);
|
|
79
84
|
|
|
80
85
|
function getBit(bits: Uint32Array, idx: number): boolean {
|
|
@@ -115,9 +120,10 @@ export function _lineHashesPure(content: string): string[] {
|
|
|
115
120
|
const hashes = new Array<string>(lines.length);
|
|
116
121
|
const used = new Uint32Array(BITSET_WORDS);
|
|
117
122
|
const hint = { value: 0 };
|
|
123
|
+
const hashSourceCache = new Map<string, string>();
|
|
118
124
|
|
|
119
125
|
for (let i = 0; i < lines.length; i++) {
|
|
120
|
-
const c =
|
|
126
|
+
const c = getCached(hashSourceCache, lines[i]!, hashSource);
|
|
121
127
|
const baseIdx = (xxh32(c) >>> 14) % HASH_SPACE;
|
|
122
128
|
hashes[i] = assignHash(used, baseIdx, hint);
|
|
123
129
|
}
|
|
@@ -146,7 +152,7 @@ export async function lineHashes(
|
|
|
146
152
|
);
|
|
147
153
|
if (persist !== false) {
|
|
148
154
|
try {
|
|
149
|
-
|
|
155
|
+
persistSnapshot(hashStore, path, content, newHashes);
|
|
150
156
|
} catch (error) {
|
|
151
157
|
console.error("Failed to persist hash snapshot:", error);
|
|
152
158
|
}
|
|
@@ -167,7 +173,7 @@ export async function lineHashes(
|
|
|
167
173
|
const newHashes = _lineHashesPure(content);
|
|
168
174
|
if (persist !== false) {
|
|
169
175
|
try {
|
|
170
|
-
|
|
176
|
+
persistSnapshot(hashStore, path, content, newHashes);
|
|
171
177
|
} catch (error) {
|
|
172
178
|
console.error("Failed to persist hash snapshot:", error);
|
|
173
179
|
}
|
|
@@ -219,6 +225,7 @@ function mapStableHashes(
|
|
|
219
225
|
const newHashes = new Array<string>(newLines.length);
|
|
220
226
|
const used = new Uint32Array(BITSET_WORDS);
|
|
221
227
|
const hint = { value: 0 };
|
|
228
|
+
const hashSourceCache = new Map<string, string>();
|
|
222
229
|
const removed = removedHashes ?? new Set<string>();
|
|
223
230
|
|
|
224
231
|
const oldHashIndex = new Map<string, number>();
|
|
@@ -255,7 +262,7 @@ function mapStableHashes(
|
|
|
255
262
|
|
|
256
263
|
const newByContent = new Map<string, number[]>();
|
|
257
264
|
for (let i = 0; i < newLines.length; i++) {
|
|
258
|
-
const key =
|
|
265
|
+
const key = getCached(hashSourceCache, newLines[i]!, hashSource);
|
|
259
266
|
const list = newByContent.get(key);
|
|
260
267
|
if (list) list.push(i);
|
|
261
268
|
else newByContent.set(key, [i]);
|
|
@@ -270,7 +277,7 @@ function mapStableHashes(
|
|
|
270
277
|
};
|
|
271
278
|
|
|
272
279
|
for (const entry of survivors) {
|
|
273
|
-
const candidates = newByContent.get(
|
|
280
|
+
const candidates = newByContent.get(getCached(hashSourceCache, oldLines[entry.index]!, hashSource));
|
|
274
281
|
if (!candidates || candidates.length === 0) continue;
|
|
275
282
|
const target = entry.index > spanEnd ? entry.index + shiftAfterSpan : entry.index;
|
|
276
283
|
const pos = nearestNew(candidates, target);
|
|
@@ -301,7 +308,7 @@ function mapStableHashes(
|
|
|
301
308
|
|
|
302
309
|
for (let i = 0; i < newLines.length; i++) {
|
|
303
310
|
if (newHashes[i]) continue;
|
|
304
|
-
const c =
|
|
311
|
+
const c = getCached(hashSourceCache, newLines[i]!, hashSource);
|
|
305
312
|
const baseIdx = (xxh32(c) >>> 14) % HASH_SPACE;
|
|
306
313
|
newHashes[i] = assignHash(used, baseIdx, hint);
|
|
307
314
|
}
|