pi-hashline-edit-pro 2.8.1 → 2.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/grep.ts CHANGED
@@ -1,20 +1,20 @@
1
1
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
2
2
  import { formatSize, DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, type TruncationResult } from "@earendil-works/pi-coding-agent";
3
3
  import { Type } from "typebox";
4
- import { readdir, stat } from "fs/promises";
4
+ import { stat } from "fs/promises";
5
5
  import { dirname, join, relative } from "path";
6
+ import { spawn, spawnSync } from "child_process";
7
+ import { createInterface } from "readline";
6
8
  import { tryReadNormFile } from "./file-reader";
7
9
  import { MAX_HASH_LINES, fmtRow, HASH_LEN, HASH_SEP } from "./hashline";
8
10
  import { MAX_GREP_LINE_BYTES } from "./constants";
9
11
  import { toCwd } from "./paths";
10
12
  import { loadP, loadGuide } from "./prompts";
11
13
  import { normReq } from "./payload-contract";
12
- import { recordServedSafe } from "./served";
14
+ import { recordServedSafe, buildServedMap } from "./served";
13
15
  import { abortIf, errCode, isRec, makePrepareArguments, rejectUnknownFields, truncateToBytes, visLines } from "./utils";
14
16
 
15
17
  const GREP_KS = new Set(["pattern", "path", "glob", "context", "ignoreCase", "literal", "limit"]);
16
- const SKIP_DIRS = new Set(["node_modules", ".git", ".tmp", "coverage"]);
17
- const MAX_SCAN_FILES = 4000;
18
18
 
19
19
  function cmp(a: string, b: string): number {
20
20
  return a < b ? -1 : a > b ? 1 : 0;
@@ -69,13 +69,11 @@ function unsafeRegex(pattern: string): never {
69
69
 
70
70
  function assertSafeRegex(pattern: string): void {
71
71
  if (pattern.length > 4096) unsafeRegex(pattern);
72
-
73
72
  const groups: RegexGroupRisk[] = [];
74
73
  let inClass = false;
75
74
  let escaped = false;
76
75
  let variableQuantifiers = 0;
77
76
  let lastAtom: { groupRisky: boolean; quantified: boolean } | undefined;
78
-
79
77
  for (let i = 0; i < pattern.length; i++) {
80
78
  const ch = pattern[i]!;
81
79
  if (escaped) {
@@ -122,13 +120,17 @@ function assertSafeRegex(pattern: string): void {
122
120
  lastAtom = undefined;
123
121
  continue;
124
122
  }
125
-
126
123
  let quantifierLength = 0;
127
124
  if (ch === "*" || ch === "+" || ch === "?") {
128
125
  quantifierLength = 1;
129
126
  } else if (ch === "{") {
130
127
  quantifierLength = /^\{\d+(?:,\d*)?\}/.exec(pattern.slice(i))?.[0].length ?? 0;
131
128
  }
129
+ if (ch === "{" && quantifierLength > 0) {
130
+ const quant = pattern.slice(i, i + quantifierLength);
131
+ const m = /^\{(\d+)/.exec(quant);
132
+ if (m && Number(m[1]) > 1000) unsafeRegex(pattern);
133
+ }
132
134
  if (quantifierLength > 0 && lastAtom) {
133
135
  if (ch === "?" && lastAtom.quantified) continue;
134
136
  const variable = ch !== "{" || pattern.slice(i, i + quantifierLength).includes(",");
@@ -140,7 +142,6 @@ function assertSafeRegex(pattern: string): void {
140
142
  i += quantifierLength - 1;
141
143
  continue;
142
144
  }
143
-
144
145
  lastAtom = { groupRisky: false, quantified: false };
145
146
  }
146
147
  }
@@ -177,8 +178,10 @@ interface FileHit {
177
178
  path: string;
178
179
  displayPath: string;
179
180
  fileHashes: string[];
181
+ fileLines: string[];
180
182
  rows: string[];
181
183
  hashes: string[];
184
+ lineNumbers: number[];
182
185
  matchCount: number;
183
186
  totalMatchCount: number;
184
187
  fragmented: boolean[];
@@ -219,96 +222,41 @@ function grepHeadFragment(line: string): string {
219
222
  return head.length < line.length ? `${head}...` : head;
220
223
  }
221
224
 
222
- interface ScanState {
223
- scanned: number;
224
- stopped: boolean;
225
- }
226
-
227
- async function walkFiles(
228
- root: string,
229
- state: ScanState,
230
- onFile: (absPath: string) => Promise<void>,
231
- signal?: AbortSignal,
232
- ): Promise<void> {
233
- const queue: string[] = [root];
234
- let head = 0;
235
- while (head < queue.length && !state.stopped) {
236
- abortIf(signal);
237
- const dir = queue[head++]!;
238
- let entries;
239
- try {
240
- entries = await readdir(dir, { withFileTypes: true });
241
- } catch {
242
- continue;
243
- }
244
- entries.sort((a, b) => cmp(a.name, b.name));
245
- for (let ei = 0; ei < entries.length; ei++) {
246
- if ((ei & 127) === 0) abortIf(signal);
247
- if (state.stopped) break;
248
- const entry = entries[ei]!;
249
- const full = join(dir, entry.name);
250
- if (entry.isDirectory()) {
251
- if (SKIP_DIRS.has(entry.name)) continue;
252
- queue.push(full);
253
- } else if (entry.isFile()) {
254
- state.scanned += 1;
255
- if (state.scanned > MAX_SCAN_FILES) {
256
- state.stopped = true;
257
- break;
258
- }
259
- await onFile(full);
260
- }
261
- }
262
- }
263
- }
264
-
265
- async function searchFile(
266
- absPath: string,
267
- globRoot: string,
268
- cwd: string,
269
- regex: RegExp,
270
- globRegex: RegExp | undefined,
225
+ function makeHitFromIndices(
226
+ norm: { normalized: string; fileHashes: string[]; absolutePath: string },
227
+ displayPath: string,
228
+ matchIndices: number[],
271
229
  context: number,
272
- maxMatches: number,
273
- signal?: AbortSignal,
274
- ): Promise<FileHit | undefined> {
275
- const displayPath = relative(cwd, absPath).replace(/\\/g, "/");
276
- if (globRegex) {
277
- const globPath = relative(globRoot, absPath).replace(/\\/g, "/");
278
- if (!globRegex.test(globPath) && !globRegex.test(displayPath)) return undefined;
279
- }
280
- const norm = await tryReadNormFile(absPath, cwd, { maxLines: MAX_HASH_LINES, noPersist: true, signal });
281
- if (!norm) return undefined;
230
+ regex: RegExp | undefined,
231
+ totalMatchCount: number,
232
+ keptMatchCount: number,
233
+ ): FileHit {
282
234
  const lines = visLines(norm.normalized);
283
- const matchLines: number[] = [];
284
- for (let i = 0; i < lines.length; i++) {
285
- if ((i & 1023) === 0) abortIf(signal);
286
- if (i !== 0 && (i & 4095) === 0) await new Promise<void>((r) => setImmediate(r));
287
- if (regex.test(lines[i]!)) matchLines.push(i);
288
- }
289
- if (matchLines.length === 0) return undefined;
290
- const keptMatches = matchLines.length > maxMatches ? matchLines.slice(0, maxMatches) : matchLines;
291
235
  const shown = new Set<number>();
292
- for (const i of keptMatches) {
236
+ const kept = matchIndices.slice(0, keptMatchCount);
237
+ for (const i of kept) {
293
238
  for (let j = Math.max(0, i - context); j <= Math.min(lines.length - 1, i + context); j++) shown.add(j);
294
239
  }
295
240
  const sorted = [...shown].sort((a, b) => a - b);
296
- const matchSet = new Set(matchLines);
241
+ const matchSet = new Set(matchIndices);
297
242
  const rows: string[] = [];
298
243
  const hashes: string[] = [];
244
+ const lineNumbers: number[] = [];
299
245
  const fragmented: boolean[] = [];
300
246
  for (const idx of sorted) {
301
247
  const hash = norm.fileHashes[idx]!;
302
248
  const line = lines[idx]!;
303
249
  const row = fmtRow(hash, line);
304
250
  if (Buffer.byteLength(row, "utf-8") > MAX_GREP_LINE_BYTES) {
305
- const content = matchSet.has(idx) ? grepMatchFragment(line, regex) : grepHeadFragment(line);
251
+ const content = matchSet.has(idx) && regex ? grepMatchFragment(line, regex) : grepHeadFragment(line);
306
252
  rows.push(fmtRow(hash, content));
307
253
  hashes.push(hash);
254
+ lineNumbers.push(idx + 1);
308
255
  fragmented.push(true);
309
256
  } else {
310
257
  rows.push(row);
311
258
  hashes.push(hash);
259
+ lineNumbers.push(idx + 1);
312
260
  fragmented.push(false);
313
261
  }
314
262
  }
@@ -316,14 +264,142 @@ async function searchFile(
316
264
  path: norm.absolutePath,
317
265
  displayPath,
318
266
  fileHashes: norm.fileHashes,
267
+ fileLines: lines,
319
268
  rows,
320
269
  hashes,
321
- matchCount: keptMatches.length,
322
- totalMatchCount: matchLines.length,
270
+ lineNumbers,
271
+ matchCount: kept.length,
272
+ totalMatchCount,
323
273
  fragmented,
324
274
  };
325
275
  }
326
276
 
277
+ async function resolveRgPath(): Promise<string> {
278
+ try {
279
+ const r = spawnSync("rg", ["--version"], { stdio: "pipe" });
280
+ if (!r.error && r.status === 0) return "rg";
281
+ } catch {}
282
+ try {
283
+ const { homedir } = await import("os");
284
+ const { existsSync } = await import("fs");
285
+ const home = process.env.HOME ?? homedir();
286
+ const base = process.env.PI_CODING_AGENT_DIR ?? join(home, ".pi", "agent");
287
+ const bin = join(base, "bin", process.platform === "win32" ? "rg.exe" : "rg");
288
+ if (existsSync(bin)) {
289
+ const r = spawnSync(bin, ["--version"], { stdio: "pipe" });
290
+ if (!r.error && r.status === 0) return bin;
291
+ }
292
+ } catch {}
293
+ try {
294
+ const { createRequire } = await import("module");
295
+ const require = createRequire(import.meta.url);
296
+ const pkgPath = require.resolve("@earendil-works/pi-coding-agent/package.json");
297
+ const { dirname } = await import("path");
298
+ const piDir = dirname(pkgPath);
299
+ const toolsManagerPath = join(piDir, "dist/utils/tools-manager.js");
300
+ const mod = await import("file://" + toolsManagerPath);
301
+ if (mod.ensureTool) {
302
+ const p = await mod.ensureTool("rg", true);
303
+ if (p) return p;
304
+ }
305
+ } catch {}
306
+ throw new Error("[E_ACCESS] ripgrep (rg) is required for grep but was not found. Install ripgrep or ensure pi can download it to ~/.pi/agent/bin.");
307
+ }
308
+
309
+ async function collectRgMatches(
310
+ rgPath: string,
311
+ pattern: string,
312
+ searchPath: string,
313
+ req: GrepReq,
314
+ signal?: AbortSignal,
315
+ ): Promise<Map<string, number[]>> {
316
+ const args = ["--json", "--line-number", "--color=never", "--hidden", "--glob", "!.git"];
317
+ if (req.ignoreCase) args.push("--ignore-case");
318
+ if (req.literal) args.push("--fixed-strings");
319
+ args.push("--", pattern, searchPath);
320
+ const result = new Map<string, number[]>();
321
+ return await new Promise<Map<string, number[]>>((resolve, reject) => {
322
+ const child = spawn(rgPath, args, { stdio: ["ignore", "pipe", "pipe"] });
323
+ const rl = createInterface({ input: child.stdout });
324
+ let stderr = "";
325
+ let timedOut = false;
326
+ const rgTimeout = setTimeout(() => {
327
+ timedOut = true;
328
+ if (!child.killed) child.kill("SIGKILL");
329
+ reject(new Error("rg timeout"));
330
+ }, 10000);
331
+ child.stderr?.on("data", (chunk) => {
332
+ stderr += chunk.toString();
333
+ });
334
+ const onAbort = () => {
335
+ if (!child.killed) child.kill("SIGKILL");
336
+ };
337
+ signal?.addEventListener("abort", onAbort, { once: true });
338
+ const cleanup = () => {
339
+ clearTimeout(rgTimeout);
340
+ rl.close();
341
+ signal?.removeEventListener("abort", onAbort);
342
+ };
343
+ rl.on("line", (line) => {
344
+ if (!line.trim()) return;
345
+ let event: { type?: string; data?: { path?: { text?: string }; line_number?: number } };
346
+ try {
347
+ event = JSON.parse(line);
348
+ } catch {
349
+ return;
350
+ }
351
+ if (event.type === "match") {
352
+ const filePath = event.data?.path?.text;
353
+ const lineNumber = event.data?.line_number;
354
+ if (typeof filePath === "string" && typeof lineNumber === "number") {
355
+ let abs: string;
356
+ try {
357
+ abs = filePath.startsWith("/") || /^[A-Za-z]:\\/.test(filePath) ? filePath : join(searchPath, filePath);
358
+ } catch {
359
+ abs = filePath;
360
+ }
361
+ const list = result.get(abs) ?? [];
362
+ list.push(lineNumber);
363
+ result.set(abs, list);
364
+ }
365
+ }
366
+ });
367
+ child.on("error", (error) => {
368
+ cleanup();
369
+ reject(error);
370
+ });
371
+ child.on("close", (code) => {
372
+ cleanup();
373
+ if (timedOut) return;
374
+ if (signal?.aborted) {
375
+ reject(new Error("Operation aborted"));
376
+ return;
377
+ }
378
+ if (code !== 0 && code !== 1) {
379
+ const msg = stderr.trim() || `ripgrep exited with code ${code}`;
380
+ reject(new Error(msg));
381
+ return;
382
+ }
383
+ resolve(result);
384
+ });
385
+ });
386
+ }
387
+
388
+ function gutterWidthFor(numbers: number[]): number {
389
+ let max = 0;
390
+ for (const n of numbers) if (n > max) max = n;
391
+ return String(max || 1).length;
392
+ }
393
+
394
+ function displayRowsForHit(hit: FileHit): string[] {
395
+ const width = gutterWidthFor(hit.lineNumbers);
396
+ return hit.rows.map((row, i) => {
397
+ const n = hit.lineNumbers[i]!;
398
+ const padded = String(n).padStart(width, " ");
399
+ return `${padded} │ ${row}`;
400
+ });
401
+ }
402
+
327
403
  const grepToolSchema = Type.Object(
328
404
  {
329
405
  pattern: Type.String({
@@ -367,8 +443,8 @@ const grepToolSchema = Type.Object(
367
443
 
368
444
  export function regGrep(pi: ExtensionAPI): void {
369
445
  pi.registerTool({
370
- name: "grep",
371
- label: "Grep",
446
+ name: "anchor_grep",
447
+ label: "Anchor Grep",
372
448
  description: loadP("../prompts/grep.md"),
373
449
  promptSnippet: loadP("../prompts/grep-snippet.md"),
374
450
  promptGuidelines: loadGuide("../prompts/grep-guidelines.md"),
@@ -380,10 +456,8 @@ export function regGrep(pi: ExtensionAPI): void {
380
456
  const canonical = normReq(params);
381
457
  assertGrepReq(canonical);
382
458
  const req = canonical;
383
- const regex = buildRegex(req.pattern, req.literal === true, req.ignoreCase === true);
384
459
  const context = req.context ?? 0;
385
460
  const limit = req.limit ?? 100;
386
- const globRegex = req.glob === undefined ? undefined : globToRegex(req.glob);
387
461
  const base = req.path ? toCwd(req.path, ctx.cwd) : ctx.cwd;
388
462
  abortIf(signal);
389
463
  let baseStat;
@@ -396,16 +470,9 @@ export function regGrep(pi: ExtensionAPI): void {
396
470
  throw new Error(`[E_ACCESS] Cannot access path: ${req.path ?? ctx.cwd}`);
397
471
  }
398
472
  const globRoot = baseStat.isFile() ? dirname(base) : base;
399
- const state: ScanState = { scanned: 0, stopped: false };
400
- const files: string[] = [];
401
- if (baseStat.isFile()) {
402
- files.push(base);
403
- } else {
404
- await walkFiles(base, state, async (absPath) => {
405
- files.push(absPath);
406
- }, signal);
407
- files.sort(cmp);
408
- }
473
+ const globRegex = req.glob === undefined ? undefined : globToRegex(req.glob);
474
+ const validatedRegex = buildRegex(req.pattern, req.literal === true, req.ignoreCase === true);
475
+ const rgPath = await resolveRgPath();
409
476
  const hits: FileHit[] = [];
410
477
  let matches = 0;
411
478
  let limitTruncated = false;
@@ -417,17 +484,26 @@ export function regGrep(pi: ExtensionAPI): void {
417
484
  let truncatedBy: "lines" | "bytes" | null = null;
418
485
  let linesReplaced = 0;
419
486
  let countOnly = false;
420
- for (let f = 0; f < files.length; f++) {
487
+ const rgMatches = await collectRgMatches(rgPath, req.pattern, base, req, signal);
488
+ const sortedFiles = [...rgMatches.keys()].sort(cmp);
489
+ for (let f = 0; f < sortedFiles.length; f++) {
421
490
  abortIf(signal);
422
- const absPath = files[f]!;
491
+ const absPath = sortedFiles[f]!;
492
+ const allNums = rgMatches.get(absPath) ?? [];
493
+ const totalForFile = allNums.length;
494
+ const sortedNums = [...allNums].sort((a, b) => a - b);
495
+ const indices = sortedNums.map((n) => n - 1).filter((n) => n >= 0);
423
496
  if (countOnly) {
424
- const hit = await searchFile(absPath, globRoot, ctx.cwd, regex, globRegex, context, Number.MAX_SAFE_INTEGER, signal);
425
- if (!hit) continue;
426
- totalRows += hit.rows.length;
427
- for (const row of hit.rows) totalBytes += Buffer.byteLength(row, "utf-8") + 1;
497
+ const norm = await tryReadNormFile(absPath, ctx.cwd, { maxLines: MAX_HASH_LINES, noPersist: true, signal });
498
+ if (!norm) continue;
499
+ const hit = makeHitFromIndices(norm, relative(ctx.cwd, absPath).replace(/\\/g, "/"), indices, context, validatedRegex, totalForFile, indices.length);
500
+ const display = displayRowsForHit(hit);
501
+ totalRows += display.length;
502
+ for (const r of display) totalBytes += Buffer.byteLength(r, "utf-8") + 1;
428
503
  const remaining = limit - matches;
429
504
  if (remaining > 0) {
430
- matches += Math.min(hit.matchCount, remaining);
505
+ const add = Math.min(hit.matchCount, remaining);
506
+ matches += add;
431
507
  if (hit.matchCount > remaining) limitTruncated = true;
432
508
  } else {
433
509
  limitTruncated = true;
@@ -439,24 +515,36 @@ export function regGrep(pi: ExtensionAPI): void {
439
515
  limitTruncated = true;
440
516
  break;
441
517
  }
442
- const hit = await searchFile(absPath, globRoot, ctx.cwd, regex, globRegex, context, remaining, signal);
518
+ const norm = await tryReadNormFile(absPath, ctx.cwd, { maxLines: MAX_HASH_LINES, noPersist: true, signal });
519
+ if (!norm) continue;
520
+ if (globRegex) {
521
+ const displayPath = relative(ctx.cwd, absPath).replace(/\\/g, "/");
522
+ const globPath = relative(globRoot, absPath).replace(/\\/g, "/");
523
+ if (!globRegex.test(globPath) && !globRegex.test(displayPath)) continue;
524
+ }
525
+ const hit = makeHitFromIndices(norm, relative(ctx.cwd, absPath).replace(/\\/g, "/"), indices, context, validatedRegex, totalForFile, Math.min(totalForFile, remaining));
443
526
  if (!hit) continue;
527
+ const display = displayRowsForHit(hit);
444
528
  const keptRows: string[] = [];
445
529
  const keptHashes: string[] = [];
446
- for (let i = 0; i < hit.rows.length; i++) {
447
- const row = hit.rows[i]!;
530
+ const keptLineNumbers: number[] = [];
531
+ const keptFragmented: boolean[] = [];
532
+ for (let i = 0; i < display.length; i++) {
533
+ const row = display[i]!;
448
534
  const rowBytes = Buffer.byteLength(row, "utf-8") + 1;
449
535
  if (rowCount >= DEFAULT_MAX_LINES || byteCount + rowBytes > DEFAULT_MAX_BYTES) {
450
536
  rowTruncated = true;
451
537
  if (truncatedBy === null) truncatedBy = byteCount + rowBytes > DEFAULT_MAX_BYTES ? "bytes" : "lines";
452
- for (let j = i; j < hit.rows.length; j++) {
538
+ for (let j = i; j < display.length; j++) {
453
539
  totalRows += 1;
454
- totalBytes += Buffer.byteLength(hit.rows[j]!, "utf-8") + 1;
540
+ totalBytes += Buffer.byteLength(display[j]!, "utf-8") + 1;
455
541
  }
456
542
  break;
457
543
  }
458
544
  keptRows.push(row);
459
- keptHashes.push(hit.hashes[i]);
545
+ keptHashes.push(hit.hashes[i]!);
546
+ keptLineNumbers.push(hit.lineNumbers[i]!);
547
+ keptFragmented.push(hit.fragmented[i]!);
460
548
  if (hit.fragmented[i]) linesReplaced += 1;
461
549
  rowCount += 1;
462
550
  byteCount += rowBytes;
@@ -465,12 +553,14 @@ export function regGrep(pi: ExtensionAPI): void {
465
553
  }
466
554
  if (hit.totalMatchCount > hit.matchCount) limitTruncated = true;
467
555
  matches += hit.matchCount;
468
- hits.push({ ...hit, rows: keptRows, hashes: keptHashes });
556
+ const displayHit: FileHit = { ...hit, rows: keptRows, hashes: keptHashes, lineNumbers: keptLineNumbers, fragmented: keptFragmented };
557
+ hits.push(displayHit);
469
558
  if (rowTruncated) countOnly = true;
470
559
  }
471
560
  hits.sort((a, b) => cmp(a.displayPath, b.displayPath));
472
561
  for (const hit of hits) {
473
- await recordServedSafe(hit.path, hit.hashes, "grep", new Set(hit.fileHashes));
562
+ const servedMap = buildServedMap(hit.fileHashes, hit.fileLines, hit.hashes);
563
+ await recordServedSafe(hit.path, servedMap, "anchor_grep", new Set(hit.fileHashes));
474
564
  }
475
565
  const blocks = hits
476
566
  .map((hit) => `=== ${hit.displayPath} ===\n${hit.rows.join("\n")}`)
@@ -478,7 +568,6 @@ export function regGrep(pi: ExtensionAPI): void {
478
568
  const notes: string[] = [];
479
569
  if (rowTruncated) notes.push(`[grep: output truncated at ${DEFAULT_MAX_LINES} rows or ${formatSize(DEFAULT_MAX_BYTES)}; refine the pattern to see more.]`);
480
570
  if (limitTruncated) notes.push(`[grep: showing first ${limit} matches; increase limit to see more.]`);
481
- if (state.stopped) notes.push(`[grep: scan cap of ${MAX_SCAN_FILES} files reached; results may be incomplete.]`);
482
571
  if (linesReplaced > 0) notes.push(`[grep: ${linesReplaced} line(s) exceed ${formatSize(MAX_GREP_LINE_BYTES)} and are shown as truncated fragments; use read to see the full lines.]`);
483
572
  const truncated = limitTruncated || rowTruncated;
484
573
  const truncation: TruncationResult | undefined = rowTruncated
@@ -505,7 +594,7 @@ export function regGrep(pi: ExtensionAPI): void {
505
594
  metrics: {
506
595
  matches,
507
596
  files: hits.length,
508
- truncated: truncated || state.stopped,
597
+ truncated,
509
598
  },
510
599
  },
511
600
  };
@@ -9,6 +9,15 @@ export function isValidHashList(value: unknown): value is string[] {
9
9
  return true;
10
10
  }
11
11
 
12
+ export function isValidServedMap(value: unknown): value is Record<string, string> {
13
+ if (typeof value !== "object" || value === null || Array.isArray(value)) return false;
14
+ for (const [k, v] of Object.entries(value as Record<string, unknown>)) {
15
+ if (typeof k !== "string" || !HASH_RE.test(k)) return false;
16
+ if (typeof v !== "string") return false;
17
+ }
18
+ return true;
19
+ }
20
+
12
21
  export function parseHashList(raw: string, onInvalid: () => void, context?: string): string[] | undefined {
13
22
  let parsed: unknown;
14
23
  try {
@@ -26,11 +35,33 @@ export function parseHashList(raw: string, onInvalid: () => void, context?: stri
26
35
  return parsed;
27
36
  }
28
37
 
38
+ export function parseServedMap(raw: string, onInvalid: () => void, context?: string): Map<string, string> | undefined {
39
+ let parsed: unknown;
40
+ try {
41
+ parsed = JSON.parse(raw);
42
+ } catch (error) {
43
+ console.error(`[parseServedMap]${context ? ` ${context}:` : ""} failed to parse stored served JSON:`, error);
44
+ onInvalid();
45
+ return undefined;
46
+ }
47
+ if (!isValidServedMap(parsed)) {
48
+ console.error(`[parseServedMap]${context ? ` ${context}:` : ""} stored served did not pass validation:`, (() => { try { return JSON.stringify(parsed)?.slice(0, 500) ?? String(parsed).slice(0, 500); } catch { return String(parsed).slice(0, 500); } })());
49
+ onInvalid();
50
+ return undefined;
51
+ }
52
+ return new Map(Object.entries(parsed as Record<string, string>));
53
+ }
54
+
29
55
  export function parseStoredHashes(row: Record<string, unknown> | undefined, onInvalid: () => void): string[] | undefined {
30
56
  if (!row) return undefined;
31
57
  return parseHashList(row.hashes as string, onInvalid);
32
58
  }
33
59
 
60
+ export function parseStoredServed(row: Record<string, unknown> | undefined, onInvalid: () => void): Map<string, string> | undefined {
61
+ if (!row) return undefined;
62
+ return parseServedMap(row.hashes as string, onInvalid);
63
+ }
64
+
34
65
  export function isValidSnapshot(value: unknown): value is { content: string; hashes: string[] } {
35
66
  if (typeof value !== "object" || value === null) return false;
36
67
  const v = value as Record<string, unknown>;
package/src/hash-store.ts CHANGED
@@ -6,10 +6,13 @@ import { initHasher, contentChecksum } from "./hashline/hasher";
6
6
  import { HASH_STORE_VERSION, HASH_STORE_BUSY_TIMEOUT } from "./constants";
7
7
  import {
8
8
  isValidHashList,
9
+ isValidServedMap,
9
10
  parseStoredHashes,
11
+ parseStoredServed,
10
12
  isValidSnapshot,
11
13
  isCorruptionError,
12
14
  parseHashList,
15
+ parseServedMap,
13
16
  } from "./hash-store/validation";
14
17
  import {
15
18
  withBusyRetry,
@@ -22,7 +25,7 @@ import {
22
25
  SNAPSHOT_CACHE_LIMIT,
23
26
  } from "./hash-store/cache";
24
27
 
25
- export { isValidHashList, parseHashList, parseStoredHashes, isCorruptionError };
28
+ export { isValidHashList, isValidServedMap, parseHashList, parseServedMap, parseStoredHashes, parseStoredServed, isCorruptionError };
26
29
  export { SNAPSHOT_CACHE_LIMIT };
27
30
  export const STORE_NOT_OPEN_MESSAGE = "Hash store is not open; transactional update aborted";
28
31
 
@@ -272,6 +275,20 @@ async function openStore(storePath: string): Promise<HashStore> {
272
275
  opened = await openDbWithBusyRetryAsync(() => openDb(storePath));
273
276
  }
274
277
  const { db, stmts } = opened;
278
+ try {
279
+ const autoVacuum = (db.prepare("PRAGMA auto_vacuum").get() as { auto_vacuum: number }).auto_vacuum;
280
+ const pageCount = (db.prepare("PRAGMA page_count").get() as { page_count: number }).page_count;
281
+ const freelist = (db.prepare("PRAGMA freelist_count").get() as { freelist_count: number }).freelist_count;
282
+ if (autoVacuum === 0 && !existed) {
283
+ db.exec("PRAGMA auto_vacuum=INCREMENTAL");
284
+ } else if (freelist > 50 && freelist * 5 > pageCount) {
285
+ try {
286
+ db.exec("PRAGMA incremental_vacuum(50)");
287
+ } catch {
288
+ db.exec("VACUUM");
289
+ }
290
+ }
291
+ } catch {}
275
292
 
276
293
  if (process.platform !== "win32") {
277
294
  for (const candidate of [storePath, `${storePath}-wal`, `${storePath}-shm`]) {
@@ -551,10 +568,37 @@ function matchPathsByHashes(
551
568
  return matches;
552
569
  }
553
570
 
571
+ function matchPathsByServed(
572
+ rows: { path: string; hashes: string }[],
573
+ hashes: string[],
574
+ ): string[] {
575
+ const needed = new Set(hashes);
576
+ if (needed.size === 0) return [];
577
+ const matches: string[] = [];
578
+ for (const row of rows) {
579
+ try {
580
+ const parsed = JSON.parse(row.hashes) as unknown;
581
+ if (!isValidServedMap(parsed)) continue;
582
+ const keySet = new Set(Object.keys(parsed as Record<string, unknown>));
583
+ let ok = true;
584
+ for (const h of needed) {
585
+ if (!keySet.has(h)) {
586
+ ok = false;
587
+ break;
588
+ }
589
+ }
590
+ if (ok) matches.push(row.path);
591
+ } catch {
592
+ continue;
593
+ }
594
+ }
595
+ return matches;
596
+ }
597
+
554
598
  export function findSnapshotPaths(store: HashStore, hashes: string[]): string[] {
555
599
  return matchPathsByHashes(store.stmts.allHashes() as { path: string; hashes: string }[], hashes);
556
600
  }
557
601
 
558
602
  export function findServedPaths(store: HashStore, hashes: string[]): string[] {
559
- return matchPathsByHashes(store.stmts.allServed() as { path: string; hashes: string }[], hashes);
603
+ return matchPathsByServed(store.stmts.allServed() as { path: string; hashes: string }[], hashes);
560
604
  }
@@ -154,7 +154,7 @@ export function applyEdit(
154
154
  signal?: AbortSignal,
155
155
  precomputedHashes?: string[],
156
156
  filePath?: string,
157
- servedHashes?: ReadonlySet<string>,
157
+ servedHashes?: ReadonlyMap<string, string>,
158
158
  skipBoundaryDedup?: boolean,
159
159
  ): {
160
160
  content: string;
@@ -190,7 +190,7 @@ export function applyEdit(
190
190
  fileHashes,
191
191
  filePath,
192
192
  );
193
- throw new AnchorMismatchError(feedback.text, feedback.hashes);
193
+ throw new AnchorMismatchError(feedback.text, feedback.hashes, feedback.servedMap);
194
194
  }
195
195
 
196
196
  warnUnicodeEsc(prefixFixed, warnings);
@@ -233,7 +233,7 @@ export function applyEdit(
233
233
  fileHashes,
234
234
  filePath,
235
235
  );
236
- throw new AnchorMismatchError(feedback.text, feedback.hashes);
236
+ throw new AnchorMismatchError(feedback.text, feedback.hashes, feedback.servedMap);
237
237
  }
238
238
  resolved = correctedResult.resolved;
239
239
  }