harnery 0.7.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/README.md +2 -1
  2. package/dist/commander.d.ts +9 -0
  3. package/dist/commander.d.ts.map +1 -1
  4. package/dist/commander.js +2 -0
  5. package/dist/commands/callers.d.ts.map +1 -1
  6. package/dist/commands/callers.js +69 -17
  7. package/dist/commands/doctor.d.ts.map +1 -1
  8. package/dist/commands/doctor.js +84 -2
  9. package/dist/commands/grep.d.ts +77 -0
  10. package/dist/commands/grep.d.ts.map +1 -1
  11. package/dist/commands/grep.js +598 -109
  12. package/dist/commands/init.d.ts +15 -5
  13. package/dist/commands/init.d.ts.map +1 -1
  14. package/dist/commands/init.js +125 -14
  15. package/dist/commands/workflow.d.ts +4 -0
  16. package/dist/commands/workflow.d.ts.map +1 -0
  17. package/dist/commands/workflow.js +89 -0
  18. package/dist/core/agents/cli.js +10 -3
  19. package/dist/core/agents/rules/claim-conflict.d.ts.map +1 -1
  20. package/dist/core/agents/rules/claim-conflict.js +26 -1
  21. package/dist/core/agents/rules/stop-hook.d.ts +8 -0
  22. package/dist/core/agents/rules/stop-hook.d.ts.map +1 -1
  23. package/dist/core/agents/rules/stop-hook.js +8 -0
  24. package/dist/core/agents/state/heartbeat-projector.d.ts +2 -0
  25. package/dist/core/agents/state/heartbeat-projector.d.ts.map +1 -1
  26. package/dist/core/agents/state/heartbeat-projector.js +13 -2
  27. package/dist/core/agents/state/heartbeat-writer.d.ts +16 -2
  28. package/dist/core/agents/state/heartbeat-writer.d.ts.map +1 -1
  29. package/dist/core/agents/state/heartbeat-writer.js +21 -4
  30. package/dist/core/config.d.ts +18 -0
  31. package/dist/core/config.d.ts.map +1 -1
  32. package/dist/core/config.js +38 -0
  33. package/dist/core/hooks/cli.js +29 -3
  34. package/dist/core/hooks/events/schema.d.ts +4 -0
  35. package/dist/core/hooks/events/schema.d.ts.map +1 -1
  36. package/dist/core/hooks/harness/events.d.ts +7 -0
  37. package/dist/core/hooks/harness/events.d.ts.map +1 -1
  38. package/dist/core/hooks/harness/events.js +1 -0
  39. package/dist/core/hooks/resolve/coord-root.d.ts +11 -0
  40. package/dist/core/hooks/resolve/coord-root.d.ts.map +1 -1
  41. package/dist/core/hooks/resolve/coord-root.js +28 -6
  42. package/dist/core/workflow/billing.d.ts +48 -0
  43. package/dist/core/workflow/billing.d.ts.map +1 -0
  44. package/dist/core/workflow/billing.js +102 -0
  45. package/dist/core/workflow/child-env.d.ts +30 -0
  46. package/dist/core/workflow/child-env.d.ts.map +1 -0
  47. package/dist/core/workflow/child-env.js +43 -0
  48. package/dist/core/workflow/engine.d.ts +21 -0
  49. package/dist/core/workflow/engine.d.ts.map +1 -0
  50. package/dist/core/workflow/engine.js +340 -0
  51. package/dist/core/workflow/harnesses.d.ts +17 -0
  52. package/dist/core/workflow/harnesses.d.ts.map +1 -0
  53. package/dist/core/workflow/harnesses.js +30 -0
  54. package/dist/core/workflow/spawn-claude.d.ts +22 -0
  55. package/dist/core/workflow/spawn-claude.d.ts.map +1 -0
  56. package/dist/core/workflow/spawn-claude.js +82 -0
  57. package/dist/core/workflow/spawn-codex.d.ts +19 -0
  58. package/dist/core/workflow/spawn-codex.d.ts.map +1 -0
  59. package/dist/core/workflow/spawn-codex.js +66 -0
  60. package/dist/core/workflow/spawn-cursor.d.ts +26 -0
  61. package/dist/core/workflow/spawn-cursor.d.ts.map +1 -0
  62. package/dist/core/workflow/spawn-cursor.js +72 -0
  63. package/dist/core/workflow/types.d.ts +145 -0
  64. package/dist/core/workflow/types.d.ts.map +1 -0
  65. package/dist/core/workflow/types.js +9 -0
  66. package/dist/core/workflow/validate.d.ts +15 -0
  67. package/dist/core/workflow/validate.d.ts.map +1 -0
  68. package/dist/core/workflow/validate.js +70 -0
  69. package/dist/lib/tools/ripgrep.d.ts +61 -0
  70. package/dist/lib/tools/ripgrep.d.ts.map +1 -0
  71. package/dist/lib/tools/ripgrep.js +217 -0
  72. package/package.json +1 -1
  73. package/src/commander.ts +11 -0
  74. package/src/commands/callers.ts +90 -26
  75. package/src/commands/doctor.ts +95 -2
  76. package/src/commands/grep.ts +737 -113
  77. package/src/commands/init.ts +138 -17
  78. package/src/commands/workflow.ts +132 -0
  79. package/src/core/agents/cli.ts +11 -4
  80. package/src/core/agents/rules/claim-conflict.ts +26 -1
  81. package/src/core/agents/rules/stop-hook.ts +17 -0
  82. package/src/core/agents/state/heartbeat-projector.ts +13 -1
  83. package/src/core/agents/state/heartbeat-writer.ts +30 -5
  84. package/src/core/config.ts +47 -0
  85. package/src/core/hooks/cli.ts +30 -3
  86. package/src/core/hooks/events/schema.ts +4 -0
  87. package/src/core/hooks/harness/events.ts +8 -0
  88. package/src/core/hooks/resolve/coord-root.ts +28 -6
  89. package/src/core/workflow/billing.ts +146 -0
  90. package/src/core/workflow/child-env.ts +47 -0
  91. package/src/core/workflow/engine.ts +394 -0
  92. package/src/core/workflow/harnesses.ts +38 -0
  93. package/src/core/workflow/spawn-claude.ts +99 -0
  94. package/src/core/workflow/spawn-codex.ts +74 -0
  95. package/src/core/workflow/spawn-cursor.ts +89 -0
  96. package/src/core/workflow/types.ts +153 -0
  97. package/src/core/workflow/validate.ts +75 -0
  98. package/src/lib/tools/ripgrep.ts +244 -0
@@ -1,15 +1,40 @@
1
1
  import { spawn } from "node:child_process";
2
+ import { readFile } from "node:fs/promises";
3
+ import { isAbsolute, join } from "node:path";
2
4
  import type { Command } from "commander";
3
5
  import type { EmitContext, HarneryProgramContext } from "../commander.ts";
6
+ import { resolveSearchEngine, type SearchEngine } from "../lib/tools/ripgrep.ts";
4
7
 
5
8
  /**
6
- * `grep`: monorepo-aware code search. Thin wrapper over grep -rn with
7
- * smart default excludes (skip dist/.next/node_modules/.git/...), repo
8
- * scoping (`--repo <name>` or `--all-repos`), and language presets.
9
+ * `grep`: monorepo-aware code search. Prefers ripgrep (`rg`) when it is on
10
+ * PATH and falls back to GNU/BSD grep transparently — both engines are driven
11
+ * with equivalent flags and their output is parsed into the same envelope, so
12
+ * results are identical (pinned by tests/unit/grep-engine.test.ts).
13
+ * Smart default excludes (skip dist/.next/node_modules/.git/...), repo
14
+ * scoping (`--repo <name>` or `--all-repos`), language presets, file-level
15
+ * boolean composition (`--and` / `--without`), and context (`-C`/`-A`/`-B`).
9
16
  *
10
- * Default behavior matches grep's "regex" semantics (`-E` extended). Use
11
- * `-F` / `--literal` to pin to literal-string mode. Output is line-oriented
12
- * `file:line:content` in TTY mode, `{rows, total, truncated}` in --json mode.
17
+ * Engine selection: `HARNERY_GREP_ENGINE=rg|grep` forces one; otherwise `rg`
18
+ * is resolved (managed install, PATH, or opt-in auto-provision; see
19
+ * resolveEngine) and used when available. Repos are searched in parallel, and
20
+ * in `--all-repos` mode the parent scan prunes submodule directories so each
21
+ * match is attributed to exactly one repo.
22
+ *
23
+ * Output framing: content searches request the NUL filename delimiter
24
+ * (`--null` on both engines — the long spelling, because BSD grep repurposes
25
+ * `-Z` for decompression) so filename boundaries are never inferred from
26
+ * punctuation. Context lines are NOT requested from the engines: matches are
27
+ * selected (and budgeted) first, then context windows are materialized from
28
+ * one file read per selected file, which keeps `--limit` semantics exact and
29
+ * both engines byte-identical.
30
+ *
31
+ * Default behavior matches extended-regex semantics (`grep -E` / ripgrep's
32
+ * default). Use `-F` / `--literal` to pin to literal-string mode. Output is
33
+ * line-oriented `file:line:content` in TTY mode (context rows render
34
+ * grep-style as `file-line-content`); `--json` emits the full GrepResult
35
+ * envelope `{ pattern, mode, engine, and_patterns, without_patterns, repos,
36
+ * total_matches, total_files, truncated, elapsed_ms }`. Matches are sorted
37
+ * (file, then line) for stable output across runs and engines.
13
38
  */
14
39
 
15
40
  const DEFAULT_EXCLUDE_DIRS = [
@@ -51,29 +76,40 @@ const LANG_GLOBS: Record<string, string[]> = {
51
76
  rs: ["*.rs"],
52
77
  };
53
78
 
54
- interface GrepOpts {
79
+ export type GrepEngine = SearchEngine;
80
+
81
+ export interface GrepOpts {
55
82
  repo?: string;
56
83
  allRepos?: boolean;
57
- lang?: string;
84
+ /** Repeatable and/or comma-separated (`--lang ts,tsx`). A bare string is accepted for direct callers. */
85
+ lang?: string | string[];
58
86
  ignoreCase?: boolean;
59
87
  wholeWord?: boolean;
60
88
  literal?: boolean;
61
89
  filesOnly?: boolean;
90
+ files?: boolean;
62
91
  count?: boolean;
63
92
  context?: string;
93
+ afterContext?: string;
94
+ beforeContext?: string;
64
95
  maxCount?: string;
65
96
  limit?: string;
66
97
  include?: string[];
67
98
  exclude?: string[];
99
+ and?: string[];
100
+ without?: string[];
101
+ quiet?: boolean;
68
102
  noDefaultExcludes?: boolean;
69
103
  json?: boolean;
70
104
  }
71
105
 
72
- interface Match {
106
+ export interface Match {
73
107
  repo: string;
74
108
  file: string;
75
109
  line: number;
76
110
  text: string;
111
+ /** "match" = primary result row (counts toward totals + --limit); "context" = free -C/-A/-B row. */
112
+ kind: "match" | "context";
77
113
  }
78
114
 
79
115
  export function registerGrepCommand(
@@ -84,20 +120,50 @@ export function registerGrepCommand(
84
120
  program
85
121
  .command("grep <pattern> [paths...]")
86
122
  .description(
87
- "Monorepo-aware code search. Skips dist/.next/node_modules/.git/... by default. " +
88
- "Use --repo, --all-repos, --lang for scoping. Regex by default; -F for literal.",
123
+ "Monorepo-aware code search (ripgrep when available, GNU grep fallback). " +
124
+ "Skips dist/.next/node_modules/.git/... by default. " +
125
+ "Use --repo, --all-repos, --lang for scoping; --and/--without for file-level composition. " +
126
+ "Regex by default; -F for literal.",
89
127
  )
90
128
  .option("--repo <name>", "Scope to one submodule (`.` = parent repo root)")
91
129
  .option("--all-repos", "Search parent + every submodule")
92
- .option("--lang <lang>", `File type preset (${Object.keys(LANG_GLOBS).join(", ")})`)
130
+ .option(
131
+ "--lang <lang>",
132
+ `File type preset, repeatable or comma-separated (${Object.keys(LANG_GLOBS).join(", ")})`,
133
+ collect,
134
+ [] as string[],
135
+ )
93
136
  .option("-i, --ignore-case", "Case-insensitive match")
94
137
  .option("-w, --whole-word", "Match whole words only")
95
138
  .option("-F, --literal", "Treat <pattern> as a literal string (no regex)")
96
139
  .option("-l, --files-only", "Only print file names containing a match")
140
+ .option(
141
+ "--files",
142
+ "Filename search: treat <pattern> as a filename glob and list matching files " +
143
+ "(rg --files when available, POSIX find fallback)",
144
+ )
97
145
  .option("-c, --count", "Print match count per file (suppresses content)")
98
146
  .option("-C, --context <n>", "Print N lines of context around each match", "0")
147
+ .option("-A, --after-context <n>", "Print N lines after each match (overrides -C's after side)")
148
+ .option(
149
+ "-B, --before-context <n>",
150
+ "Print N lines before each match (overrides -C's before side)",
151
+ )
99
152
  .option("--max-count <n>", "Stop after N matches per file")
100
153
  .option("--limit <n>", "Truncate output to N matches total")
154
+ .option(
155
+ "--and <pattern>",
156
+ "Only show files that ALSO contain this pattern (file-level, repeatable)",
157
+ collect,
158
+ [] as string[],
159
+ )
160
+ .option(
161
+ "--without <pattern>",
162
+ "Drop files that contain this pattern (file-level, repeatable)",
163
+ collect,
164
+ [] as string[],
165
+ )
166
+ .option("-q, --quiet", "No output; exit 0 if any match exists, 1 if none")
101
167
  .option("--include <glob>", "Extra --include glob (repeatable)", collect, [] as string[])
102
168
  .option("--exclude <glob>", "Extra --exclude glob (repeatable)", collect, [] as string[])
103
169
  .option("--no-default-excludes", "Disable the default skip list (node_modules, dist, etc.)")
@@ -105,6 +171,12 @@ export function registerGrepCommand(
105
171
  .action(async (pattern: string, paths: string[], opts: GrepOpts) => {
106
172
  try {
107
173
  const result = await runGrep(pattern, paths, opts, context);
174
+ if (opts.quiet) {
175
+ // No output by contract; grep-conventional status. exitCode (not
176
+ // process.exit) so an embedding host isn't terminated mid-flush.
177
+ process.exitCode = result.total_matches > 0 ? 0 : 1;
178
+ return;
179
+ }
108
180
  if (opts.json) {
109
181
  emit.config({ format: "json" });
110
182
  emit.data(result);
@@ -122,9 +194,12 @@ function collect(value: string, prev: string[]): string[] {
122
194
  return [...prev, value];
123
195
  }
124
196
 
125
- interface GrepResult {
197
+ export interface GrepResult {
126
198
  pattern: string;
127
- mode: "regex" | "literal";
199
+ mode: "regex" | "literal" | "files";
200
+ engine: GrepEngine;
201
+ and_patterns: string[];
202
+ without_patterns: string[];
128
203
  repos: { name: string; cwd: string; matches: Match[]; truncated: boolean }[];
129
204
  total_matches: number;
130
205
  total_files: number;
@@ -132,47 +207,204 @@ interface GrepResult {
132
207
  elapsed_ms: number;
133
208
  }
134
209
 
135
- async function runGrep(
210
+ /** Validated, resolved options — built once before repo fan-out. */
211
+ interface NormOpts {
212
+ limit: number; // Infinity when unset
213
+ maxCount: number | undefined;
214
+ before: number;
215
+ after: number;
216
+ langGlobs: string[] | undefined;
217
+ andPatterns: string[];
218
+ withoutPatterns: string[];
219
+ }
220
+
221
+ /**
222
+ * Parse a strictly-decimal integer option. Rejects empty, negative, signed,
223
+ * fractional, and trailing-junk values before any engine is spawned.
224
+ */
225
+ function parseIntOpt(flag: string, raw: string, min: number): number {
226
+ if (!/^\d+$/.test(raw.trim())) {
227
+ throw new Error(`${flag} expects a non-negative integer, got "${raw}"`);
228
+ }
229
+ const n = Number.parseInt(raw.trim(), 10);
230
+ if (n < min) throw new Error(`${flag} must be >= ${min}, got ${n}`);
231
+ return n;
232
+ }
233
+
234
+ function normalizeLangs(lang: string | string[] | undefined): string[] | undefined {
235
+ const rawValues = typeof lang === "string" ? [lang] : (lang ?? []);
236
+ const keys: string[] = [];
237
+ for (const raw of rawValues) {
238
+ for (const piece of raw.split(",")) {
239
+ const key = piece.trim();
240
+ if (key === "") continue;
241
+ if (!LANG_GLOBS[key]) {
242
+ throw new Error(`unknown --lang "${key}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
243
+ }
244
+ if (!keys.includes(key)) keys.push(key);
245
+ }
246
+ }
247
+ if (keys.length === 0) return undefined;
248
+ const globs: string[] = [];
249
+ for (const key of keys) {
250
+ const langGlobs = LANG_GLOBS[key];
251
+ if (!langGlobs) continue;
252
+ for (const g of langGlobs) if (!globs.includes(g)) globs.push(g);
253
+ }
254
+ return globs;
255
+ }
256
+
257
+ /** Validate numerics, resolve context sides, expand langs, enforce the compatibility matrix. */
258
+ function normalizeOpts(opts: GrepOpts): NormOpts {
259
+ const limit = opts.limit ? parseIntOpt("--limit", opts.limit, 1) : Number.POSITIVE_INFINITY;
260
+ const maxCount = opts.maxCount ? parseIntOpt("--max-count", opts.maxCount, 1) : undefined;
261
+ const c = opts.context !== undefined ? parseIntOpt("-C/--context", opts.context, 0) : 0;
262
+ const before =
263
+ opts.beforeContext !== undefined
264
+ ? parseIntOpt("-B/--before-context", opts.beforeContext, 0)
265
+ : c;
266
+ const after =
267
+ opts.afterContext !== undefined ? parseIntOpt("-A/--after-context", opts.afterContext, 0) : c;
268
+ const andPatterns = opts.and ?? [];
269
+ const withoutPatterns = opts.without ?? [];
270
+ const contextActive = before > 0 || after > 0;
271
+
272
+ if (opts.files) {
273
+ // Filename mode lists files by name glob; content-search flags make no
274
+ // sense here — reject loudly rather than silently ignoring them.
275
+ const incompatible: [unknown, string][] = [
276
+ [normalizeLangs(opts.lang)?.length, "--lang"],
277
+ [opts.count, "-c/--count"],
278
+ [opts.wholeWord, "-w/--whole-word"],
279
+ [opts.literal, "-F/--literal"],
280
+ [opts.maxCount, "--max-count"],
281
+ [opts.include?.length, "--include"],
282
+ [contextActive ? true : undefined, "-C/-A/-B context"],
283
+ [andPatterns.length ? true : undefined, "--and"],
284
+ [withoutPatterns.length ? true : undefined, "--without"],
285
+ ];
286
+ for (const [set, flag] of incompatible) {
287
+ if (set) throw new Error(`${flag} does not apply to --files (filename glob) mode`);
288
+ }
289
+ }
290
+ if (contextActive) {
291
+ const rejected: [unknown, string][] = [
292
+ [opts.filesOnly, "-l/--files-only"],
293
+ [opts.count, "-c/--count"],
294
+ [opts.quiet, "-q/--quiet"],
295
+ ];
296
+ for (const [set, flag] of rejected) {
297
+ if (set) throw new Error(`-C/-A/-B context does not combine with ${flag}`);
298
+ }
299
+ }
300
+ if (opts.quiet) {
301
+ const rejected: [unknown, string][] = [
302
+ [opts.json, "--json"],
303
+ [opts.filesOnly, "-l/--files-only"],
304
+ [opts.count, "-c/--count"],
305
+ [opts.limit, "--limit"],
306
+ ];
307
+ for (const [set, flag] of rejected) {
308
+ if (set) throw new Error(`-q/--quiet does not combine with ${flag}`);
309
+ }
310
+ }
311
+
312
+ return {
313
+ // -q only needs existence: one accepted primary row settles the exit code.
314
+ limit: opts.quiet ? 1 : limit,
315
+ maxCount,
316
+ before,
317
+ after,
318
+ langGlobs: normalizeLangs(opts.lang),
319
+ andPatterns,
320
+ withoutPatterns,
321
+ };
322
+ }
323
+
324
+ /** Exported for tests (not part of the package exports map). */
325
+ export async function runGrep(
136
326
  pattern: string,
137
327
  paths: string[],
138
328
  opts: GrepOpts,
139
329
  context: HarneryProgramContext | undefined,
140
330
  ): Promise<GrepResult> {
141
331
  if (!pattern) throw new Error("pattern required");
332
+ const norm = normalizeOpts(opts);
142
333
  const started = Date.now();
143
334
 
335
+ const { engine, rgBin } = await resolveSearchEngine("grep");
144
336
  const repos = resolveRepos(opts, context);
145
- const limit = opts.limit ? Number.parseInt(opts.limit, 10) : Number.POSITIVE_INFINITY;
337
+
338
+ // Host-injected default excludes (generated mirrors, vendored trees, ...)
339
+ // ride the same --no-default-excludes gate as the built-in list.
340
+ const hostExcludeDirs = opts.noDefaultExcludes ? [] : (context?.grepExcludeDirs ?? []);
341
+
342
+ // All repos are searched concurrently; each collects at most limit+1
343
+ // accepted rows (one row of lookahead, so "exactly N results" is
344
+ // distinguishable from "more than N"), then the global budget is applied
345
+ // in repo order below so `--limit` semantics stay deterministic.
346
+ const acceptLimit = Number.isFinite(norm.limit) ? norm.limit + 1 : Number.POSITIVE_INFINITY;
347
+ const perRepo = await Promise.all(
348
+ repos.map((repo) => {
349
+ // In --all-repos mode the parent scan prunes submodule dirs — each
350
+ // submodule gets its own scoped scan, so descending from the parent
351
+ // would double-scan and double-report every submodule match. This
352
+ // pruning is correctness (one repo owns each match), so it applies
353
+ // even when --no-default-excludes is set.
354
+ const dedupeDirs = opts.allRepos && repo.name === "parent" ? (context?.submodules ?? []) : [];
355
+ const extraDirs = [...hostExcludeDirs, ...dedupeDirs];
356
+ return runGrepInRepo(
357
+ pattern,
358
+ paths,
359
+ opts,
360
+ norm,
361
+ repo.cwd,
362
+ repo.name,
363
+ acceptLimit,
364
+ engine,
365
+ rgBin,
366
+ extraDirs,
367
+ );
368
+ }),
369
+ );
146
370
 
147
371
  const allRepoResults: GrepResult["repos"] = [];
148
372
  let totalMatches = 0;
149
373
  const filesSeen = new Set<string>();
150
374
  let truncated = false;
375
+ let budget = norm.limit;
151
376
 
152
- for (const repo of repos) {
153
- if (truncated) break;
154
- const repoLimit = Number.isFinite(limit)
155
- ? Math.max(0, limit - totalMatches)
156
- : Number.POSITIVE_INFINITY;
157
- if (repoLimit === 0) {
158
- truncated = true;
159
- allRepoResults.push({ name: repo.name, cwd: repo.cwd, matches: [], truncated: true });
160
- break;
161
- }
162
- const matches = await runGrepInRepo(pattern, paths, opts, repo.cwd, repo.name, repoLimit);
163
- let repoTruncated = false;
164
- if (Number.isFinite(repoLimit) && matches.length >= repoLimit) {
165
- repoTruncated = true;
166
- truncated = true;
167
- }
377
+ for (let i = 0; i < repos.length; i++) {
378
+ const repo = repos[i];
379
+ if (!repo) continue;
380
+ const collected = perRepo[i] ?? [];
381
+ collected.sort((a, b) => (a.file < b.file ? -1 : a.file > b.file ? 1 : a.line - b.line));
382
+ const take = Number.isFinite(budget)
383
+ ? Math.min(collected.length, Math.max(0, budget))
384
+ : collected.length;
385
+ let matches = collected.slice(0, take);
386
+ // truncated only when an accepted primary row was actually omitted —
387
+ // the lookahead row (or a later-repo surplus) is the proof.
388
+ const repoTruncated = collected.length > take;
389
+ if (repoTruncated) truncated = true;
390
+ if (Number.isFinite(budget)) budget -= take;
168
391
  totalMatches += matches.length;
169
392
  for (const m of matches) filesSeen.add(`${repo.name}/${m.file}`);
393
+ // Context is materialized only for the SELECTED rows, after budgeting,
394
+ // so context rows never consume the limit and trailing context is never
395
+ // cut off by an engine kill.
396
+ if (!opts.files && !opts.filesOnly && !opts.count && (norm.before > 0 || norm.after > 0)) {
397
+ matches = await materializeContext(matches, repo.cwd, norm.before, norm.after);
398
+ }
170
399
  allRepoResults.push({ name: repo.name, cwd: repo.cwd, matches, truncated: repoTruncated });
171
400
  }
172
401
 
173
402
  return {
174
403
  pattern,
175
- mode: opts.literal ? "literal" : "regex",
404
+ mode: opts.files ? "files" : opts.literal ? "literal" : "regex",
405
+ engine,
406
+ and_patterns: norm.andPatterns,
407
+ without_patterns: norm.withoutPatterns,
176
408
  repos: allRepoResults,
177
409
  total_matches: totalMatches,
178
410
  total_files: filesSeen.size,
@@ -220,71 +452,403 @@ async function runGrepInRepo(
220
452
  pattern: string,
221
453
  paths: string[],
222
454
  opts: GrepOpts,
455
+ norm: NormOpts,
223
456
  cwd: string,
224
457
  repoName: string,
225
- limit: number,
458
+ acceptLimit: number,
459
+ engine: GrepEngine,
460
+ rgBin: string,
461
+ extraExcludeDirs: readonly string[],
226
462
  ): Promise<Match[]> {
227
- const args: string[] = ["-rn", "--color=never", "-I"]; // -I: skip binary files
463
+ // File-level boolean composition: build the complete auxiliary file sets
464
+ // FIRST (no limit — a partial set can't prove absence), then feed the
465
+ // primary scan an in-stream predicate. Running the primary with its limit
466
+ // and filtering afterwards would let non-qualifying early hits consume the
467
+ // budget and hide later valid results.
468
+ let predicate: ((file: string) => boolean) | undefined;
469
+ if (norm.andPatterns.length > 0 || norm.withoutPatterns.length > 0) {
470
+ const requiredSets: Set<string>[] = [];
471
+ // Sequential, not parallel: repeated flags must not multiply peak child
472
+ // process concurrency by repos x patterns. Repo workers stay parallel.
473
+ for (const p of norm.andPatterns) {
474
+ const files = await runFileSetScan(
475
+ p,
476
+ paths,
477
+ opts,
478
+ norm,
479
+ cwd,
480
+ engine,
481
+ rgBin,
482
+ extraExcludeDirs,
483
+ );
484
+ if (files.size === 0) return []; // intersection is already empty
485
+ requiredSets.push(files);
486
+ }
487
+ const forbidden = new Set<string>();
488
+ for (const p of norm.withoutPatterns) {
489
+ const files = await runFileSetScan(
490
+ p,
491
+ paths,
492
+ opts,
493
+ norm,
494
+ cwd,
495
+ engine,
496
+ rgBin,
497
+ extraExcludeDirs,
498
+ );
499
+ for (const f of files) forbidden.add(f);
500
+ }
501
+ predicate = (file) => requiredSets.every((s) => s.has(file)) && !forbidden.has(file);
502
+ }
503
+
504
+ const [bin, args, decodeMode] = buildEngineInvocation(
505
+ pattern,
506
+ paths,
507
+ opts,
508
+ norm,
509
+ engine,
510
+ rgBin,
511
+ extraExcludeDirs,
512
+ false,
513
+ );
514
+
515
+ let matches = await execSearch(bin, args, cwd, acceptLimit, decodeMode, predicate, false);
516
+ // -c mode: GNU grep prints a `path:0` row for every searched file; ripgrep
517
+ // omits zero-count rows. Filter zeros on both engines so output matches.
518
+ if (opts.count) matches = matches.filter((m) => m.line > 0);
519
+ return matches.map((m) => ({ ...m, repo: repoName }));
520
+ }
521
+
522
+ /** Membership scan for one --and/--without pattern: complete files-only set, strict failure. */
523
+ async function runFileSetScan(
524
+ pattern: string,
525
+ paths: string[],
526
+ opts: GrepOpts,
527
+ norm: NormOpts,
528
+ cwd: string,
529
+ engine: GrepEngine,
530
+ rgBin: string,
531
+ extraExcludeDirs: readonly string[],
532
+ ): Promise<Set<string>> {
533
+ const [bin, args, decodeMode] = buildEngineInvocation(
534
+ pattern,
535
+ paths,
536
+ opts,
537
+ norm,
538
+ engine,
539
+ rgBin,
540
+ extraExcludeDirs,
541
+ true,
542
+ );
543
+ const rows = await execSearch(
544
+ bin,
545
+ args,
546
+ cwd,
547
+ Number.POSITIVE_INFINITY,
548
+ decodeMode,
549
+ undefined,
550
+ true,
551
+ );
552
+ return new Set(rows.map((r) => r.file));
553
+ }
554
+
555
+ type DecodeMode = "content" | "count" | "filesOnly" | "plainLines";
556
+
557
+ /**
558
+ * Build one engine invocation. `membership: true` forces files-only mode for
559
+ * --and/--without set scans (match-shaping + scope flags shared with the
560
+ * primary; output flags not applied).
561
+ */
562
+ function buildEngineInvocation(
563
+ pattern: string,
564
+ paths: string[],
565
+ opts: GrepOpts,
566
+ norm: NormOpts,
567
+ engine: GrepEngine,
568
+ rgBin: string,
569
+ extraExcludeDirs: readonly string[],
570
+ membership: boolean,
571
+ ): [string, string[], DecodeMode] {
572
+ if (opts.files) {
573
+ // Filename mode keeps its existing rg/find framing (plain lines).
574
+ return engine === "rg"
575
+ ? [rgBin, buildRgFilesArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"]
576
+ : ["find", buildFindArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"];
577
+ }
578
+ const filesOnly = membership || Boolean(opts.filesOnly);
579
+ const count = !membership && Boolean(opts.count);
580
+ const decodeMode: DecodeMode = filesOnly ? "filesOnly" : count ? "count" : "content";
581
+ const args =
582
+ engine === "rg"
583
+ ? buildRgArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count })
584
+ : buildGrepArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count });
585
+ return [engine === "rg" ? rgBin : "grep", args, decodeMode];
586
+ }
587
+
588
+ interface OutputShape {
589
+ filesOnly: boolean;
590
+ count: boolean;
591
+ }
592
+
593
+ function buildGrepArgs(
594
+ pattern: string,
595
+ paths: string[],
596
+ opts: GrepOpts,
597
+ norm: NormOpts,
598
+ extraExcludeDirs: readonly string[],
599
+ shape: OutputShape,
600
+ ): string[] {
601
+ // --null (NOT -Z: BSD grep repurposes -Z for decompression): NUL after the
602
+ // filename, so paths containing colons/dashes can't confuse the decoder.
603
+ // -I: skip binary files. Context is never requested from the engine — see
604
+ // materializeContext.
605
+ const args: string[] = ["-rn", "-H", "--null", "--color=never", "-I"];
228
606
  if (opts.ignoreCase) args.push("-i");
229
607
  if (opts.wholeWord) args.push("-w");
230
608
  if (opts.literal) args.push("-F");
231
609
  else args.push("-E");
232
- if (opts.filesOnly) args.push("-l");
233
- if (opts.count) args.push("-c");
234
- const contextN = Number.parseInt(opts.context ?? "0", 10);
235
- if (Number.isFinite(contextN) && contextN > 0) args.push(`-C${contextN}`);
236
- if (opts.maxCount) {
237
- const n = Number.parseInt(opts.maxCount, 10);
238
- if (Number.isFinite(n) && n > 0) args.push(`-m${n}`);
239
- }
610
+ if (shape.filesOnly) args.push("-l");
611
+ if (shape.count) args.push("-c");
612
+ if (norm.maxCount !== undefined) args.push(`-m${norm.maxCount}`);
240
613
 
241
614
  if (!opts.noDefaultExcludes) {
242
615
  for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--exclude-dir=${d}`);
243
616
  for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--exclude=${f}`);
244
617
  }
618
+ for (const d of extraExcludeDirs) args.push(`--exclude-dir=${d}`);
245
619
  for (const e of opts.exclude ?? []) args.push(`--exclude=${e}`);
246
620
 
247
- const langGlobs = opts.lang ? LANG_GLOBS[opts.lang] : undefined;
248
- if (opts.lang && !langGlobs) {
249
- throw new Error(`unknown --lang "${opts.lang}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
250
- }
251
- if (langGlobs) for (const g of langGlobs) args.push(`--include=${g}`);
621
+ if (norm.langGlobs) for (const g of norm.langGlobs) args.push(`--include=${g}`);
252
622
  for (const g of opts.include ?? []) args.push(`--include=${g}`);
253
623
 
254
624
  args.push("--", pattern);
255
625
  if (paths.length > 0) args.push(...paths);
256
626
  else args.push(".");
627
+ return args;
628
+ }
257
629
 
258
- const matches = await execGrep(args, cwd, limit);
259
- return matches.map((m) => ({ ...m, repo: repoName }));
630
+ function buildRgArgs(
631
+ pattern: string,
632
+ paths: string[],
633
+ opts: GrepOpts,
634
+ norm: NormOpts,
635
+ extraExcludeDirs: readonly string[],
636
+ shape: OutputShape,
637
+ ): string[] {
638
+ // --hidden --no-ignore: match GNU grep's semantics (search dotdirs, ignore
639
+ // .gitignore) so the only filters are the explicit exclude lists.
640
+ // --no-config: a user's ripgrep config file must not skew results.
641
+ // --with-filename: rg drops the file prefix for a single explicit file arg,
642
+ // which would break the shared decoder.
643
+ // --null: NUL after the filename (same spelling as GNU/BSD grep's long flag).
644
+ const args: string[] = [
645
+ "-n",
646
+ "--no-heading",
647
+ "--with-filename",
648
+ "--null",
649
+ "--color=never",
650
+ "--no-config",
651
+ "--hidden",
652
+ "--no-ignore",
653
+ ];
654
+ if (opts.ignoreCase) args.push("-i");
655
+ if (opts.wholeWord) args.push("-w");
656
+ if (opts.literal) args.push("-F");
657
+ if (shape.filesOnly) args.push("-l");
658
+ if (shape.count) args.push("-c");
659
+ if (norm.maxCount !== undefined) args.push(`-m${norm.maxCount}`);
660
+
661
+ // Gitignore-style globs: a bare name matches (and prunes) at any depth,
662
+ // mirroring grep's --exclude-dir / --exclude basename semantics. ORDER
663
+ // MATTERS: rg globs are last-match-wins, so positives (includes/lang) go
664
+ // first and negatives (excludes) last — otherwise `--include '*.md'` would
665
+ // re-include a .md file inside an excluded node_modules/. grep's
666
+ // --exclude-dir always beats --include, so this keeps the engines aligned.
667
+ if (norm.langGlobs) for (const g of norm.langGlobs) args.push(`--glob=${g}`);
668
+ for (const g of opts.include ?? []) args.push(`--glob=${g}`);
669
+
670
+ if (!opts.noDefaultExcludes) {
671
+ for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--glob=!${d}`);
672
+ for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--glob=!${f}`);
673
+ }
674
+ for (const d of extraExcludeDirs) args.push(`--glob=!${d}`);
675
+ for (const e of opts.exclude ?? []) args.push(`--glob=!${e}`);
676
+
677
+ args.push("--", pattern);
678
+ if (paths.length > 0) args.push(...paths);
679
+ else args.push(".");
680
+ return args;
260
681
  }
261
682
 
262
- function execGrep(args: string[], cwd: string, limit: number): Promise<Omit<Match, "repo">[]> {
683
+ /**
684
+ * Filename mode via ripgrep: `--files` lists files; a single positive glob
685
+ * acts as a whitelist, negative globs prune. `--iglob` gives -i semantics.
686
+ */
687
+ function buildRgFilesArgs(
688
+ pattern: string,
689
+ paths: string[],
690
+ opts: GrepOpts,
691
+ extraExcludeDirs: readonly string[],
692
+ ): string[] {
693
+ const args: string[] = ["--files", "--hidden", "--no-ignore", "--no-config", "--color=never"];
694
+ // Positive pattern FIRST, negatives last (rg globs are last-match-wins;
695
+ // excludes must beat the pattern — see the ordering note in buildRgArgs).
696
+ args.push(`${opts.ignoreCase ? "--iglob" : "--glob"}=${pattern}`);
697
+ if (!opts.noDefaultExcludes) {
698
+ for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--glob=!${d}`);
699
+ for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--glob=!${f}`);
700
+ }
701
+ for (const d of extraExcludeDirs) args.push(`--glob=!${d}`);
702
+ for (const e of opts.exclude ?? []) args.push(`--glob=!${e}`);
703
+ if (paths.length > 0) args.push(...paths);
704
+ else args.push(".");
705
+ return args;
706
+ }
707
+
708
+ /**
709
+ * Filename mode via POSIX find (the no-ripgrep fallback):
710
+ * find <paths> ( -name d1 -o -name d2 ... ) -prune -o -type f -name <glob> [! -name <ex>]... -print
711
+ * Sticks to POSIX operators (`(`, `-o`, `!`) so BSD/macOS find behaves the same.
712
+ */
713
+ function buildFindArgs(
714
+ pattern: string,
715
+ paths: string[],
716
+ opts: GrepOpts,
717
+ extraExcludeDirs: readonly string[],
718
+ ): string[] {
719
+ const args: string[] = paths.length > 0 ? [...paths] : ["."];
720
+ const pruneDirs = [...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_DIRS), ...extraExcludeDirs];
721
+ if (pruneDirs.length > 0) {
722
+ args.push("(");
723
+ pruneDirs.forEach((d, i) => {
724
+ if (i > 0) args.push("-o");
725
+ args.push("-name", d);
726
+ });
727
+ args.push(")", "-prune", "-o");
728
+ }
729
+ args.push("-type", "f", opts.ignoreCase ? "-iname" : "-name", pattern);
730
+ const fileExcludes = [
731
+ ...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_FILES),
732
+ ...(opts.exclude ?? []),
733
+ ];
734
+ for (const e of fileExcludes) args.push("!", "-name", e);
735
+ args.push("-print");
736
+ return args;
737
+ }
738
+
739
+ /**
740
+ * Streaming decoder for NUL-framed engine output. Exported for direct
741
+ * chunk-boundary tests. Record shapes (both engines, pinned by the parity
742
+ * suite):
743
+ * content: <path>NUL<line>:<text>LF
744
+ * count: <path>NUL<count>LF
745
+ * filesOnly: <path>NUL (NUL is the record terminator)
746
+ * plainLines: <path>LF (files mode: rg --files / find)
747
+ *
748
+ * Operates on Buffers so a chunk boundary can fall anywhere — including
749
+ * inside a multi-byte UTF-8 sequence — without corrupting a record: bytes
750
+ * are only decoded to strings once a full record is framed.
751
+ */
752
+ export class NulDecoder {
753
+ private buffer: Buffer = Buffer.alloc(0);
754
+ constructor(private readonly mode: DecodeMode) {}
755
+
756
+ push(chunk: Buffer): Omit<Match, "repo" | "kind">[] {
757
+ this.buffer = this.buffer.length === 0 ? chunk : Buffer.concat([this.buffer, chunk]);
758
+ const rows: Omit<Match, "repo" | "kind">[] = [];
759
+ if (this.mode === "filesOnly") {
760
+ let nul = this.buffer.indexOf(0);
761
+ while (nul >= 0) {
762
+ const path = this.buffer.subarray(0, nul).toString("utf8");
763
+ this.buffer = this.buffer.subarray(nul + 1);
764
+ // Engines may still newline-separate NUL-terminated records in edge
765
+ // configurations; a bare leading LF is framing residue, not a path.
766
+ const cleaned = path.startsWith("\n") ? path.slice(1) : path;
767
+ if (cleaned.length > 0) rows.push({ file: normalizeFile(cleaned), line: 0, text: "" });
768
+ nul = this.buffer.indexOf(0);
769
+ }
770
+ return rows;
771
+ }
772
+ let nl = this.buffer.indexOf(0x0a);
773
+ while (nl >= 0) {
774
+ const record = this.buffer.subarray(0, nl);
775
+ this.buffer = this.buffer.subarray(nl + 1);
776
+ const row = this.decodeRecord(record);
777
+ if (row) rows.push(row);
778
+ nl = this.buffer.indexOf(0x0a);
779
+ }
780
+ return rows;
781
+ }
782
+
783
+ /** Flush a trailing unterminated record (engine killed mid-write, or no final LF). */
784
+ flush(): Omit<Match, "repo" | "kind">[] {
785
+ if (this.buffer.length === 0) return [];
786
+ const record = this.buffer;
787
+ this.buffer = Buffer.alloc(0);
788
+ if (this.mode === "filesOnly") {
789
+ // A partial filesOnly record has no terminating NUL — an engine killed
790
+ // mid-path would yield a corrupt name; drop it.
791
+ return [];
792
+ }
793
+ const row = this.decodeRecord(record);
794
+ return row ? [row] : [];
795
+ }
796
+
797
+ private decodeRecord(record: Buffer): Omit<Match, "repo" | "kind"> | null {
798
+ if (this.mode === "plainLines") {
799
+ const path = record.toString("utf8");
800
+ if (path.length === 0) return null;
801
+ return { file: normalizeFile(path), line: 0, text: "" };
802
+ }
803
+ const nul = record.indexOf(0);
804
+ if (nul < 0) return null; // not a NUL-framed record (stray engine chatter)
805
+ const file = normalizeFile(record.subarray(0, nul).toString("utf8"));
806
+ const rest = record.subarray(nul + 1).toString("utf8");
807
+ if (this.mode === "count") {
808
+ const count = Number.parseInt(rest, 10);
809
+ if (!Number.isFinite(count)) return null;
810
+ return { file, line: count, text: "" };
811
+ }
812
+ // content: <line>:<text>. The engines never emit context/separator rows
813
+ // (context is materialized from file reads), so `:` is always present.
814
+ const colon = rest.indexOf(":");
815
+ if (colon < 0) return null;
816
+ const lineNum = Number.parseInt(rest.slice(0, colon), 10);
817
+ if (!Number.isFinite(lineNum)) return null;
818
+ return { file, line: lineNum, text: rest.slice(colon + 1) };
819
+ }
820
+ }
821
+
822
+ function execSearch(
823
+ bin: string,
824
+ args: string[],
825
+ cwd: string,
826
+ acceptLimit: number,
827
+ decodeMode: DecodeMode,
828
+ filePredicate: ((file: string) => boolean) | undefined,
829
+ strict: boolean,
830
+ ): Promise<Match[]> {
263
831
  return new Promise((resolveP, reject) => {
264
- const proc = spawn("grep", args, { cwd, stdio: ["ignore", "pipe", "pipe"] });
265
- const matches: Omit<Match, "repo">[] = [];
266
- let buffer = "";
832
+ const proc = spawn(bin, args, { cwd, stdio: ["ignore", "pipe", "pipe"] });
833
+ const decoder = new NulDecoder(decodeMode);
834
+ const matches: Match[] = [];
267
835
  let stderr = "";
268
- let truncated = false;
269
-
270
- proc.stdout.setEncoding("utf8");
271
- proc.stdout.on("data", (chunk: string) => {
272
- if (truncated) return;
273
- buffer += chunk;
274
- let nl = buffer.indexOf("\n");
275
- while (nl >= 0) {
276
- const line = buffer.slice(0, nl);
277
- buffer = buffer.slice(nl + 1);
278
- if (line.length !== 0) {
279
- const parsed = parseGrepLine(line);
280
- if (parsed) matches.push(parsed);
281
- if (Number.isFinite(limit) && matches.length >= limit) {
282
- truncated = true;
283
- proc.kill("SIGTERM");
284
- return;
285
- }
286
- }
287
- nl = buffer.indexOf("\n");
836
+ let killed = false;
837
+
838
+ const accept = (rows: Omit<Match, "repo" | "kind">[]): boolean => {
839
+ for (const row of rows) {
840
+ if (filePredicate && !filePredicate(row.file)) continue;
841
+ matches.push({ ...row, repo: "", kind: "match" });
842
+ if (Number.isFinite(acceptLimit) && matches.length >= acceptLimit) return true;
843
+ }
844
+ return false;
845
+ };
846
+
847
+ proc.stdout.on("data", (chunk: Buffer) => {
848
+ if (killed) return;
849
+ if (accept(decoder.push(chunk))) {
850
+ killed = true;
851
+ proc.kill("SIGTERM");
288
852
  }
289
853
  });
290
854
  proc.stderr.setEncoding("utf8");
@@ -293,13 +857,15 @@ function execGrep(args: string[], cwd: string, limit: number): Promise<Omit<Matc
293
857
  });
294
858
  proc.on("error", reject);
295
859
  proc.on("close", (code) => {
296
- if (buffer.length > 0 && !truncated) {
297
- const parsed = parseGrepLine(buffer);
298
- if (parsed) matches.push(parsed);
299
- }
300
- // grep exits 1 on "no matches": normal, not an error.
301
- if (code !== null && code !== 0 && code !== 1 && !truncated) {
302
- reject(new Error(`grep exited ${code}: ${stderr.trim() || "(no stderr)"}`));
860
+ if (!killed) accept(decoder.flush());
861
+ // Exit 1 is "no matches" on both engines: normal, not an error.
862
+ // Primary scans tolerate exit 2 with matches collected (an unreadable
863
+ // file mid-walk): surface the results rather than throwing them away.
864
+ // STRICT scans (--and/--without membership) must reject on ANY partial
865
+ // failure — an incomplete set cannot prove that a file lacks a pattern.
866
+ const failed = code !== null && code !== 0 && code !== 1 && !killed;
867
+ if (failed && (strict || matches.length === 0)) {
868
+ reject(new Error(`${bin} exited ${code}: ${stderr.trim() || "(no stderr)"}`));
303
869
  return;
304
870
  }
305
871
  resolveP(matches);
@@ -307,32 +873,81 @@ function execGrep(args: string[], cwd: string, limit: number): Promise<Omit<Matc
307
873
  });
308
874
  }
309
875
 
310
- /** Parse `path:line:content` from grep -n output. Returns null for ambiguous (no line number) lines. */
311
- function parseGrepLine(line: string): Omit<Match, "repo"> | null {
312
- // First colon ends path; second colon ends line number (if numeric).
313
- const firstColon = line.indexOf(":");
314
- if (firstColon < 0) {
315
- // -l mode: just a path. Encode as line=0, text="".
316
- return { file: line, line: 0, text: "" };
317
- }
318
- const after = line.slice(firstColon + 1);
319
- const secondColon = after.indexOf(":");
320
- if (secondColon < 0) {
321
- // -c mode: `path:count`.
322
- const count = Number.parseInt(after, 10);
323
- if (Number.isFinite(count)) return { file: line.slice(0, firstColon), line: count, text: "" };
324
- return { file: line.slice(0, firstColon), line: 0, text: after };
876
+ /**
877
+ * Materialize -C/-A/-B context for already-selected match rows: one file
878
+ * read per selected file, intervals clamped + merged, match rows keep the
879
+ * engine-captured text, other window lines become kind:"context" rows.
880
+ * An unreadable file (deleted since the scan — the accepted TOCTOU window)
881
+ * degrades to its match rows without context rather than failing the search.
882
+ */
883
+ async function materializeContext(
884
+ selected: Match[],
885
+ cwd: string,
886
+ before: number,
887
+ after: number,
888
+ ): Promise<Match[]> {
889
+ const byFile = new Map<string, Match[]>();
890
+ for (const m of selected) {
891
+ const rows = byFile.get(m.file);
892
+ if (rows) rows.push(m);
893
+ else byFile.set(m.file, [m]);
325
894
  }
326
- const lineNum = Number.parseInt(after.slice(0, secondColon), 10);
327
- if (!Number.isFinite(lineNum)) {
328
- // Probably a context separator like `--`. Skip.
329
- return null;
895
+
896
+ const out: Match[] = [];
897
+ for (const [file, rows] of byFile) {
898
+ const abs = isAbsolute(file) ? file : join(cwd, file);
899
+ let lines: string[] | null = null;
900
+ try {
901
+ const content = await readFile(abs, "utf8");
902
+ lines = content.split("\n");
903
+ if (lines.at(-1) === "") lines.pop();
904
+ } catch {
905
+ lines = null;
906
+ }
907
+ if (lines === null) {
908
+ out.push(...rows);
909
+ continue;
910
+ }
911
+ // Merge overlapping/adjacent [line-before, line+after] windows.
912
+ const intervals: [number, number][] = rows
913
+ .map((m): [number, number] => [
914
+ Math.max(1, m.line - before),
915
+ Math.min(lines.length, m.line + after),
916
+ ])
917
+ .sort((a, b) => a[0] - b[0]);
918
+ const merged: [number, number][] = [];
919
+ for (const iv of intervals) {
920
+ const last = merged.at(-1);
921
+ if (last && iv[0] <= last[1] + 1) last[1] = Math.max(last[1], iv[1]);
922
+ else merged.push([iv[0], iv[1]]);
923
+ }
924
+ const matchByLine = new Map(rows.map((m) => [m.line, m]));
925
+ const emitted = new Set<Match>();
926
+ for (const [start, end] of merged) {
927
+ if (start > end) continue;
928
+ for (let n = start; n <= end; n++) {
929
+ const m = matchByLine.get(n);
930
+ if (m) {
931
+ out.push(m);
932
+ emitted.add(m);
933
+ } else {
934
+ const repo = rows[0]?.repo ?? "";
935
+ out.push({ repo, file, line: n, text: lines[n - 1] ?? "", kind: "context" });
936
+ }
937
+ }
938
+ }
939
+ // A match whose line exceeds the file's current length (file shrank
940
+ // since the scan) sits outside every clamped window — still emit it.
941
+ for (const m of rows) {
942
+ if (!emitted.has(m)) out.push(m);
943
+ }
330
944
  }
331
- return {
332
- file: line.slice(0, firstColon),
333
- line: lineNum,
334
- text: after.slice(secondColon + 1),
335
- };
945
+ return out;
946
+ }
947
+
948
+ /** Both engines may emit a leading `./` for the default path; strip it so output is engine-identical. */
949
+ function normalizeFile(file: string): string {
950
+ return file.startsWith("./") ? file.slice(2) : file;
336
951
  }
337
952
 
338
953
  function renderResult(r: GrepResult, opts: GrepOpts): string {
@@ -340,32 +955,41 @@ function renderResult(r: GrepResult, opts: GrepOpts): string {
340
955
  const showRepoHeader = r.repos.length > 1;
341
956
  for (const repo of r.repos) {
342
957
  if (repo.matches.length === 0) continue;
958
+ const primary = repo.matches.filter((m) => m.kind === "match").length;
343
959
  if (showRepoHeader) {
344
960
  lines.push("");
345
- lines.push(
346
- `── ${repo.name} (${repo.matches.length} match${repo.matches.length === 1 ? "" : "es"}) ──`,
347
- );
961
+ lines.push(`── ${repo.name} (${primary} match${primary === 1 ? "" : "es"}) ──`);
348
962
  }
963
+ // Context mode: `--` between disjoint windows (file change or line gap).
964
+ const contextMode = repo.matches.some((x) => x.kind === "context");
965
+ let prev: Match | undefined;
349
966
  for (const m of repo.matches) {
350
- if (opts.filesOnly) {
967
+ if (contextMode && prev && (prev.file !== m.file || m.line !== prev.line + 1)) {
968
+ lines.push("--");
969
+ }
970
+ if (opts.filesOnly || opts.files) {
351
971
  lines.push(m.file);
352
972
  } else if (opts.count) {
353
973
  lines.push(`${m.file}:${m.line}`);
974
+ } else if (m.kind === "context") {
975
+ lines.push(`${m.file}-${m.line}-${m.text}`);
354
976
  } else {
355
977
  lines.push(`${m.file}:${m.line}:${m.text}`);
356
978
  }
979
+ prev = m;
357
980
  }
358
981
  if (repo.truncated) {
359
- lines.push(" (truncated; pass --limit to widen)");
982
+ lines.push(" (truncated; raise --limit)");
360
983
  }
361
984
  }
362
985
  if (r.total_matches === 0) {
363
- lines.push(`(no matches for /${r.pattern}/${r.mode === "literal" ? " (literal)" : ""})`);
986
+ const modeTag = r.mode === "literal" ? " (literal)" : r.mode === "files" ? " (files)" : "";
987
+ lines.push(`(no matches for /${r.pattern}/${modeTag})`);
364
988
  } else if (r.repos.length > 1 || r.truncated) {
365
989
  lines.push("");
366
990
  const summary = `${r.total_matches} match${r.total_matches === 1 ? "" : "es"} across ${r.total_files} file${r.total_files === 1 ? "" : "s"}`;
367
- const tail = r.truncated ? " (truncated; use --limit)" : "";
368
- lines.push(`${summary}${tail} (${r.elapsed_ms}ms)`);
991
+ const tail = r.truncated ? " (truncated; raise --limit)" : "";
992
+ lines.push(`${summary}${tail} (${r.elapsed_ms}ms, ${r.engine})`);
369
993
  }
370
994
  return lines.join("\n");
371
995
  }