harnery 0.7.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/README.md +2 -1
  2. package/dist/commander.d.ts +9 -0
  3. package/dist/commander.d.ts.map +1 -1
  4. package/dist/commander.js +2 -0
  5. package/dist/commands/callers.d.ts.map +1 -1
  6. package/dist/commands/callers.js +69 -17
  7. package/dist/commands/doctor.d.ts.map +1 -1
  8. package/dist/commands/doctor.js +84 -2
  9. package/dist/commands/grep.d.ts +77 -0
  10. package/dist/commands/grep.d.ts.map +1 -1
  11. package/dist/commands/grep.js +598 -109
  12. package/dist/commands/init.d.ts +15 -5
  13. package/dist/commands/init.d.ts.map +1 -1
  14. package/dist/commands/init.js +125 -14
  15. package/dist/commands/workflow.d.ts +4 -0
  16. package/dist/commands/workflow.d.ts.map +1 -0
  17. package/dist/commands/workflow.js +89 -0
  18. package/dist/core/agents/cli.js +10 -3
  19. package/dist/core/agents/rules/claim-conflict.d.ts.map +1 -1
  20. package/dist/core/agents/rules/claim-conflict.js +26 -1
  21. package/dist/core/agents/rules/stop-hook.d.ts +8 -0
  22. package/dist/core/agents/rules/stop-hook.d.ts.map +1 -1
  23. package/dist/core/agents/rules/stop-hook.js +8 -0
  24. package/dist/core/agents/state/heartbeat-projector.d.ts +2 -0
  25. package/dist/core/agents/state/heartbeat-projector.d.ts.map +1 -1
  26. package/dist/core/agents/state/heartbeat-projector.js +13 -2
  27. package/dist/core/agents/state/heartbeat-writer.d.ts +16 -2
  28. package/dist/core/agents/state/heartbeat-writer.d.ts.map +1 -1
  29. package/dist/core/agents/state/heartbeat-writer.js +21 -4
  30. package/dist/core/config.d.ts +18 -0
  31. package/dist/core/config.d.ts.map +1 -1
  32. package/dist/core/config.js +38 -0
  33. package/dist/core/hooks/cli.js +29 -3
  34. package/dist/core/hooks/events/schema.d.ts +4 -0
  35. package/dist/core/hooks/events/schema.d.ts.map +1 -1
  36. package/dist/core/hooks/harness/events.d.ts +7 -0
  37. package/dist/core/hooks/harness/events.d.ts.map +1 -1
  38. package/dist/core/hooks/harness/events.js +1 -0
  39. package/dist/core/hooks/resolve/coord-root.d.ts +11 -0
  40. package/dist/core/hooks/resolve/coord-root.d.ts.map +1 -1
  41. package/dist/core/hooks/resolve/coord-root.js +28 -6
  42. package/dist/core/workflow/billing.d.ts +48 -0
  43. package/dist/core/workflow/billing.d.ts.map +1 -0
  44. package/dist/core/workflow/billing.js +102 -0
  45. package/dist/core/workflow/child-env.d.ts +30 -0
  46. package/dist/core/workflow/child-env.d.ts.map +1 -0
  47. package/dist/core/workflow/child-env.js +43 -0
  48. package/dist/core/workflow/engine.d.ts +21 -0
  49. package/dist/core/workflow/engine.d.ts.map +1 -0
  50. package/dist/core/workflow/engine.js +340 -0
  51. package/dist/core/workflow/harnesses.d.ts +17 -0
  52. package/dist/core/workflow/harnesses.d.ts.map +1 -0
  53. package/dist/core/workflow/harnesses.js +30 -0
  54. package/dist/core/workflow/spawn-claude.d.ts +22 -0
  55. package/dist/core/workflow/spawn-claude.d.ts.map +1 -0
  56. package/dist/core/workflow/spawn-claude.js +82 -0
  57. package/dist/core/workflow/spawn-codex.d.ts +19 -0
  58. package/dist/core/workflow/spawn-codex.d.ts.map +1 -0
  59. package/dist/core/workflow/spawn-codex.js +66 -0
  60. package/dist/core/workflow/spawn-cursor.d.ts +26 -0
  61. package/dist/core/workflow/spawn-cursor.d.ts.map +1 -0
  62. package/dist/core/workflow/spawn-cursor.js +72 -0
  63. package/dist/core/workflow/types.d.ts +145 -0
  64. package/dist/core/workflow/types.d.ts.map +1 -0
  65. package/dist/core/workflow/types.js +9 -0
  66. package/dist/core/workflow/validate.d.ts +15 -0
  67. package/dist/core/workflow/validate.d.ts.map +1 -0
  68. package/dist/core/workflow/validate.js +70 -0
  69. package/dist/lib/tools/ripgrep.d.ts +61 -0
  70. package/dist/lib/tools/ripgrep.d.ts.map +1 -0
  71. package/dist/lib/tools/ripgrep.js +217 -0
  72. package/package.json +1 -1
  73. package/src/commander.ts +11 -0
  74. package/src/commands/callers.ts +90 -26
  75. package/src/commands/doctor.ts +95 -2
  76. package/src/commands/grep.ts +737 -113
  77. package/src/commands/init.ts +138 -17
  78. package/src/commands/workflow.ts +132 -0
  79. package/src/core/agents/cli.ts +11 -4
  80. package/src/core/agents/rules/claim-conflict.ts +26 -1
  81. package/src/core/agents/rules/stop-hook.ts +17 -0
  82. package/src/core/agents/state/heartbeat-projector.ts +13 -1
  83. package/src/core/agents/state/heartbeat-writer.ts +30 -5
  84. package/src/core/config.ts +47 -0
  85. package/src/core/hooks/cli.ts +30 -3
  86. package/src/core/hooks/events/schema.ts +4 -0
  87. package/src/core/hooks/harness/events.ts +8 -0
  88. package/src/core/hooks/resolve/coord-root.ts +28 -6
  89. package/src/core/workflow/billing.ts +146 -0
  90. package/src/core/workflow/child-env.ts +47 -0
  91. package/src/core/workflow/engine.ts +394 -0
  92. package/src/core/workflow/harnesses.ts +38 -0
  93. package/src/core/workflow/spawn-claude.ts +99 -0
  94. package/src/core/workflow/spawn-codex.ts +74 -0
  95. package/src/core/workflow/spawn-cursor.ts +89 -0
  96. package/src/core/workflow/types.ts +153 -0
  97. package/src/core/workflow/validate.ts +75 -0
  98. package/src/lib/tools/ripgrep.ts +244 -0
@@ -1,12 +1,37 @@
1
1
  import { spawn } from "node:child_process";
2
+ import { readFile } from "node:fs/promises";
3
+ import { isAbsolute, join } from "node:path";
4
+ import { resolveSearchEngine } from "../lib/tools/ripgrep.js";
2
5
  /**
3
- * `grep`: monorepo-aware code search. Thin wrapper over grep -rn with
4
- * smart default excludes (skip dist/.next/node_modules/.git/...), repo
5
- * scoping (`--repo <name>` or `--all-repos`), and language presets.
6
+ * `grep`: monorepo-aware code search. Prefers ripgrep (`rg`) when it is on
7
+ * PATH and falls back to GNU/BSD grep transparently — both engines are driven
8
+ * with equivalent flags and their output is parsed into the same envelope, so
9
+ * results are identical (pinned by tests/unit/grep-engine.test.ts).
10
+ * Smart default excludes (skip dist/.next/node_modules/.git/...), repo
11
+ * scoping (`--repo <name>` or `--all-repos`), language presets, file-level
12
+ * boolean composition (`--and` / `--without`), and context (`-C`/`-A`/`-B`).
6
13
  *
7
- * Default behavior matches grep's "regex" semantics (`-E` extended). Use
8
- * `-F` / `--literal` to pin to literal-string mode. Output is line-oriented
9
- * `file:line:content` in TTY mode, `{rows, total, truncated}` in --json mode.
14
+ * Engine selection: `HARNERY_GREP_ENGINE=rg|grep` forces one; otherwise `rg`
15
+ * is resolved (managed install, PATH, or opt-in auto-provision; see
16
+ * resolveEngine) and used when available. Repos are searched in parallel, and
17
+ * in `--all-repos` mode the parent scan prunes submodule directories so each
18
+ * match is attributed to exactly one repo.
19
+ *
20
+ * Output framing: content searches request the NUL filename delimiter
21
+ * (`--null` on both engines — the long spelling, because BSD grep repurposes
22
+ * `-Z` for decompression) so filename boundaries are never inferred from
23
+ * punctuation. Context lines are NOT requested from the engines: matches are
24
+ * selected (and budgeted) first, then context windows are materialized from
25
+ * one file read per selected file, which keeps `--limit` semantics exact and
26
+ * both engines byte-identical.
27
+ *
28
+ * Default behavior matches extended-regex semantics (`grep -E` / ripgrep's
29
+ * default). Use `-F` / `--literal` to pin to literal-string mode. Output is
30
+ * line-oriented `file:line:content` in TTY mode (context rows render
31
+ * grep-style as `file-line-content`); `--json` emits the full GrepResult
32
+ * envelope `{ pattern, mode, engine, and_patterns, without_patterns, repos,
33
+ * total_matches, total_files, truncated, elapsed_ms }`. Matches are sorted
34
+ * (file, then line) for stable output across runs and engines.
10
35
  */
11
36
  const DEFAULT_EXCLUDE_DIRS = [
12
37
  ".git",
@@ -47,19 +72,28 @@ const LANG_GLOBS = {
47
72
  export function registerGrepCommand(program, emit, context) {
48
73
  program
49
74
  .command("grep <pattern> [paths...]")
50
- .description("Monorepo-aware code search. Skips dist/.next/node_modules/.git/... by default. " +
51
- "Use --repo, --all-repos, --lang for scoping. Regex by default; -F for literal.")
75
+ .description("Monorepo-aware code search (ripgrep when available, GNU grep fallback). " +
76
+ "Skips dist/.next/node_modules/.git/... by default. " +
77
+ "Use --repo, --all-repos, --lang for scoping; --and/--without for file-level composition. " +
78
+ "Regex by default; -F for literal.")
52
79
  .option("--repo <name>", "Scope to one submodule (`.` = parent repo root)")
53
80
  .option("--all-repos", "Search parent + every submodule")
54
- .option("--lang <lang>", `File type preset (${Object.keys(LANG_GLOBS).join(", ")})`)
81
+ .option("--lang <lang>", `File type preset, repeatable or comma-separated (${Object.keys(LANG_GLOBS).join(", ")})`, collect, [])
55
82
  .option("-i, --ignore-case", "Case-insensitive match")
56
83
  .option("-w, --whole-word", "Match whole words only")
57
84
  .option("-F, --literal", "Treat <pattern> as a literal string (no regex)")
58
85
  .option("-l, --files-only", "Only print file names containing a match")
86
+ .option("--files", "Filename search: treat <pattern> as a filename glob and list matching files " +
87
+ "(rg --files when available, POSIX find fallback)")
59
88
  .option("-c, --count", "Print match count per file (suppresses content)")
60
89
  .option("-C, --context <n>", "Print N lines of context around each match", "0")
90
+ .option("-A, --after-context <n>", "Print N lines after each match (overrides -C's after side)")
91
+ .option("-B, --before-context <n>", "Print N lines before each match (overrides -C's before side)")
61
92
  .option("--max-count <n>", "Stop after N matches per file")
62
93
  .option("--limit <n>", "Truncate output to N matches total")
94
+ .option("--and <pattern>", "Only show files that ALSO contain this pattern (file-level, repeatable)", collect, [])
95
+ .option("--without <pattern>", "Drop files that contain this pattern (file-level, repeatable)", collect, [])
96
+ .option("-q, --quiet", "No output; exit 0 if any match exists, 1 if none")
63
97
  .option("--include <glob>", "Extra --include glob (repeatable)", collect, [])
64
98
  .option("--exclude <glob>", "Extra --exclude glob (repeatable)", collect, [])
65
99
  .option("--no-default-excludes", "Disable the default skip list (node_modules, dist, etc.)")
@@ -67,6 +101,12 @@ export function registerGrepCommand(program, emit, context) {
67
101
  .action(async (pattern, paths, opts) => {
68
102
  try {
69
103
  const result = await runGrep(pattern, paths, opts, context);
104
+ if (opts.quiet) {
105
+ // No output by contract; grep-conventional status. exitCode (not
106
+ // process.exit) so an embedding host isn't terminated mid-flush.
107
+ process.exitCode = result.total_matches > 0 ? 0 : 1;
108
+ return;
109
+ }
70
110
  if (opts.json) {
71
111
  emit.config({ format: "json" });
72
112
  emit.data(result);
@@ -83,41 +123,177 @@ export function registerGrepCommand(program, emit, context) {
83
123
  function collect(value, prev) {
84
124
  return [...prev, value];
85
125
  }
86
- async function runGrep(pattern, paths, opts, context) {
126
+ /**
127
+ * Parse a strictly-decimal integer option. Rejects empty, negative, signed,
128
+ * fractional, and trailing-junk values before any engine is spawned.
129
+ */
130
+ function parseIntOpt(flag, raw, min) {
131
+ if (!/^\d+$/.test(raw.trim())) {
132
+ throw new Error(`${flag} expects a non-negative integer, got "${raw}"`);
133
+ }
134
+ const n = Number.parseInt(raw.trim(), 10);
135
+ if (n < min)
136
+ throw new Error(`${flag} must be >= ${min}, got ${n}`);
137
+ return n;
138
+ }
139
+ function normalizeLangs(lang) {
140
+ const rawValues = typeof lang === "string" ? [lang] : (lang ?? []);
141
+ const keys = [];
142
+ for (const raw of rawValues) {
143
+ for (const piece of raw.split(",")) {
144
+ const key = piece.trim();
145
+ if (key === "")
146
+ continue;
147
+ if (!LANG_GLOBS[key]) {
148
+ throw new Error(`unknown --lang "${key}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
149
+ }
150
+ if (!keys.includes(key))
151
+ keys.push(key);
152
+ }
153
+ }
154
+ if (keys.length === 0)
155
+ return undefined;
156
+ const globs = [];
157
+ for (const key of keys) {
158
+ const langGlobs = LANG_GLOBS[key];
159
+ if (!langGlobs)
160
+ continue;
161
+ for (const g of langGlobs)
162
+ if (!globs.includes(g))
163
+ globs.push(g);
164
+ }
165
+ return globs;
166
+ }
167
+ /** Validate numerics, resolve context sides, expand langs, enforce the compatibility matrix. */
168
+ function normalizeOpts(opts) {
169
+ const limit = opts.limit ? parseIntOpt("--limit", opts.limit, 1) : Number.POSITIVE_INFINITY;
170
+ const maxCount = opts.maxCount ? parseIntOpt("--max-count", opts.maxCount, 1) : undefined;
171
+ const c = opts.context !== undefined ? parseIntOpt("-C/--context", opts.context, 0) : 0;
172
+ const before = opts.beforeContext !== undefined
173
+ ? parseIntOpt("-B/--before-context", opts.beforeContext, 0)
174
+ : c;
175
+ const after = opts.afterContext !== undefined ? parseIntOpt("-A/--after-context", opts.afterContext, 0) : c;
176
+ const andPatterns = opts.and ?? [];
177
+ const withoutPatterns = opts.without ?? [];
178
+ const contextActive = before > 0 || after > 0;
179
+ if (opts.files) {
180
+ // Filename mode lists files by name glob; content-search flags make no
181
+ // sense here — reject loudly rather than silently ignoring them.
182
+ const incompatible = [
183
+ [normalizeLangs(opts.lang)?.length, "--lang"],
184
+ [opts.count, "-c/--count"],
185
+ [opts.wholeWord, "-w/--whole-word"],
186
+ [opts.literal, "-F/--literal"],
187
+ [opts.maxCount, "--max-count"],
188
+ [opts.include?.length, "--include"],
189
+ [contextActive ? true : undefined, "-C/-A/-B context"],
190
+ [andPatterns.length ? true : undefined, "--and"],
191
+ [withoutPatterns.length ? true : undefined, "--without"],
192
+ ];
193
+ for (const [set, flag] of incompatible) {
194
+ if (set)
195
+ throw new Error(`${flag} does not apply to --files (filename glob) mode`);
196
+ }
197
+ }
198
+ if (contextActive) {
199
+ const rejected = [
200
+ [opts.filesOnly, "-l/--files-only"],
201
+ [opts.count, "-c/--count"],
202
+ [opts.quiet, "-q/--quiet"],
203
+ ];
204
+ for (const [set, flag] of rejected) {
205
+ if (set)
206
+ throw new Error(`-C/-A/-B context does not combine with ${flag}`);
207
+ }
208
+ }
209
+ if (opts.quiet) {
210
+ const rejected = [
211
+ [opts.json, "--json"],
212
+ [opts.filesOnly, "-l/--files-only"],
213
+ [opts.count, "-c/--count"],
214
+ [opts.limit, "--limit"],
215
+ ];
216
+ for (const [set, flag] of rejected) {
217
+ if (set)
218
+ throw new Error(`-q/--quiet does not combine with ${flag}`);
219
+ }
220
+ }
221
+ return {
222
+ // -q only needs existence: one accepted primary row settles the exit code.
223
+ limit: opts.quiet ? 1 : limit,
224
+ maxCount,
225
+ before,
226
+ after,
227
+ langGlobs: normalizeLangs(opts.lang),
228
+ andPatterns,
229
+ withoutPatterns,
230
+ };
231
+ }
232
+ /** Exported for tests (not part of the package exports map). */
233
+ export async function runGrep(pattern, paths, opts, context) {
87
234
  if (!pattern)
88
235
  throw new Error("pattern required");
236
+ const norm = normalizeOpts(opts);
89
237
  const started = Date.now();
238
+ const { engine, rgBin } = await resolveSearchEngine("grep");
90
239
  const repos = resolveRepos(opts, context);
91
- const limit = opts.limit ? Number.parseInt(opts.limit, 10) : Number.POSITIVE_INFINITY;
240
+ // Host-injected default excludes (generated mirrors, vendored trees, ...)
241
+ // ride the same --no-default-excludes gate as the built-in list.
242
+ const hostExcludeDirs = opts.noDefaultExcludes ? [] : (context?.grepExcludeDirs ?? []);
243
+ // All repos are searched concurrently; each collects at most limit+1
244
+ // accepted rows (one row of lookahead, so "exactly N results" is
245
+ // distinguishable from "more than N"), then the global budget is applied
246
+ // in repo order below so `--limit` semantics stay deterministic.
247
+ const acceptLimit = Number.isFinite(norm.limit) ? norm.limit + 1 : Number.POSITIVE_INFINITY;
248
+ const perRepo = await Promise.all(repos.map((repo) => {
249
+ // In --all-repos mode the parent scan prunes submodule dirs — each
250
+ // submodule gets its own scoped scan, so descending from the parent
251
+ // would double-scan and double-report every submodule match. This
252
+ // pruning is correctness (one repo owns each match), so it applies
253
+ // even when --no-default-excludes is set.
254
+ const dedupeDirs = opts.allRepos && repo.name === "parent" ? (context?.submodules ?? []) : [];
255
+ const extraDirs = [...hostExcludeDirs, ...dedupeDirs];
256
+ return runGrepInRepo(pattern, paths, opts, norm, repo.cwd, repo.name, acceptLimit, engine, rgBin, extraDirs);
257
+ }));
92
258
  const allRepoResults = [];
93
259
  let totalMatches = 0;
94
260
  const filesSeen = new Set();
95
261
  let truncated = false;
96
- for (const repo of repos) {
97
- if (truncated)
98
- break;
99
- const repoLimit = Number.isFinite(limit)
100
- ? Math.max(0, limit - totalMatches)
101
- : Number.POSITIVE_INFINITY;
102
- if (repoLimit === 0) {
103
- truncated = true;
104
- allRepoResults.push({ name: repo.name, cwd: repo.cwd, matches: [], truncated: true });
105
- break;
106
- }
107
- const matches = await runGrepInRepo(pattern, paths, opts, repo.cwd, repo.name, repoLimit);
108
- let repoTruncated = false;
109
- if (Number.isFinite(repoLimit) && matches.length >= repoLimit) {
110
- repoTruncated = true;
262
+ let budget = norm.limit;
263
+ for (let i = 0; i < repos.length; i++) {
264
+ const repo = repos[i];
265
+ if (!repo)
266
+ continue;
267
+ const collected = perRepo[i] ?? [];
268
+ collected.sort((a, b) => (a.file < b.file ? -1 : a.file > b.file ? 1 : a.line - b.line));
269
+ const take = Number.isFinite(budget)
270
+ ? Math.min(collected.length, Math.max(0, budget))
271
+ : collected.length;
272
+ let matches = collected.slice(0, take);
273
+ // truncated only when an accepted primary row was actually omitted —
274
+ // the lookahead row (or a later-repo surplus) is the proof.
275
+ const repoTruncated = collected.length > take;
276
+ if (repoTruncated)
111
277
  truncated = true;
112
- }
278
+ if (Number.isFinite(budget))
279
+ budget -= take;
113
280
  totalMatches += matches.length;
114
281
  for (const m of matches)
115
282
  filesSeen.add(`${repo.name}/${m.file}`);
283
+ // Context is materialized only for the SELECTED rows, after budgeting,
284
+ // so context rows never consume the limit and trailing context is never
285
+ // cut off by an engine kill.
286
+ if (!opts.files && !opts.filesOnly && !opts.count && (norm.before > 0 || norm.after > 0)) {
287
+ matches = await materializeContext(matches, repo.cwd, norm.before, norm.after);
288
+ }
116
289
  allRepoResults.push({ name: repo.name, cwd: repo.cwd, matches, truncated: repoTruncated });
117
290
  }
118
291
  return {
119
292
  pattern,
120
- mode: opts.literal ? "literal" : "regex",
293
+ mode: opts.files ? "files" : opts.literal ? "literal" : "regex",
294
+ engine,
295
+ and_patterns: norm.andPatterns,
296
+ without_patterns: norm.withoutPatterns,
121
297
  repos: allRepoResults,
122
298
  total_matches: totalMatches,
123
299
  total_files: filesSeen.size,
@@ -155,8 +331,71 @@ function resolveRepos(opts, context) {
155
331
  }
156
332
  return [{ name: "cwd", cwd: process.cwd() }];
157
333
  }
158
- async function runGrepInRepo(pattern, paths, opts, cwd, repoName, limit) {
159
- const args = ["-rn", "--color=never", "-I"]; // -I: skip binary files
334
+ async function runGrepInRepo(pattern, paths, opts, norm, cwd, repoName, acceptLimit, engine, rgBin, extraExcludeDirs) {
335
+ // File-level boolean composition: build the complete auxiliary file sets
336
+ // FIRST (no limit — a partial set can't prove absence), then feed the
337
+ // primary scan an in-stream predicate. Running the primary with its limit
338
+ // and filtering afterwards would let non-qualifying early hits consume the
339
+ // budget and hide later valid results.
340
+ let predicate;
341
+ if (norm.andPatterns.length > 0 || norm.withoutPatterns.length > 0) {
342
+ const requiredSets = [];
343
+ // Sequential, not parallel: repeated flags must not multiply peak child
344
+ // process concurrency by repos x patterns. Repo workers stay parallel.
345
+ for (const p of norm.andPatterns) {
346
+ const files = await runFileSetScan(p, paths, opts, norm, cwd, engine, rgBin, extraExcludeDirs);
347
+ if (files.size === 0)
348
+ return []; // intersection is already empty
349
+ requiredSets.push(files);
350
+ }
351
+ const forbidden = new Set();
352
+ for (const p of norm.withoutPatterns) {
353
+ const files = await runFileSetScan(p, paths, opts, norm, cwd, engine, rgBin, extraExcludeDirs);
354
+ for (const f of files)
355
+ forbidden.add(f);
356
+ }
357
+ predicate = (file) => requiredSets.every((s) => s.has(file)) && !forbidden.has(file);
358
+ }
359
+ const [bin, args, decodeMode] = buildEngineInvocation(pattern, paths, opts, norm, engine, rgBin, extraExcludeDirs, false);
360
+ let matches = await execSearch(bin, args, cwd, acceptLimit, decodeMode, predicate, false);
361
+ // -c mode: GNU grep prints a `path:0` row for every searched file; ripgrep
362
+ // omits zero-count rows. Filter zeros on both engines so output matches.
363
+ if (opts.count)
364
+ matches = matches.filter((m) => m.line > 0);
365
+ return matches.map((m) => ({ ...m, repo: repoName }));
366
+ }
367
+ /** Membership scan for one --and/--without pattern: complete files-only set, strict failure. */
368
+ async function runFileSetScan(pattern, paths, opts, norm, cwd, engine, rgBin, extraExcludeDirs) {
369
+ const [bin, args, decodeMode] = buildEngineInvocation(pattern, paths, opts, norm, engine, rgBin, extraExcludeDirs, true);
370
+ const rows = await execSearch(bin, args, cwd, Number.POSITIVE_INFINITY, decodeMode, undefined, true);
371
+ return new Set(rows.map((r) => r.file));
372
+ }
373
+ /**
374
+ * Build one engine invocation. `membership: true` forces files-only mode for
375
+ * --and/--without set scans (match-shaping + scope flags shared with the
376
+ * primary; output flags not applied).
377
+ */
378
+ function buildEngineInvocation(pattern, paths, opts, norm, engine, rgBin, extraExcludeDirs, membership) {
379
+ if (opts.files) {
380
+ // Filename mode keeps its existing rg/find framing (plain lines).
381
+ return engine === "rg"
382
+ ? [rgBin, buildRgFilesArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"]
383
+ : ["find", buildFindArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"];
384
+ }
385
+ const filesOnly = membership || Boolean(opts.filesOnly);
386
+ const count = !membership && Boolean(opts.count);
387
+ const decodeMode = filesOnly ? "filesOnly" : count ? "count" : "content";
388
+ const args = engine === "rg"
389
+ ? buildRgArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count })
390
+ : buildGrepArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count });
391
+ return [engine === "rg" ? rgBin : "grep", args, decodeMode];
392
+ }
393
+ function buildGrepArgs(pattern, paths, opts, norm, extraExcludeDirs, shape) {
394
+ // --null (NOT -Z: BSD grep repurposes -Z for decompression): NUL after the
395
+ // filename, so paths containing colons/dashes can't confuse the decoder.
396
+ // -I: skip binary files. Context is never requested from the engine — see
397
+ // materializeContext.
398
+ const args = ["-rn", "-H", "--null", "--color=never", "-I"];
160
399
  if (opts.ignoreCase)
161
400
  args.push("-i");
162
401
  if (opts.wholeWord)
@@ -165,32 +404,24 @@ async function runGrepInRepo(pattern, paths, opts, cwd, repoName, limit) {
165
404
  args.push("-F");
166
405
  else
167
406
  args.push("-E");
168
- if (opts.filesOnly)
407
+ if (shape.filesOnly)
169
408
  args.push("-l");
170
- if (opts.count)
409
+ if (shape.count)
171
410
  args.push("-c");
172
- const contextN = Number.parseInt(opts.context ?? "0", 10);
173
- if (Number.isFinite(contextN) && contextN > 0)
174
- args.push(`-C${contextN}`);
175
- if (opts.maxCount) {
176
- const n = Number.parseInt(opts.maxCount, 10);
177
- if (Number.isFinite(n) && n > 0)
178
- args.push(`-m${n}`);
179
- }
411
+ if (norm.maxCount !== undefined)
412
+ args.push(`-m${norm.maxCount}`);
180
413
  if (!opts.noDefaultExcludes) {
181
414
  for (const d of DEFAULT_EXCLUDE_DIRS)
182
415
  args.push(`--exclude-dir=${d}`);
183
416
  for (const f of DEFAULT_EXCLUDE_FILES)
184
417
  args.push(`--exclude=${f}`);
185
418
  }
419
+ for (const d of extraExcludeDirs)
420
+ args.push(`--exclude-dir=${d}`);
186
421
  for (const e of opts.exclude ?? [])
187
422
  args.push(`--exclude=${e}`);
188
- const langGlobs = opts.lang ? LANG_GLOBS[opts.lang] : undefined;
189
- if (opts.lang && !langGlobs) {
190
- throw new Error(`unknown --lang "${opts.lang}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
191
- }
192
- if (langGlobs)
193
- for (const g of langGlobs)
423
+ if (norm.langGlobs)
424
+ for (const g of norm.langGlobs)
194
425
  args.push(`--include=${g}`);
195
426
  for (const g of opts.include ?? [])
196
427
  args.push(`--include=${g}`);
@@ -199,36 +430,230 @@ async function runGrepInRepo(pattern, paths, opts, cwd, repoName, limit) {
199
430
  args.push(...paths);
200
431
  else
201
432
  args.push(".");
202
- const matches = await execGrep(args, cwd, limit);
203
- return matches.map((m) => ({ ...m, repo: repoName }));
433
+ return args;
434
+ }
435
+ function buildRgArgs(pattern, paths, opts, norm, extraExcludeDirs, shape) {
436
+ // --hidden --no-ignore: match GNU grep's semantics (search dotdirs, ignore
437
+ // .gitignore) so the only filters are the explicit exclude lists.
438
+ // --no-config: a user's ripgrep config file must not skew results.
439
+ // --with-filename: rg drops the file prefix for a single explicit file arg,
440
+ // which would break the shared decoder.
441
+ // --null: NUL after the filename (same spelling as GNU/BSD grep's long flag).
442
+ const args = [
443
+ "-n",
444
+ "--no-heading",
445
+ "--with-filename",
446
+ "--null",
447
+ "--color=never",
448
+ "--no-config",
449
+ "--hidden",
450
+ "--no-ignore",
451
+ ];
452
+ if (opts.ignoreCase)
453
+ args.push("-i");
454
+ if (opts.wholeWord)
455
+ args.push("-w");
456
+ if (opts.literal)
457
+ args.push("-F");
458
+ if (shape.filesOnly)
459
+ args.push("-l");
460
+ if (shape.count)
461
+ args.push("-c");
462
+ if (norm.maxCount !== undefined)
463
+ args.push(`-m${norm.maxCount}`);
464
+ // Gitignore-style globs: a bare name matches (and prunes) at any depth,
465
+ // mirroring grep's --exclude-dir / --exclude basename semantics. ORDER
466
+ // MATTERS: rg globs are last-match-wins, so positives (includes/lang) go
467
+ // first and negatives (excludes) last — otherwise `--include '*.md'` would
468
+ // re-include a .md file inside an excluded node_modules/. grep's
469
+ // --exclude-dir always beats --include, so this keeps the engines aligned.
470
+ if (norm.langGlobs)
471
+ for (const g of norm.langGlobs)
472
+ args.push(`--glob=${g}`);
473
+ for (const g of opts.include ?? [])
474
+ args.push(`--glob=${g}`);
475
+ if (!opts.noDefaultExcludes) {
476
+ for (const d of DEFAULT_EXCLUDE_DIRS)
477
+ args.push(`--glob=!${d}`);
478
+ for (const f of DEFAULT_EXCLUDE_FILES)
479
+ args.push(`--glob=!${f}`);
480
+ }
481
+ for (const d of extraExcludeDirs)
482
+ args.push(`--glob=!${d}`);
483
+ for (const e of opts.exclude ?? [])
484
+ args.push(`--glob=!${e}`);
485
+ args.push("--", pattern);
486
+ if (paths.length > 0)
487
+ args.push(...paths);
488
+ else
489
+ args.push(".");
490
+ return args;
491
+ }
492
+ /**
493
+ * Filename mode via ripgrep: `--files` lists files; a single positive glob
494
+ * acts as a whitelist, negative globs prune. `--iglob` gives -i semantics.
495
+ */
496
+ function buildRgFilesArgs(pattern, paths, opts, extraExcludeDirs) {
497
+ const args = ["--files", "--hidden", "--no-ignore", "--no-config", "--color=never"];
498
+ // Positive pattern FIRST, negatives last (rg globs are last-match-wins;
499
+ // excludes must beat the pattern — see the ordering note in buildRgArgs).
500
+ args.push(`${opts.ignoreCase ? "--iglob" : "--glob"}=${pattern}`);
501
+ if (!opts.noDefaultExcludes) {
502
+ for (const d of DEFAULT_EXCLUDE_DIRS)
503
+ args.push(`--glob=!${d}`);
504
+ for (const f of DEFAULT_EXCLUDE_FILES)
505
+ args.push(`--glob=!${f}`);
506
+ }
507
+ for (const d of extraExcludeDirs)
508
+ args.push(`--glob=!${d}`);
509
+ for (const e of opts.exclude ?? [])
510
+ args.push(`--glob=!${e}`);
511
+ if (paths.length > 0)
512
+ args.push(...paths);
513
+ else
514
+ args.push(".");
515
+ return args;
516
+ }
517
+ /**
518
+ * Filename mode via POSIX find (the no-ripgrep fallback):
519
+ * find <paths> ( -name d1 -o -name d2 ... ) -prune -o -type f -name <glob> [! -name <ex>]... -print
520
+ * Sticks to POSIX operators (`(`, `-o`, `!`) so BSD/macOS find behaves the same.
521
+ */
522
+ function buildFindArgs(pattern, paths, opts, extraExcludeDirs) {
523
+ const args = paths.length > 0 ? [...paths] : ["."];
524
+ const pruneDirs = [...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_DIRS), ...extraExcludeDirs];
525
+ if (pruneDirs.length > 0) {
526
+ args.push("(");
527
+ pruneDirs.forEach((d, i) => {
528
+ if (i > 0)
529
+ args.push("-o");
530
+ args.push("-name", d);
531
+ });
532
+ args.push(")", "-prune", "-o");
533
+ }
534
+ args.push("-type", "f", opts.ignoreCase ? "-iname" : "-name", pattern);
535
+ const fileExcludes = [
536
+ ...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_FILES),
537
+ ...(opts.exclude ?? []),
538
+ ];
539
+ for (const e of fileExcludes)
540
+ args.push("!", "-name", e);
541
+ args.push("-print");
542
+ return args;
204
543
  }
205
- function execGrep(args, cwd, limit) {
544
+ /**
545
+ * Streaming decoder for NUL-framed engine output. Exported for direct
546
+ * chunk-boundary tests. Record shapes (both engines, pinned by the parity
547
+ * suite):
548
+ * content: <path>NUL<line>:<text>LF
549
+ * count: <path>NUL<count>LF
550
+ * filesOnly: <path>NUL (NUL is the record terminator)
551
+ * plainLines: <path>LF (files mode: rg --files / find)
552
+ *
553
+ * Operates on Buffers so a chunk boundary can fall anywhere — including
554
+ * inside a multi-byte UTF-8 sequence — without corrupting a record: bytes
555
+ * are only decoded to strings once a full record is framed.
556
+ */
557
+ export class NulDecoder {
558
+ mode;
559
+ buffer = Buffer.alloc(0);
560
+ constructor(mode) {
561
+ this.mode = mode;
562
+ }
563
+ push(chunk) {
564
+ this.buffer = this.buffer.length === 0 ? chunk : Buffer.concat([this.buffer, chunk]);
565
+ const rows = [];
566
+ if (this.mode === "filesOnly") {
567
+ let nul = this.buffer.indexOf(0);
568
+ while (nul >= 0) {
569
+ const path = this.buffer.subarray(0, nul).toString("utf8");
570
+ this.buffer = this.buffer.subarray(nul + 1);
571
+ // Engines may still newline-separate NUL-terminated records in edge
572
+ // configurations; a bare leading LF is framing residue, not a path.
573
+ const cleaned = path.startsWith("\n") ? path.slice(1) : path;
574
+ if (cleaned.length > 0)
575
+ rows.push({ file: normalizeFile(cleaned), line: 0, text: "" });
576
+ nul = this.buffer.indexOf(0);
577
+ }
578
+ return rows;
579
+ }
580
+ let nl = this.buffer.indexOf(0x0a);
581
+ while (nl >= 0) {
582
+ const record = this.buffer.subarray(0, nl);
583
+ this.buffer = this.buffer.subarray(nl + 1);
584
+ const row = this.decodeRecord(record);
585
+ if (row)
586
+ rows.push(row);
587
+ nl = this.buffer.indexOf(0x0a);
588
+ }
589
+ return rows;
590
+ }
591
+ /** Flush a trailing unterminated record (engine killed mid-write, or no final LF). */
592
+ flush() {
593
+ if (this.buffer.length === 0)
594
+ return [];
595
+ const record = this.buffer;
596
+ this.buffer = Buffer.alloc(0);
597
+ if (this.mode === "filesOnly") {
598
+ // A partial filesOnly record has no terminating NUL — an engine killed
599
+ // mid-path would yield a corrupt name; drop it.
600
+ return [];
601
+ }
602
+ const row = this.decodeRecord(record);
603
+ return row ? [row] : [];
604
+ }
605
+ decodeRecord(record) {
606
+ if (this.mode === "plainLines") {
607
+ const path = record.toString("utf8");
608
+ if (path.length === 0)
609
+ return null;
610
+ return { file: normalizeFile(path), line: 0, text: "" };
611
+ }
612
+ const nul = record.indexOf(0);
613
+ if (nul < 0)
614
+ return null; // not a NUL-framed record (stray engine chatter)
615
+ const file = normalizeFile(record.subarray(0, nul).toString("utf8"));
616
+ const rest = record.subarray(nul + 1).toString("utf8");
617
+ if (this.mode === "count") {
618
+ const count = Number.parseInt(rest, 10);
619
+ if (!Number.isFinite(count))
620
+ return null;
621
+ return { file, line: count, text: "" };
622
+ }
623
+ // content: <line>:<text>. The engines never emit context/separator rows
624
+ // (context is materialized from file reads), so `:` is always present.
625
+ const colon = rest.indexOf(":");
626
+ if (colon < 0)
627
+ return null;
628
+ const lineNum = Number.parseInt(rest.slice(0, colon), 10);
629
+ if (!Number.isFinite(lineNum))
630
+ return null;
631
+ return { file, line: lineNum, text: rest.slice(colon + 1) };
632
+ }
633
+ }
634
+ function execSearch(bin, args, cwd, acceptLimit, decodeMode, filePredicate, strict) {
206
635
  return new Promise((resolveP, reject) => {
207
- const proc = spawn("grep", args, { cwd, stdio: ["ignore", "pipe", "pipe"] });
636
+ const proc = spawn(bin, args, { cwd, stdio: ["ignore", "pipe", "pipe"] });
637
+ const decoder = new NulDecoder(decodeMode);
208
638
  const matches = [];
209
- let buffer = "";
210
639
  let stderr = "";
211
- let truncated = false;
212
- proc.stdout.setEncoding("utf8");
640
+ let killed = false;
641
+ const accept = (rows) => {
642
+ for (const row of rows) {
643
+ if (filePredicate && !filePredicate(row.file))
644
+ continue;
645
+ matches.push({ ...row, repo: "", kind: "match" });
646
+ if (Number.isFinite(acceptLimit) && matches.length >= acceptLimit)
647
+ return true;
648
+ }
649
+ return false;
650
+ };
213
651
  proc.stdout.on("data", (chunk) => {
214
- if (truncated)
652
+ if (killed)
215
653
  return;
216
- buffer += chunk;
217
- let nl = buffer.indexOf("\n");
218
- while (nl >= 0) {
219
- const line = buffer.slice(0, nl);
220
- buffer = buffer.slice(nl + 1);
221
- if (line.length !== 0) {
222
- const parsed = parseGrepLine(line);
223
- if (parsed)
224
- matches.push(parsed);
225
- if (Number.isFinite(limit) && matches.length >= limit) {
226
- truncated = true;
227
- proc.kill("SIGTERM");
228
- return;
229
- }
230
- }
231
- nl = buffer.indexOf("\n");
654
+ if (accept(decoder.push(chunk))) {
655
+ killed = true;
656
+ proc.kill("SIGTERM");
232
657
  }
233
658
  });
234
659
  proc.stderr.setEncoding("utf8");
@@ -237,47 +662,99 @@ function execGrep(args, cwd, limit) {
237
662
  });
238
663
  proc.on("error", reject);
239
664
  proc.on("close", (code) => {
240
- if (buffer.length > 0 && !truncated) {
241
- const parsed = parseGrepLine(buffer);
242
- if (parsed)
243
- matches.push(parsed);
244
- }
245
- // grep exits 1 on "no matches": normal, not an error.
246
- if (code !== null && code !== 0 && code !== 1 && !truncated) {
247
- reject(new Error(`grep exited ${code}: ${stderr.trim() || "(no stderr)"}`));
665
+ if (!killed)
666
+ accept(decoder.flush());
667
+ // Exit 1 is "no matches" on both engines: normal, not an error.
668
+ // Primary scans tolerate exit 2 with matches collected (an unreadable
669
+ // file mid-walk): surface the results rather than throwing them away.
670
+ // STRICT scans (--and/--without membership) must reject on ANY partial
671
+ // failure — an incomplete set cannot prove that a file lacks a pattern.
672
+ const failed = code !== null && code !== 0 && code !== 1 && !killed;
673
+ if (failed && (strict || matches.length === 0)) {
674
+ reject(new Error(`${bin} exited ${code}: ${stderr.trim() || "(no stderr)"}`));
248
675
  return;
249
676
  }
250
677
  resolveP(matches);
251
678
  });
252
679
  });
253
680
  }
254
- /** Parse `path:line:content` from grep -n output. Returns null for ambiguous (no line number) lines. */
255
- function parseGrepLine(line) {
256
- // First colon ends path; second colon ends line number (if numeric).
257
- const firstColon = line.indexOf(":");
258
- if (firstColon < 0) {
259
- // -l mode: just a path. Encode as line=0, text="".
260
- return { file: line, line: 0, text: "" };
261
- }
262
- const after = line.slice(firstColon + 1);
263
- const secondColon = after.indexOf(":");
264
- if (secondColon < 0) {
265
- // -c mode: `path:count`.
266
- const count = Number.parseInt(after, 10);
267
- if (Number.isFinite(count))
268
- return { file: line.slice(0, firstColon), line: count, text: "" };
269
- return { file: line.slice(0, firstColon), line: 0, text: after };
681
+ /**
682
+ * Materialize -C/-A/-B context for already-selected match rows: one file
683
+ * read per selected file, intervals clamped + merged, match rows keep the
684
+ * engine-captured text, other window lines become kind:"context" rows.
685
+ * An unreadable file (deleted since the scan — the accepted TOCTOU window)
686
+ * degrades to its match rows without context rather than failing the search.
687
+ */
688
+ async function materializeContext(selected, cwd, before, after) {
689
+ const byFile = new Map();
690
+ for (const m of selected) {
691
+ const rows = byFile.get(m.file);
692
+ if (rows)
693
+ rows.push(m);
694
+ else
695
+ byFile.set(m.file, [m]);
270
696
  }
271
- const lineNum = Number.parseInt(after.slice(0, secondColon), 10);
272
- if (!Number.isFinite(lineNum)) {
273
- // Probably a context separator like `--`. Skip.
274
- return null;
697
+ const out = [];
698
+ for (const [file, rows] of byFile) {
699
+ const abs = isAbsolute(file) ? file : join(cwd, file);
700
+ let lines = null;
701
+ try {
702
+ const content = await readFile(abs, "utf8");
703
+ lines = content.split("\n");
704
+ if (lines.at(-1) === "")
705
+ lines.pop();
706
+ }
707
+ catch {
708
+ lines = null;
709
+ }
710
+ if (lines === null) {
711
+ out.push(...rows);
712
+ continue;
713
+ }
714
+ // Merge overlapping/adjacent [line-before, line+after] windows.
715
+ const intervals = rows
716
+ .map((m) => [
717
+ Math.max(1, m.line - before),
718
+ Math.min(lines.length, m.line + after),
719
+ ])
720
+ .sort((a, b) => a[0] - b[0]);
721
+ const merged = [];
722
+ for (const iv of intervals) {
723
+ const last = merged.at(-1);
724
+ if (last && iv[0] <= last[1] + 1)
725
+ last[1] = Math.max(last[1], iv[1]);
726
+ else
727
+ merged.push([iv[0], iv[1]]);
728
+ }
729
+ const matchByLine = new Map(rows.map((m) => [m.line, m]));
730
+ const emitted = new Set();
731
+ for (const [start, end] of merged) {
732
+ if (start > end)
733
+ continue;
734
+ for (let n = start; n <= end; n++) {
735
+ const m = matchByLine.get(n);
736
+ if (m) {
737
+ out.push(m);
738
+ emitted.add(m);
739
+ }
740
+ else {
741
+ const repo = rows[0]?.repo ?? "";
742
+ out.push({ repo, file, line: n, text: lines[n - 1] ?? "", kind: "context" });
743
+ }
744
+ }
745
+ }
746
+ // A match whose line exceeds the file's current length (file shrank
747
+ // since the scan) sits outside every clamped window — still emit it.
748
+ for (const m of rows) {
749
+ if (!emitted.has(m))
750
+ out.push(m);
751
+ }
275
752
  }
276
- return {
277
- file: line.slice(0, firstColon),
278
- line: lineNum,
279
- text: after.slice(secondColon + 1),
280
- };
753
+ return out;
754
+ }
755
+ /** Both engines may emit a leading `./` for the default path; strip it so output is engine-identical. */
756
+ function normalizeFile(file) {
757
+ return file.startsWith("./") ? file.slice(2) : file;
281
758
  }
282
759
  function renderResult(r, opts) {
283
760
  const lines = [];
@@ -285,33 +762,45 @@ function renderResult(r, opts) {
285
762
  for (const repo of r.repos) {
286
763
  if (repo.matches.length === 0)
287
764
  continue;
765
+ const primary = repo.matches.filter((m) => m.kind === "match").length;
288
766
  if (showRepoHeader) {
289
767
  lines.push("");
290
- lines.push(`── ${repo.name} (${repo.matches.length} match${repo.matches.length === 1 ? "" : "es"}) ──`);
768
+ lines.push(`── ${repo.name} (${primary} match${primary === 1 ? "" : "es"}) ──`);
291
769
  }
770
+ // Context mode: `--` between disjoint windows (file change or line gap).
771
+ const contextMode = repo.matches.some((x) => x.kind === "context");
772
+ let prev;
292
773
  for (const m of repo.matches) {
293
- if (opts.filesOnly) {
774
+ if (contextMode && prev && (prev.file !== m.file || m.line !== prev.line + 1)) {
775
+ lines.push("--");
776
+ }
777
+ if (opts.filesOnly || opts.files) {
294
778
  lines.push(m.file);
295
779
  }
296
780
  else if (opts.count) {
297
781
  lines.push(`${m.file}:${m.line}`);
298
782
  }
783
+ else if (m.kind === "context") {
784
+ lines.push(`${m.file}-${m.line}-${m.text}`);
785
+ }
299
786
  else {
300
787
  lines.push(`${m.file}:${m.line}:${m.text}`);
301
788
  }
789
+ prev = m;
302
790
  }
303
791
  if (repo.truncated) {
304
- lines.push(" (truncated; pass --limit to widen)");
792
+ lines.push(" (truncated; raise --limit)");
305
793
  }
306
794
  }
307
795
  if (r.total_matches === 0) {
308
- lines.push(`(no matches for /${r.pattern}/${r.mode === "literal" ? " (literal)" : ""})`);
796
+ const modeTag = r.mode === "literal" ? " (literal)" : r.mode === "files" ? " (files)" : "";
797
+ lines.push(`(no matches for /${r.pattern}/${modeTag})`);
309
798
  }
310
799
  else if (r.repos.length > 1 || r.truncated) {
311
800
  lines.push("");
312
801
  const summary = `${r.total_matches} match${r.total_matches === 1 ? "" : "es"} across ${r.total_files} file${r.total_files === 1 ? "" : "s"}`;
313
- const tail = r.truncated ? " (truncated; use --limit)" : "";
314
- lines.push(`${summary}${tail} (${r.elapsed_ms}ms)`);
802
+ const tail = r.truncated ? " (truncated; raise --limit)" : "";
803
+ lines.push(`${summary}${tail} (${r.elapsed_ms}ms, ${r.engine})`);
315
804
  }
316
805
  return lines.join("\n");
317
806
  }