harnery 0.7.1 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/commander.d.ts +9 -0
- package/dist/commander.d.ts.map +1 -1
- package/dist/commander.js +2 -0
- package/dist/commands/callers.d.ts.map +1 -1
- package/dist/commands/callers.js +69 -17
- package/dist/commands/doctor.d.ts.map +1 -1
- package/dist/commands/doctor.js +84 -2
- package/dist/commands/grep.d.ts +77 -0
- package/dist/commands/grep.d.ts.map +1 -1
- package/dist/commands/grep.js +598 -109
- package/dist/commands/init.d.ts +15 -5
- package/dist/commands/init.d.ts.map +1 -1
- package/dist/commands/init.js +125 -14
- package/dist/commands/workflow.d.ts +4 -0
- package/dist/commands/workflow.d.ts.map +1 -0
- package/dist/commands/workflow.js +89 -0
- package/dist/core/agents/cli.js +10 -3
- package/dist/core/agents/rules/claim-conflict.d.ts.map +1 -1
- package/dist/core/agents/rules/claim-conflict.js +26 -1
- package/dist/core/agents/rules/stop-hook.d.ts +8 -0
- package/dist/core/agents/rules/stop-hook.d.ts.map +1 -1
- package/dist/core/agents/rules/stop-hook.js +8 -0
- package/dist/core/agents/state/heartbeat-projector.d.ts +2 -0
- package/dist/core/agents/state/heartbeat-projector.d.ts.map +1 -1
- package/dist/core/agents/state/heartbeat-projector.js +13 -2
- package/dist/core/agents/state/heartbeat-writer.d.ts +16 -2
- package/dist/core/agents/state/heartbeat-writer.d.ts.map +1 -1
- package/dist/core/agents/state/heartbeat-writer.js +21 -4
- package/dist/core/config.d.ts +18 -0
- package/dist/core/config.d.ts.map +1 -1
- package/dist/core/config.js +38 -0
- package/dist/core/hooks/cli.js +29 -3
- package/dist/core/hooks/events/schema.d.ts +4 -0
- package/dist/core/hooks/events/schema.d.ts.map +1 -1
- package/dist/core/hooks/harness/events.d.ts +7 -0
- package/dist/core/hooks/harness/events.d.ts.map +1 -1
- package/dist/core/hooks/harness/events.js +1 -0
- package/dist/core/hooks/resolve/coord-root.d.ts +11 -0
- package/dist/core/hooks/resolve/coord-root.d.ts.map +1 -1
- package/dist/core/hooks/resolve/coord-root.js +28 -6
- package/dist/core/workflow/billing.d.ts +48 -0
- package/dist/core/workflow/billing.d.ts.map +1 -0
- package/dist/core/workflow/billing.js +102 -0
- package/dist/core/workflow/child-env.d.ts +30 -0
- package/dist/core/workflow/child-env.d.ts.map +1 -0
- package/dist/core/workflow/child-env.js +43 -0
- package/dist/core/workflow/engine.d.ts +21 -0
- package/dist/core/workflow/engine.d.ts.map +1 -0
- package/dist/core/workflow/engine.js +340 -0
- package/dist/core/workflow/harnesses.d.ts +17 -0
- package/dist/core/workflow/harnesses.d.ts.map +1 -0
- package/dist/core/workflow/harnesses.js +30 -0
- package/dist/core/workflow/spawn-claude.d.ts +22 -0
- package/dist/core/workflow/spawn-claude.d.ts.map +1 -0
- package/dist/core/workflow/spawn-claude.js +82 -0
- package/dist/core/workflow/spawn-codex.d.ts +19 -0
- package/dist/core/workflow/spawn-codex.d.ts.map +1 -0
- package/dist/core/workflow/spawn-codex.js +66 -0
- package/dist/core/workflow/spawn-cursor.d.ts +26 -0
- package/dist/core/workflow/spawn-cursor.d.ts.map +1 -0
- package/dist/core/workflow/spawn-cursor.js +72 -0
- package/dist/core/workflow/types.d.ts +145 -0
- package/dist/core/workflow/types.d.ts.map +1 -0
- package/dist/core/workflow/types.js +9 -0
- package/dist/core/workflow/validate.d.ts +15 -0
- package/dist/core/workflow/validate.d.ts.map +1 -0
- package/dist/core/workflow/validate.js +70 -0
- package/dist/lib/tools/ripgrep.d.ts +61 -0
- package/dist/lib/tools/ripgrep.d.ts.map +1 -0
- package/dist/lib/tools/ripgrep.js +217 -0
- package/package.json +1 -1
- package/src/commander.ts +11 -0
- package/src/commands/callers.ts +90 -26
- package/src/commands/doctor.ts +95 -2
- package/src/commands/grep.ts +737 -113
- package/src/commands/init.ts +138 -17
- package/src/commands/workflow.ts +132 -0
- package/src/core/agents/cli.ts +11 -4
- package/src/core/agents/rules/claim-conflict.ts +26 -1
- package/src/core/agents/rules/stop-hook.ts +17 -0
- package/src/core/agents/state/heartbeat-projector.ts +13 -1
- package/src/core/agents/state/heartbeat-writer.ts +30 -5
- package/src/core/config.ts +47 -0
- package/src/core/hooks/cli.ts +30 -3
- package/src/core/hooks/events/schema.ts +4 -0
- package/src/core/hooks/harness/events.ts +8 -0
- package/src/core/hooks/resolve/coord-root.ts +28 -6
- package/src/core/workflow/billing.ts +146 -0
- package/src/core/workflow/child-env.ts +47 -0
- package/src/core/workflow/engine.ts +394 -0
- package/src/core/workflow/harnesses.ts +38 -0
- package/src/core/workflow/spawn-claude.ts +99 -0
- package/src/core/workflow/spawn-codex.ts +74 -0
- package/src/core/workflow/spawn-cursor.ts +89 -0
- package/src/core/workflow/types.ts +153 -0
- package/src/core/workflow/validate.ts +75 -0
- package/src/lib/tools/ripgrep.ts +244 -0
package/src/commands/grep.ts
CHANGED
|
@@ -1,15 +1,40 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
|
+
import { readFile } from "node:fs/promises";
|
|
3
|
+
import { isAbsolute, join } from "node:path";
|
|
2
4
|
import type { Command } from "commander";
|
|
3
5
|
import type { EmitContext, HarneryProgramContext } from "../commander.ts";
|
|
6
|
+
import { resolveSearchEngine, type SearchEngine } from "../lib/tools/ripgrep.ts";
|
|
4
7
|
|
|
5
8
|
/**
|
|
6
|
-
* `grep`: monorepo-aware code search.
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
+
* `grep`: monorepo-aware code search. Prefers ripgrep (`rg`) when it is on
|
|
10
|
+
* PATH and falls back to GNU/BSD grep transparently — both engines are driven
|
|
11
|
+
* with equivalent flags and their output is parsed into the same envelope, so
|
|
12
|
+
* results are identical (pinned by tests/unit/grep-engine.test.ts).
|
|
13
|
+
* Smart default excludes (skip dist/.next/node_modules/.git/...), repo
|
|
14
|
+
* scoping (`--repo <name>` or `--all-repos`), language presets, file-level
|
|
15
|
+
* boolean composition (`--and` / `--without`), and context (`-C`/`-A`/`-B`).
|
|
9
16
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
17
|
+
* Engine selection: `HARNERY_GREP_ENGINE=rg|grep` forces one; otherwise `rg`
|
|
18
|
+
* is resolved (managed install, PATH, or opt-in auto-provision; see
|
|
19
|
+
* resolveEngine) and used when available. Repos are searched in parallel, and
|
|
20
|
+
* in `--all-repos` mode the parent scan prunes submodule directories so each
|
|
21
|
+
* match is attributed to exactly one repo.
|
|
22
|
+
*
|
|
23
|
+
* Output framing: content searches request the NUL filename delimiter
|
|
24
|
+
* (`--null` on both engines — the long spelling, because BSD grep repurposes
|
|
25
|
+
* `-Z` for decompression) so filename boundaries are never inferred from
|
|
26
|
+
* punctuation. Context lines are NOT requested from the engines: matches are
|
|
27
|
+
* selected (and budgeted) first, then context windows are materialized from
|
|
28
|
+
* one file read per selected file, which keeps `--limit` semantics exact and
|
|
29
|
+
* both engines byte-identical.
|
|
30
|
+
*
|
|
31
|
+
* Default behavior matches extended-regex semantics (`grep -E` / ripgrep's
|
|
32
|
+
* default). Use `-F` / `--literal` to pin to literal-string mode. Output is
|
|
33
|
+
* line-oriented `file:line:content` in TTY mode (context rows render
|
|
34
|
+
* grep-style as `file-line-content`); `--json` emits the full GrepResult
|
|
35
|
+
* envelope `{ pattern, mode, engine, and_patterns, without_patterns, repos,
|
|
36
|
+
* total_matches, total_files, truncated, elapsed_ms }`. Matches are sorted
|
|
37
|
+
* (file, then line) for stable output across runs and engines.
|
|
13
38
|
*/
|
|
14
39
|
|
|
15
40
|
const DEFAULT_EXCLUDE_DIRS = [
|
|
@@ -51,29 +76,40 @@ const LANG_GLOBS: Record<string, string[]> = {
|
|
|
51
76
|
rs: ["*.rs"],
|
|
52
77
|
};
|
|
53
78
|
|
|
54
|
-
|
|
79
|
+
export type GrepEngine = SearchEngine;
|
|
80
|
+
|
|
81
|
+
export interface GrepOpts {
|
|
55
82
|
repo?: string;
|
|
56
83
|
allRepos?: boolean;
|
|
57
|
-
lang
|
|
84
|
+
/** Repeatable and/or comma-separated (`--lang ts,tsx`). A bare string is accepted for direct callers. */
|
|
85
|
+
lang?: string | string[];
|
|
58
86
|
ignoreCase?: boolean;
|
|
59
87
|
wholeWord?: boolean;
|
|
60
88
|
literal?: boolean;
|
|
61
89
|
filesOnly?: boolean;
|
|
90
|
+
files?: boolean;
|
|
62
91
|
count?: boolean;
|
|
63
92
|
context?: string;
|
|
93
|
+
afterContext?: string;
|
|
94
|
+
beforeContext?: string;
|
|
64
95
|
maxCount?: string;
|
|
65
96
|
limit?: string;
|
|
66
97
|
include?: string[];
|
|
67
98
|
exclude?: string[];
|
|
99
|
+
and?: string[];
|
|
100
|
+
without?: string[];
|
|
101
|
+
quiet?: boolean;
|
|
68
102
|
noDefaultExcludes?: boolean;
|
|
69
103
|
json?: boolean;
|
|
70
104
|
}
|
|
71
105
|
|
|
72
|
-
interface Match {
|
|
106
|
+
export interface Match {
|
|
73
107
|
repo: string;
|
|
74
108
|
file: string;
|
|
75
109
|
line: number;
|
|
76
110
|
text: string;
|
|
111
|
+
/** "match" = primary result row (counts toward totals + --limit); "context" = free -C/-A/-B row. */
|
|
112
|
+
kind: "match" | "context";
|
|
77
113
|
}
|
|
78
114
|
|
|
79
115
|
export function registerGrepCommand(
|
|
@@ -84,20 +120,50 @@ export function registerGrepCommand(
|
|
|
84
120
|
program
|
|
85
121
|
.command("grep <pattern> [paths...]")
|
|
86
122
|
.description(
|
|
87
|
-
"Monorepo-aware code search
|
|
88
|
-
"
|
|
123
|
+
"Monorepo-aware code search (ripgrep when available, GNU grep fallback). " +
|
|
124
|
+
"Skips dist/.next/node_modules/.git/... by default. " +
|
|
125
|
+
"Use --repo, --all-repos, --lang for scoping; --and/--without for file-level composition. " +
|
|
126
|
+
"Regex by default; -F for literal.",
|
|
89
127
|
)
|
|
90
128
|
.option("--repo <name>", "Scope to one submodule (`.` = parent repo root)")
|
|
91
129
|
.option("--all-repos", "Search parent + every submodule")
|
|
92
|
-
.option(
|
|
130
|
+
.option(
|
|
131
|
+
"--lang <lang>",
|
|
132
|
+
`File type preset, repeatable or comma-separated (${Object.keys(LANG_GLOBS).join(", ")})`,
|
|
133
|
+
collect,
|
|
134
|
+
[] as string[],
|
|
135
|
+
)
|
|
93
136
|
.option("-i, --ignore-case", "Case-insensitive match")
|
|
94
137
|
.option("-w, --whole-word", "Match whole words only")
|
|
95
138
|
.option("-F, --literal", "Treat <pattern> as a literal string (no regex)")
|
|
96
139
|
.option("-l, --files-only", "Only print file names containing a match")
|
|
140
|
+
.option(
|
|
141
|
+
"--files",
|
|
142
|
+
"Filename search: treat <pattern> as a filename glob and list matching files " +
|
|
143
|
+
"(rg --files when available, POSIX find fallback)",
|
|
144
|
+
)
|
|
97
145
|
.option("-c, --count", "Print match count per file (suppresses content)")
|
|
98
146
|
.option("-C, --context <n>", "Print N lines of context around each match", "0")
|
|
147
|
+
.option("-A, --after-context <n>", "Print N lines after each match (overrides -C's after side)")
|
|
148
|
+
.option(
|
|
149
|
+
"-B, --before-context <n>",
|
|
150
|
+
"Print N lines before each match (overrides -C's before side)",
|
|
151
|
+
)
|
|
99
152
|
.option("--max-count <n>", "Stop after N matches per file")
|
|
100
153
|
.option("--limit <n>", "Truncate output to N matches total")
|
|
154
|
+
.option(
|
|
155
|
+
"--and <pattern>",
|
|
156
|
+
"Only show files that ALSO contain this pattern (file-level, repeatable)",
|
|
157
|
+
collect,
|
|
158
|
+
[] as string[],
|
|
159
|
+
)
|
|
160
|
+
.option(
|
|
161
|
+
"--without <pattern>",
|
|
162
|
+
"Drop files that contain this pattern (file-level, repeatable)",
|
|
163
|
+
collect,
|
|
164
|
+
[] as string[],
|
|
165
|
+
)
|
|
166
|
+
.option("-q, --quiet", "No output; exit 0 if any match exists, 1 if none")
|
|
101
167
|
.option("--include <glob>", "Extra --include glob (repeatable)", collect, [] as string[])
|
|
102
168
|
.option("--exclude <glob>", "Extra --exclude glob (repeatable)", collect, [] as string[])
|
|
103
169
|
.option("--no-default-excludes", "Disable the default skip list (node_modules, dist, etc.)")
|
|
@@ -105,6 +171,12 @@ export function registerGrepCommand(
|
|
|
105
171
|
.action(async (pattern: string, paths: string[], opts: GrepOpts) => {
|
|
106
172
|
try {
|
|
107
173
|
const result = await runGrep(pattern, paths, opts, context);
|
|
174
|
+
if (opts.quiet) {
|
|
175
|
+
// No output by contract; grep-conventional status. exitCode (not
|
|
176
|
+
// process.exit) so an embedding host isn't terminated mid-flush.
|
|
177
|
+
process.exitCode = result.total_matches > 0 ? 0 : 1;
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
108
180
|
if (opts.json) {
|
|
109
181
|
emit.config({ format: "json" });
|
|
110
182
|
emit.data(result);
|
|
@@ -122,9 +194,12 @@ function collect(value: string, prev: string[]): string[] {
|
|
|
122
194
|
return [...prev, value];
|
|
123
195
|
}
|
|
124
196
|
|
|
125
|
-
interface GrepResult {
|
|
197
|
+
export interface GrepResult {
|
|
126
198
|
pattern: string;
|
|
127
|
-
mode: "regex" | "literal";
|
|
199
|
+
mode: "regex" | "literal" | "files";
|
|
200
|
+
engine: GrepEngine;
|
|
201
|
+
and_patterns: string[];
|
|
202
|
+
without_patterns: string[];
|
|
128
203
|
repos: { name: string; cwd: string; matches: Match[]; truncated: boolean }[];
|
|
129
204
|
total_matches: number;
|
|
130
205
|
total_files: number;
|
|
@@ -132,47 +207,204 @@ interface GrepResult {
|
|
|
132
207
|
elapsed_ms: number;
|
|
133
208
|
}
|
|
134
209
|
|
|
135
|
-
|
|
210
|
+
/** Validated, resolved options — built once before repo fan-out. */
|
|
211
|
+
interface NormOpts {
|
|
212
|
+
limit: number; // Infinity when unset
|
|
213
|
+
maxCount: number | undefined;
|
|
214
|
+
before: number;
|
|
215
|
+
after: number;
|
|
216
|
+
langGlobs: string[] | undefined;
|
|
217
|
+
andPatterns: string[];
|
|
218
|
+
withoutPatterns: string[];
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Parse a strictly-decimal integer option. Rejects empty, negative, signed,
|
|
223
|
+
* fractional, and trailing-junk values before any engine is spawned.
|
|
224
|
+
*/
|
|
225
|
+
function parseIntOpt(flag: string, raw: string, min: number): number {
|
|
226
|
+
if (!/^\d+$/.test(raw.trim())) {
|
|
227
|
+
throw new Error(`${flag} expects a non-negative integer, got "${raw}"`);
|
|
228
|
+
}
|
|
229
|
+
const n = Number.parseInt(raw.trim(), 10);
|
|
230
|
+
if (n < min) throw new Error(`${flag} must be >= ${min}, got ${n}`);
|
|
231
|
+
return n;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
function normalizeLangs(lang: string | string[] | undefined): string[] | undefined {
|
|
235
|
+
const rawValues = typeof lang === "string" ? [lang] : (lang ?? []);
|
|
236
|
+
const keys: string[] = [];
|
|
237
|
+
for (const raw of rawValues) {
|
|
238
|
+
for (const piece of raw.split(",")) {
|
|
239
|
+
const key = piece.trim();
|
|
240
|
+
if (key === "") continue;
|
|
241
|
+
if (!LANG_GLOBS[key]) {
|
|
242
|
+
throw new Error(`unknown --lang "${key}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
|
|
243
|
+
}
|
|
244
|
+
if (!keys.includes(key)) keys.push(key);
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
if (keys.length === 0) return undefined;
|
|
248
|
+
const globs: string[] = [];
|
|
249
|
+
for (const key of keys) {
|
|
250
|
+
const langGlobs = LANG_GLOBS[key];
|
|
251
|
+
if (!langGlobs) continue;
|
|
252
|
+
for (const g of langGlobs) if (!globs.includes(g)) globs.push(g);
|
|
253
|
+
}
|
|
254
|
+
return globs;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/** Validate numerics, resolve context sides, expand langs, enforce the compatibility matrix. */
|
|
258
|
+
function normalizeOpts(opts: GrepOpts): NormOpts {
|
|
259
|
+
const limit = opts.limit ? parseIntOpt("--limit", opts.limit, 1) : Number.POSITIVE_INFINITY;
|
|
260
|
+
const maxCount = opts.maxCount ? parseIntOpt("--max-count", opts.maxCount, 1) : undefined;
|
|
261
|
+
const c = opts.context !== undefined ? parseIntOpt("-C/--context", opts.context, 0) : 0;
|
|
262
|
+
const before =
|
|
263
|
+
opts.beforeContext !== undefined
|
|
264
|
+
? parseIntOpt("-B/--before-context", opts.beforeContext, 0)
|
|
265
|
+
: c;
|
|
266
|
+
const after =
|
|
267
|
+
opts.afterContext !== undefined ? parseIntOpt("-A/--after-context", opts.afterContext, 0) : c;
|
|
268
|
+
const andPatterns = opts.and ?? [];
|
|
269
|
+
const withoutPatterns = opts.without ?? [];
|
|
270
|
+
const contextActive = before > 0 || after > 0;
|
|
271
|
+
|
|
272
|
+
if (opts.files) {
|
|
273
|
+
// Filename mode lists files by name glob; content-search flags make no
|
|
274
|
+
// sense here — reject loudly rather than silently ignoring them.
|
|
275
|
+
const incompatible: [unknown, string][] = [
|
|
276
|
+
[normalizeLangs(opts.lang)?.length, "--lang"],
|
|
277
|
+
[opts.count, "-c/--count"],
|
|
278
|
+
[opts.wholeWord, "-w/--whole-word"],
|
|
279
|
+
[opts.literal, "-F/--literal"],
|
|
280
|
+
[opts.maxCount, "--max-count"],
|
|
281
|
+
[opts.include?.length, "--include"],
|
|
282
|
+
[contextActive ? true : undefined, "-C/-A/-B context"],
|
|
283
|
+
[andPatterns.length ? true : undefined, "--and"],
|
|
284
|
+
[withoutPatterns.length ? true : undefined, "--without"],
|
|
285
|
+
];
|
|
286
|
+
for (const [set, flag] of incompatible) {
|
|
287
|
+
if (set) throw new Error(`${flag} does not apply to --files (filename glob) mode`);
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
if (contextActive) {
|
|
291
|
+
const rejected: [unknown, string][] = [
|
|
292
|
+
[opts.filesOnly, "-l/--files-only"],
|
|
293
|
+
[opts.count, "-c/--count"],
|
|
294
|
+
[opts.quiet, "-q/--quiet"],
|
|
295
|
+
];
|
|
296
|
+
for (const [set, flag] of rejected) {
|
|
297
|
+
if (set) throw new Error(`-C/-A/-B context does not combine with ${flag}`);
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
if (opts.quiet) {
|
|
301
|
+
const rejected: [unknown, string][] = [
|
|
302
|
+
[opts.json, "--json"],
|
|
303
|
+
[opts.filesOnly, "-l/--files-only"],
|
|
304
|
+
[opts.count, "-c/--count"],
|
|
305
|
+
[opts.limit, "--limit"],
|
|
306
|
+
];
|
|
307
|
+
for (const [set, flag] of rejected) {
|
|
308
|
+
if (set) throw new Error(`-q/--quiet does not combine with ${flag}`);
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
return {
|
|
313
|
+
// -q only needs existence: one accepted primary row settles the exit code.
|
|
314
|
+
limit: opts.quiet ? 1 : limit,
|
|
315
|
+
maxCount,
|
|
316
|
+
before,
|
|
317
|
+
after,
|
|
318
|
+
langGlobs: normalizeLangs(opts.lang),
|
|
319
|
+
andPatterns,
|
|
320
|
+
withoutPatterns,
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
/** Exported for tests (not part of the package exports map). */
|
|
325
|
+
export async function runGrep(
|
|
136
326
|
pattern: string,
|
|
137
327
|
paths: string[],
|
|
138
328
|
opts: GrepOpts,
|
|
139
329
|
context: HarneryProgramContext | undefined,
|
|
140
330
|
): Promise<GrepResult> {
|
|
141
331
|
if (!pattern) throw new Error("pattern required");
|
|
332
|
+
const norm = normalizeOpts(opts);
|
|
142
333
|
const started = Date.now();
|
|
143
334
|
|
|
335
|
+
const { engine, rgBin } = await resolveSearchEngine("grep");
|
|
144
336
|
const repos = resolveRepos(opts, context);
|
|
145
|
-
|
|
337
|
+
|
|
338
|
+
// Host-injected default excludes (generated mirrors, vendored trees, ...)
|
|
339
|
+
// ride the same --no-default-excludes gate as the built-in list.
|
|
340
|
+
const hostExcludeDirs = opts.noDefaultExcludes ? [] : (context?.grepExcludeDirs ?? []);
|
|
341
|
+
|
|
342
|
+
// All repos are searched concurrently; each collects at most limit+1
|
|
343
|
+
// accepted rows (one row of lookahead, so "exactly N results" is
|
|
344
|
+
// distinguishable from "more than N"), then the global budget is applied
|
|
345
|
+
// in repo order below so `--limit` semantics stay deterministic.
|
|
346
|
+
const acceptLimit = Number.isFinite(norm.limit) ? norm.limit + 1 : Number.POSITIVE_INFINITY;
|
|
347
|
+
const perRepo = await Promise.all(
|
|
348
|
+
repos.map((repo) => {
|
|
349
|
+
// In --all-repos mode the parent scan prunes submodule dirs — each
|
|
350
|
+
// submodule gets its own scoped scan, so descending from the parent
|
|
351
|
+
// would double-scan and double-report every submodule match. This
|
|
352
|
+
// pruning is correctness (one repo owns each match), so it applies
|
|
353
|
+
// even when --no-default-excludes is set.
|
|
354
|
+
const dedupeDirs = opts.allRepos && repo.name === "parent" ? (context?.submodules ?? []) : [];
|
|
355
|
+
const extraDirs = [...hostExcludeDirs, ...dedupeDirs];
|
|
356
|
+
return runGrepInRepo(
|
|
357
|
+
pattern,
|
|
358
|
+
paths,
|
|
359
|
+
opts,
|
|
360
|
+
norm,
|
|
361
|
+
repo.cwd,
|
|
362
|
+
repo.name,
|
|
363
|
+
acceptLimit,
|
|
364
|
+
engine,
|
|
365
|
+
rgBin,
|
|
366
|
+
extraDirs,
|
|
367
|
+
);
|
|
368
|
+
}),
|
|
369
|
+
);
|
|
146
370
|
|
|
147
371
|
const allRepoResults: GrepResult["repos"] = [];
|
|
148
372
|
let totalMatches = 0;
|
|
149
373
|
const filesSeen = new Set<string>();
|
|
150
374
|
let truncated = false;
|
|
375
|
+
let budget = norm.limit;
|
|
151
376
|
|
|
152
|
-
for (
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
if (
|
|
165
|
-
|
|
166
|
-
truncated = true;
|
|
167
|
-
}
|
|
377
|
+
for (let i = 0; i < repos.length; i++) {
|
|
378
|
+
const repo = repos[i];
|
|
379
|
+
if (!repo) continue;
|
|
380
|
+
const collected = perRepo[i] ?? [];
|
|
381
|
+
collected.sort((a, b) => (a.file < b.file ? -1 : a.file > b.file ? 1 : a.line - b.line));
|
|
382
|
+
const take = Number.isFinite(budget)
|
|
383
|
+
? Math.min(collected.length, Math.max(0, budget))
|
|
384
|
+
: collected.length;
|
|
385
|
+
let matches = collected.slice(0, take);
|
|
386
|
+
// truncated only when an accepted primary row was actually omitted —
|
|
387
|
+
// the lookahead row (or a later-repo surplus) is the proof.
|
|
388
|
+
const repoTruncated = collected.length > take;
|
|
389
|
+
if (repoTruncated) truncated = true;
|
|
390
|
+
if (Number.isFinite(budget)) budget -= take;
|
|
168
391
|
totalMatches += matches.length;
|
|
169
392
|
for (const m of matches) filesSeen.add(`${repo.name}/${m.file}`);
|
|
393
|
+
// Context is materialized only for the SELECTED rows, after budgeting,
|
|
394
|
+
// so context rows never consume the limit and trailing context is never
|
|
395
|
+
// cut off by an engine kill.
|
|
396
|
+
if (!opts.files && !opts.filesOnly && !opts.count && (norm.before > 0 || norm.after > 0)) {
|
|
397
|
+
matches = await materializeContext(matches, repo.cwd, norm.before, norm.after);
|
|
398
|
+
}
|
|
170
399
|
allRepoResults.push({ name: repo.name, cwd: repo.cwd, matches, truncated: repoTruncated });
|
|
171
400
|
}
|
|
172
401
|
|
|
173
402
|
return {
|
|
174
403
|
pattern,
|
|
175
|
-
mode: opts.literal ? "literal" : "regex",
|
|
404
|
+
mode: opts.files ? "files" : opts.literal ? "literal" : "regex",
|
|
405
|
+
engine,
|
|
406
|
+
and_patterns: norm.andPatterns,
|
|
407
|
+
without_patterns: norm.withoutPatterns,
|
|
176
408
|
repos: allRepoResults,
|
|
177
409
|
total_matches: totalMatches,
|
|
178
410
|
total_files: filesSeen.size,
|
|
@@ -220,71 +452,403 @@ async function runGrepInRepo(
|
|
|
220
452
|
pattern: string,
|
|
221
453
|
paths: string[],
|
|
222
454
|
opts: GrepOpts,
|
|
455
|
+
norm: NormOpts,
|
|
223
456
|
cwd: string,
|
|
224
457
|
repoName: string,
|
|
225
|
-
|
|
458
|
+
acceptLimit: number,
|
|
459
|
+
engine: GrepEngine,
|
|
460
|
+
rgBin: string,
|
|
461
|
+
extraExcludeDirs: readonly string[],
|
|
226
462
|
): Promise<Match[]> {
|
|
227
|
-
|
|
463
|
+
// File-level boolean composition: build the complete auxiliary file sets
|
|
464
|
+
// FIRST (no limit — a partial set can't prove absence), then feed the
|
|
465
|
+
// primary scan an in-stream predicate. Running the primary with its limit
|
|
466
|
+
// and filtering afterwards would let non-qualifying early hits consume the
|
|
467
|
+
// budget and hide later valid results.
|
|
468
|
+
let predicate: ((file: string) => boolean) | undefined;
|
|
469
|
+
if (norm.andPatterns.length > 0 || norm.withoutPatterns.length > 0) {
|
|
470
|
+
const requiredSets: Set<string>[] = [];
|
|
471
|
+
// Sequential, not parallel: repeated flags must not multiply peak child
|
|
472
|
+
// process concurrency by repos x patterns. Repo workers stay parallel.
|
|
473
|
+
for (const p of norm.andPatterns) {
|
|
474
|
+
const files = await runFileSetScan(
|
|
475
|
+
p,
|
|
476
|
+
paths,
|
|
477
|
+
opts,
|
|
478
|
+
norm,
|
|
479
|
+
cwd,
|
|
480
|
+
engine,
|
|
481
|
+
rgBin,
|
|
482
|
+
extraExcludeDirs,
|
|
483
|
+
);
|
|
484
|
+
if (files.size === 0) return []; // intersection is already empty
|
|
485
|
+
requiredSets.push(files);
|
|
486
|
+
}
|
|
487
|
+
const forbidden = new Set<string>();
|
|
488
|
+
for (const p of norm.withoutPatterns) {
|
|
489
|
+
const files = await runFileSetScan(
|
|
490
|
+
p,
|
|
491
|
+
paths,
|
|
492
|
+
opts,
|
|
493
|
+
norm,
|
|
494
|
+
cwd,
|
|
495
|
+
engine,
|
|
496
|
+
rgBin,
|
|
497
|
+
extraExcludeDirs,
|
|
498
|
+
);
|
|
499
|
+
for (const f of files) forbidden.add(f);
|
|
500
|
+
}
|
|
501
|
+
predicate = (file) => requiredSets.every((s) => s.has(file)) && !forbidden.has(file);
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
const [bin, args, decodeMode] = buildEngineInvocation(
|
|
505
|
+
pattern,
|
|
506
|
+
paths,
|
|
507
|
+
opts,
|
|
508
|
+
norm,
|
|
509
|
+
engine,
|
|
510
|
+
rgBin,
|
|
511
|
+
extraExcludeDirs,
|
|
512
|
+
false,
|
|
513
|
+
);
|
|
514
|
+
|
|
515
|
+
let matches = await execSearch(bin, args, cwd, acceptLimit, decodeMode, predicate, false);
|
|
516
|
+
// -c mode: GNU grep prints a `path:0` row for every searched file; ripgrep
|
|
517
|
+
// omits zero-count rows. Filter zeros on both engines so output matches.
|
|
518
|
+
if (opts.count) matches = matches.filter((m) => m.line > 0);
|
|
519
|
+
return matches.map((m) => ({ ...m, repo: repoName }));
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
/** Membership scan for one --and/--without pattern: complete files-only set, strict failure. */
|
|
523
|
+
async function runFileSetScan(
|
|
524
|
+
pattern: string,
|
|
525
|
+
paths: string[],
|
|
526
|
+
opts: GrepOpts,
|
|
527
|
+
norm: NormOpts,
|
|
528
|
+
cwd: string,
|
|
529
|
+
engine: GrepEngine,
|
|
530
|
+
rgBin: string,
|
|
531
|
+
extraExcludeDirs: readonly string[],
|
|
532
|
+
): Promise<Set<string>> {
|
|
533
|
+
const [bin, args, decodeMode] = buildEngineInvocation(
|
|
534
|
+
pattern,
|
|
535
|
+
paths,
|
|
536
|
+
opts,
|
|
537
|
+
norm,
|
|
538
|
+
engine,
|
|
539
|
+
rgBin,
|
|
540
|
+
extraExcludeDirs,
|
|
541
|
+
true,
|
|
542
|
+
);
|
|
543
|
+
const rows = await execSearch(
|
|
544
|
+
bin,
|
|
545
|
+
args,
|
|
546
|
+
cwd,
|
|
547
|
+
Number.POSITIVE_INFINITY,
|
|
548
|
+
decodeMode,
|
|
549
|
+
undefined,
|
|
550
|
+
true,
|
|
551
|
+
);
|
|
552
|
+
return new Set(rows.map((r) => r.file));
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
type DecodeMode = "content" | "count" | "filesOnly" | "plainLines";
|
|
556
|
+
|
|
557
|
+
/**
|
|
558
|
+
* Build one engine invocation. `membership: true` forces files-only mode for
|
|
559
|
+
* --and/--without set scans (match-shaping + scope flags shared with the
|
|
560
|
+
* primary; output flags not applied).
|
|
561
|
+
*/
|
|
562
|
+
function buildEngineInvocation(
|
|
563
|
+
pattern: string,
|
|
564
|
+
paths: string[],
|
|
565
|
+
opts: GrepOpts,
|
|
566
|
+
norm: NormOpts,
|
|
567
|
+
engine: GrepEngine,
|
|
568
|
+
rgBin: string,
|
|
569
|
+
extraExcludeDirs: readonly string[],
|
|
570
|
+
membership: boolean,
|
|
571
|
+
): [string, string[], DecodeMode] {
|
|
572
|
+
if (opts.files) {
|
|
573
|
+
// Filename mode keeps its existing rg/find framing (plain lines).
|
|
574
|
+
return engine === "rg"
|
|
575
|
+
? [rgBin, buildRgFilesArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"]
|
|
576
|
+
: ["find", buildFindArgs(pattern, paths, opts, extraExcludeDirs), "plainLines"];
|
|
577
|
+
}
|
|
578
|
+
const filesOnly = membership || Boolean(opts.filesOnly);
|
|
579
|
+
const count = !membership && Boolean(opts.count);
|
|
580
|
+
const decodeMode: DecodeMode = filesOnly ? "filesOnly" : count ? "count" : "content";
|
|
581
|
+
const args =
|
|
582
|
+
engine === "rg"
|
|
583
|
+
? buildRgArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count })
|
|
584
|
+
: buildGrepArgs(pattern, paths, opts, norm, extraExcludeDirs, { filesOnly, count });
|
|
585
|
+
return [engine === "rg" ? rgBin : "grep", args, decodeMode];
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
interface OutputShape {
|
|
589
|
+
filesOnly: boolean;
|
|
590
|
+
count: boolean;
|
|
591
|
+
}
|
|
592
|
+
|
|
593
|
+
function buildGrepArgs(
|
|
594
|
+
pattern: string,
|
|
595
|
+
paths: string[],
|
|
596
|
+
opts: GrepOpts,
|
|
597
|
+
norm: NormOpts,
|
|
598
|
+
extraExcludeDirs: readonly string[],
|
|
599
|
+
shape: OutputShape,
|
|
600
|
+
): string[] {
|
|
601
|
+
// --null (NOT -Z: BSD grep repurposes -Z for decompression): NUL after the
|
|
602
|
+
// filename, so paths containing colons/dashes can't confuse the decoder.
|
|
603
|
+
// -I: skip binary files. Context is never requested from the engine — see
|
|
604
|
+
// materializeContext.
|
|
605
|
+
const args: string[] = ["-rn", "-H", "--null", "--color=never", "-I"];
|
|
228
606
|
if (opts.ignoreCase) args.push("-i");
|
|
229
607
|
if (opts.wholeWord) args.push("-w");
|
|
230
608
|
if (opts.literal) args.push("-F");
|
|
231
609
|
else args.push("-E");
|
|
232
|
-
if (
|
|
233
|
-
if (
|
|
234
|
-
|
|
235
|
-
if (Number.isFinite(contextN) && contextN > 0) args.push(`-C${contextN}`);
|
|
236
|
-
if (opts.maxCount) {
|
|
237
|
-
const n = Number.parseInt(opts.maxCount, 10);
|
|
238
|
-
if (Number.isFinite(n) && n > 0) args.push(`-m${n}`);
|
|
239
|
-
}
|
|
610
|
+
if (shape.filesOnly) args.push("-l");
|
|
611
|
+
if (shape.count) args.push("-c");
|
|
612
|
+
if (norm.maxCount !== undefined) args.push(`-m${norm.maxCount}`);
|
|
240
613
|
|
|
241
614
|
if (!opts.noDefaultExcludes) {
|
|
242
615
|
for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--exclude-dir=${d}`);
|
|
243
616
|
for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--exclude=${f}`);
|
|
244
617
|
}
|
|
618
|
+
for (const d of extraExcludeDirs) args.push(`--exclude-dir=${d}`);
|
|
245
619
|
for (const e of opts.exclude ?? []) args.push(`--exclude=${e}`);
|
|
246
620
|
|
|
247
|
-
|
|
248
|
-
if (opts.lang && !langGlobs) {
|
|
249
|
-
throw new Error(`unknown --lang "${opts.lang}". Valid: ${Object.keys(LANG_GLOBS).join(", ")}`);
|
|
250
|
-
}
|
|
251
|
-
if (langGlobs) for (const g of langGlobs) args.push(`--include=${g}`);
|
|
621
|
+
if (norm.langGlobs) for (const g of norm.langGlobs) args.push(`--include=${g}`);
|
|
252
622
|
for (const g of opts.include ?? []) args.push(`--include=${g}`);
|
|
253
623
|
|
|
254
624
|
args.push("--", pattern);
|
|
255
625
|
if (paths.length > 0) args.push(...paths);
|
|
256
626
|
else args.push(".");
|
|
627
|
+
return args;
|
|
628
|
+
}
|
|
257
629
|
|
|
258
|
-
|
|
259
|
-
|
|
630
|
+
function buildRgArgs(
|
|
631
|
+
pattern: string,
|
|
632
|
+
paths: string[],
|
|
633
|
+
opts: GrepOpts,
|
|
634
|
+
norm: NormOpts,
|
|
635
|
+
extraExcludeDirs: readonly string[],
|
|
636
|
+
shape: OutputShape,
|
|
637
|
+
): string[] {
|
|
638
|
+
// --hidden --no-ignore: match GNU grep's semantics (search dotdirs, ignore
|
|
639
|
+
// .gitignore) so the only filters are the explicit exclude lists.
|
|
640
|
+
// --no-config: a user's ripgrep config file must not skew results.
|
|
641
|
+
// --with-filename: rg drops the file prefix for a single explicit file arg,
|
|
642
|
+
// which would break the shared decoder.
|
|
643
|
+
// --null: NUL after the filename (same spelling as GNU/BSD grep's long flag).
|
|
644
|
+
const args: string[] = [
|
|
645
|
+
"-n",
|
|
646
|
+
"--no-heading",
|
|
647
|
+
"--with-filename",
|
|
648
|
+
"--null",
|
|
649
|
+
"--color=never",
|
|
650
|
+
"--no-config",
|
|
651
|
+
"--hidden",
|
|
652
|
+
"--no-ignore",
|
|
653
|
+
];
|
|
654
|
+
if (opts.ignoreCase) args.push("-i");
|
|
655
|
+
if (opts.wholeWord) args.push("-w");
|
|
656
|
+
if (opts.literal) args.push("-F");
|
|
657
|
+
if (shape.filesOnly) args.push("-l");
|
|
658
|
+
if (shape.count) args.push("-c");
|
|
659
|
+
if (norm.maxCount !== undefined) args.push(`-m${norm.maxCount}`);
|
|
660
|
+
|
|
661
|
+
// Gitignore-style globs: a bare name matches (and prunes) at any depth,
|
|
662
|
+
// mirroring grep's --exclude-dir / --exclude basename semantics. ORDER
|
|
663
|
+
// MATTERS: rg globs are last-match-wins, so positives (includes/lang) go
|
|
664
|
+
// first and negatives (excludes) last — otherwise `--include '*.md'` would
|
|
665
|
+
// re-include a .md file inside an excluded node_modules/. grep's
|
|
666
|
+
// --exclude-dir always beats --include, so this keeps the engines aligned.
|
|
667
|
+
if (norm.langGlobs) for (const g of norm.langGlobs) args.push(`--glob=${g}`);
|
|
668
|
+
for (const g of opts.include ?? []) args.push(`--glob=${g}`);
|
|
669
|
+
|
|
670
|
+
if (!opts.noDefaultExcludes) {
|
|
671
|
+
for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--glob=!${d}`);
|
|
672
|
+
for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--glob=!${f}`);
|
|
673
|
+
}
|
|
674
|
+
for (const d of extraExcludeDirs) args.push(`--glob=!${d}`);
|
|
675
|
+
for (const e of opts.exclude ?? []) args.push(`--glob=!${e}`);
|
|
676
|
+
|
|
677
|
+
args.push("--", pattern);
|
|
678
|
+
if (paths.length > 0) args.push(...paths);
|
|
679
|
+
else args.push(".");
|
|
680
|
+
return args;
|
|
260
681
|
}
|
|
261
682
|
|
|
262
|
-
|
|
683
|
+
/**
|
|
684
|
+
* Filename mode via ripgrep: `--files` lists files; a single positive glob
|
|
685
|
+
* acts as a whitelist, negative globs prune. `--iglob` gives -i semantics.
|
|
686
|
+
*/
|
|
687
|
+
function buildRgFilesArgs(
|
|
688
|
+
pattern: string,
|
|
689
|
+
paths: string[],
|
|
690
|
+
opts: GrepOpts,
|
|
691
|
+
extraExcludeDirs: readonly string[],
|
|
692
|
+
): string[] {
|
|
693
|
+
const args: string[] = ["--files", "--hidden", "--no-ignore", "--no-config", "--color=never"];
|
|
694
|
+
// Positive pattern FIRST, negatives last (rg globs are last-match-wins;
|
|
695
|
+
// excludes must beat the pattern — see the ordering note in buildRgArgs).
|
|
696
|
+
args.push(`${opts.ignoreCase ? "--iglob" : "--glob"}=${pattern}`);
|
|
697
|
+
if (!opts.noDefaultExcludes) {
|
|
698
|
+
for (const d of DEFAULT_EXCLUDE_DIRS) args.push(`--glob=!${d}`);
|
|
699
|
+
for (const f of DEFAULT_EXCLUDE_FILES) args.push(`--glob=!${f}`);
|
|
700
|
+
}
|
|
701
|
+
for (const d of extraExcludeDirs) args.push(`--glob=!${d}`);
|
|
702
|
+
for (const e of opts.exclude ?? []) args.push(`--glob=!${e}`);
|
|
703
|
+
if (paths.length > 0) args.push(...paths);
|
|
704
|
+
else args.push(".");
|
|
705
|
+
return args;
|
|
706
|
+
}
|
|
707
|
+
|
|
708
|
+
/**
|
|
709
|
+
* Filename mode via POSIX find (the no-ripgrep fallback):
|
|
710
|
+
* find <paths> ( -name d1 -o -name d2 ... ) -prune -o -type f -name <glob> [! -name <ex>]... -print
|
|
711
|
+
* Sticks to POSIX operators (`(`, `-o`, `!`) so BSD/macOS find behaves the same.
|
|
712
|
+
*/
|
|
713
|
+
function buildFindArgs(
|
|
714
|
+
pattern: string,
|
|
715
|
+
paths: string[],
|
|
716
|
+
opts: GrepOpts,
|
|
717
|
+
extraExcludeDirs: readonly string[],
|
|
718
|
+
): string[] {
|
|
719
|
+
const args: string[] = paths.length > 0 ? [...paths] : ["."];
|
|
720
|
+
const pruneDirs = [...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_DIRS), ...extraExcludeDirs];
|
|
721
|
+
if (pruneDirs.length > 0) {
|
|
722
|
+
args.push("(");
|
|
723
|
+
pruneDirs.forEach((d, i) => {
|
|
724
|
+
if (i > 0) args.push("-o");
|
|
725
|
+
args.push("-name", d);
|
|
726
|
+
});
|
|
727
|
+
args.push(")", "-prune", "-o");
|
|
728
|
+
}
|
|
729
|
+
args.push("-type", "f", opts.ignoreCase ? "-iname" : "-name", pattern);
|
|
730
|
+
const fileExcludes = [
|
|
731
|
+
...(opts.noDefaultExcludes ? [] : DEFAULT_EXCLUDE_FILES),
|
|
732
|
+
...(opts.exclude ?? []),
|
|
733
|
+
];
|
|
734
|
+
for (const e of fileExcludes) args.push("!", "-name", e);
|
|
735
|
+
args.push("-print");
|
|
736
|
+
return args;
|
|
737
|
+
}
|
|
738
|
+
|
|
739
|
+
/**
|
|
740
|
+
* Streaming decoder for NUL-framed engine output. Exported for direct
|
|
741
|
+
* chunk-boundary tests. Record shapes (both engines, pinned by the parity
|
|
742
|
+
* suite):
|
|
743
|
+
* content: <path>NUL<line>:<text>LF
|
|
744
|
+
* count: <path>NUL<count>LF
|
|
745
|
+
* filesOnly: <path>NUL (NUL is the record terminator)
|
|
746
|
+
* plainLines: <path>LF (files mode: rg --files / find)
|
|
747
|
+
*
|
|
748
|
+
* Operates on Buffers so a chunk boundary can fall anywhere — including
|
|
749
|
+
* inside a multi-byte UTF-8 sequence — without corrupting a record: bytes
|
|
750
|
+
* are only decoded to strings once a full record is framed.
|
|
751
|
+
*/
|
|
752
|
+
export class NulDecoder {
|
|
753
|
+
private buffer: Buffer = Buffer.alloc(0);
|
|
754
|
+
constructor(private readonly mode: DecodeMode) {}
|
|
755
|
+
|
|
756
|
+
push(chunk: Buffer): Omit<Match, "repo" | "kind">[] {
|
|
757
|
+
this.buffer = this.buffer.length === 0 ? chunk : Buffer.concat([this.buffer, chunk]);
|
|
758
|
+
const rows: Omit<Match, "repo" | "kind">[] = [];
|
|
759
|
+
if (this.mode === "filesOnly") {
|
|
760
|
+
let nul = this.buffer.indexOf(0);
|
|
761
|
+
while (nul >= 0) {
|
|
762
|
+
const path = this.buffer.subarray(0, nul).toString("utf8");
|
|
763
|
+
this.buffer = this.buffer.subarray(nul + 1);
|
|
764
|
+
// Engines may still newline-separate NUL-terminated records in edge
|
|
765
|
+
// configurations; a bare leading LF is framing residue, not a path.
|
|
766
|
+
const cleaned = path.startsWith("\n") ? path.slice(1) : path;
|
|
767
|
+
if (cleaned.length > 0) rows.push({ file: normalizeFile(cleaned), line: 0, text: "" });
|
|
768
|
+
nul = this.buffer.indexOf(0);
|
|
769
|
+
}
|
|
770
|
+
return rows;
|
|
771
|
+
}
|
|
772
|
+
let nl = this.buffer.indexOf(0x0a);
|
|
773
|
+
while (nl >= 0) {
|
|
774
|
+
const record = this.buffer.subarray(0, nl);
|
|
775
|
+
this.buffer = this.buffer.subarray(nl + 1);
|
|
776
|
+
const row = this.decodeRecord(record);
|
|
777
|
+
if (row) rows.push(row);
|
|
778
|
+
nl = this.buffer.indexOf(0x0a);
|
|
779
|
+
}
|
|
780
|
+
return rows;
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
/** Flush a trailing unterminated record (engine killed mid-write, or no final LF). */
|
|
784
|
+
flush(): Omit<Match, "repo" | "kind">[] {
|
|
785
|
+
if (this.buffer.length === 0) return [];
|
|
786
|
+
const record = this.buffer;
|
|
787
|
+
this.buffer = Buffer.alloc(0);
|
|
788
|
+
if (this.mode === "filesOnly") {
|
|
789
|
+
// A partial filesOnly record has no terminating NUL — an engine killed
|
|
790
|
+
// mid-path would yield a corrupt name; drop it.
|
|
791
|
+
return [];
|
|
792
|
+
}
|
|
793
|
+
const row = this.decodeRecord(record);
|
|
794
|
+
return row ? [row] : [];
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
private decodeRecord(record: Buffer): Omit<Match, "repo" | "kind"> | null {
|
|
798
|
+
if (this.mode === "plainLines") {
|
|
799
|
+
const path = record.toString("utf8");
|
|
800
|
+
if (path.length === 0) return null;
|
|
801
|
+
return { file: normalizeFile(path), line: 0, text: "" };
|
|
802
|
+
}
|
|
803
|
+
const nul = record.indexOf(0);
|
|
804
|
+
if (nul < 0) return null; // not a NUL-framed record (stray engine chatter)
|
|
805
|
+
const file = normalizeFile(record.subarray(0, nul).toString("utf8"));
|
|
806
|
+
const rest = record.subarray(nul + 1).toString("utf8");
|
|
807
|
+
if (this.mode === "count") {
|
|
808
|
+
const count = Number.parseInt(rest, 10);
|
|
809
|
+
if (!Number.isFinite(count)) return null;
|
|
810
|
+
return { file, line: count, text: "" };
|
|
811
|
+
}
|
|
812
|
+
// content: <line>:<text>. The engines never emit context/separator rows
|
|
813
|
+
// (context is materialized from file reads), so `:` is always present.
|
|
814
|
+
const colon = rest.indexOf(":");
|
|
815
|
+
if (colon < 0) return null;
|
|
816
|
+
const lineNum = Number.parseInt(rest.slice(0, colon), 10);
|
|
817
|
+
if (!Number.isFinite(lineNum)) return null;
|
|
818
|
+
return { file, line: lineNum, text: rest.slice(colon + 1) };
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
function execSearch(
|
|
823
|
+
bin: string,
|
|
824
|
+
args: string[],
|
|
825
|
+
cwd: string,
|
|
826
|
+
acceptLimit: number,
|
|
827
|
+
decodeMode: DecodeMode,
|
|
828
|
+
filePredicate: ((file: string) => boolean) | undefined,
|
|
829
|
+
strict: boolean,
|
|
830
|
+
): Promise<Match[]> {
|
|
263
831
|
return new Promise((resolveP, reject) => {
|
|
264
|
-
const proc = spawn(
|
|
265
|
-
const
|
|
266
|
-
|
|
832
|
+
const proc = spawn(bin, args, { cwd, stdio: ["ignore", "pipe", "pipe"] });
|
|
833
|
+
const decoder = new NulDecoder(decodeMode);
|
|
834
|
+
const matches: Match[] = [];
|
|
267
835
|
let stderr = "";
|
|
268
|
-
let
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
return;
|
|
285
|
-
}
|
|
286
|
-
}
|
|
287
|
-
nl = buffer.indexOf("\n");
|
|
836
|
+
let killed = false;
|
|
837
|
+
|
|
838
|
+
const accept = (rows: Omit<Match, "repo" | "kind">[]): boolean => {
|
|
839
|
+
for (const row of rows) {
|
|
840
|
+
if (filePredicate && !filePredicate(row.file)) continue;
|
|
841
|
+
matches.push({ ...row, repo: "", kind: "match" });
|
|
842
|
+
if (Number.isFinite(acceptLimit) && matches.length >= acceptLimit) return true;
|
|
843
|
+
}
|
|
844
|
+
return false;
|
|
845
|
+
};
|
|
846
|
+
|
|
847
|
+
proc.stdout.on("data", (chunk: Buffer) => {
|
|
848
|
+
if (killed) return;
|
|
849
|
+
if (accept(decoder.push(chunk))) {
|
|
850
|
+
killed = true;
|
|
851
|
+
proc.kill("SIGTERM");
|
|
288
852
|
}
|
|
289
853
|
});
|
|
290
854
|
proc.stderr.setEncoding("utf8");
|
|
@@ -293,13 +857,15 @@ function execGrep(args: string[], cwd: string, limit: number): Promise<Omit<Matc
|
|
|
293
857
|
});
|
|
294
858
|
proc.on("error", reject);
|
|
295
859
|
proc.on("close", (code) => {
|
|
296
|
-
if (
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
//
|
|
301
|
-
|
|
302
|
-
|
|
860
|
+
if (!killed) accept(decoder.flush());
|
|
861
|
+
// Exit 1 is "no matches" on both engines: normal, not an error.
|
|
862
|
+
// Primary scans tolerate exit 2 with matches collected (an unreadable
|
|
863
|
+
// file mid-walk): surface the results rather than throwing them away.
|
|
864
|
+
// STRICT scans (--and/--without membership) must reject on ANY partial
|
|
865
|
+
// failure — an incomplete set cannot prove that a file lacks a pattern.
|
|
866
|
+
const failed = code !== null && code !== 0 && code !== 1 && !killed;
|
|
867
|
+
if (failed && (strict || matches.length === 0)) {
|
|
868
|
+
reject(new Error(`${bin} exited ${code}: ${stderr.trim() || "(no stderr)"}`));
|
|
303
869
|
return;
|
|
304
870
|
}
|
|
305
871
|
resolveP(matches);
|
|
@@ -307,32 +873,81 @@ function execGrep(args: string[], cwd: string, limit: number): Promise<Omit<Matc
|
|
|
307
873
|
});
|
|
308
874
|
}
|
|
309
875
|
|
|
310
|
-
/**
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
876
|
+
/**
|
|
877
|
+
* Materialize -C/-A/-B context for already-selected match rows: one file
|
|
878
|
+
* read per selected file, intervals clamped + merged, match rows keep the
|
|
879
|
+
* engine-captured text, other window lines become kind:"context" rows.
|
|
880
|
+
* An unreadable file (deleted since the scan — the accepted TOCTOU window)
|
|
881
|
+
* degrades to its match rows without context rather than failing the search.
|
|
882
|
+
*/
|
|
883
|
+
async function materializeContext(
|
|
884
|
+
selected: Match[],
|
|
885
|
+
cwd: string,
|
|
886
|
+
before: number,
|
|
887
|
+
after: number,
|
|
888
|
+
): Promise<Match[]> {
|
|
889
|
+
const byFile = new Map<string, Match[]>();
|
|
890
|
+
for (const m of selected) {
|
|
891
|
+
const rows = byFile.get(m.file);
|
|
892
|
+
if (rows) rows.push(m);
|
|
893
|
+
else byFile.set(m.file, [m]);
|
|
325
894
|
}
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
895
|
+
|
|
896
|
+
const out: Match[] = [];
|
|
897
|
+
for (const [file, rows] of byFile) {
|
|
898
|
+
const abs = isAbsolute(file) ? file : join(cwd, file);
|
|
899
|
+
let lines: string[] | null = null;
|
|
900
|
+
try {
|
|
901
|
+
const content = await readFile(abs, "utf8");
|
|
902
|
+
lines = content.split("\n");
|
|
903
|
+
if (lines.at(-1) === "") lines.pop();
|
|
904
|
+
} catch {
|
|
905
|
+
lines = null;
|
|
906
|
+
}
|
|
907
|
+
if (lines === null) {
|
|
908
|
+
out.push(...rows);
|
|
909
|
+
continue;
|
|
910
|
+
}
|
|
911
|
+
// Merge overlapping/adjacent [line-before, line+after] windows.
|
|
912
|
+
const intervals: [number, number][] = rows
|
|
913
|
+
.map((m): [number, number] => [
|
|
914
|
+
Math.max(1, m.line - before),
|
|
915
|
+
Math.min(lines.length, m.line + after),
|
|
916
|
+
])
|
|
917
|
+
.sort((a, b) => a[0] - b[0]);
|
|
918
|
+
const merged: [number, number][] = [];
|
|
919
|
+
for (const iv of intervals) {
|
|
920
|
+
const last = merged.at(-1);
|
|
921
|
+
if (last && iv[0] <= last[1] + 1) last[1] = Math.max(last[1], iv[1]);
|
|
922
|
+
else merged.push([iv[0], iv[1]]);
|
|
923
|
+
}
|
|
924
|
+
const matchByLine = new Map(rows.map((m) => [m.line, m]));
|
|
925
|
+
const emitted = new Set<Match>();
|
|
926
|
+
for (const [start, end] of merged) {
|
|
927
|
+
if (start > end) continue;
|
|
928
|
+
for (let n = start; n <= end; n++) {
|
|
929
|
+
const m = matchByLine.get(n);
|
|
930
|
+
if (m) {
|
|
931
|
+
out.push(m);
|
|
932
|
+
emitted.add(m);
|
|
933
|
+
} else {
|
|
934
|
+
const repo = rows[0]?.repo ?? "";
|
|
935
|
+
out.push({ repo, file, line: n, text: lines[n - 1] ?? "", kind: "context" });
|
|
936
|
+
}
|
|
937
|
+
}
|
|
938
|
+
}
|
|
939
|
+
// A match whose line exceeds the file's current length (file shrank
|
|
940
|
+
// since the scan) sits outside every clamped window — still emit it.
|
|
941
|
+
for (const m of rows) {
|
|
942
|
+
if (!emitted.has(m)) out.push(m);
|
|
943
|
+
}
|
|
330
944
|
}
|
|
331
|
-
return
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
945
|
+
return out;
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
/** Both engines may emit a leading `./` for the default path; strip it so output is engine-identical. */
|
|
949
|
+
function normalizeFile(file: string): string {
|
|
950
|
+
return file.startsWith("./") ? file.slice(2) : file;
|
|
336
951
|
}
|
|
337
952
|
|
|
338
953
|
function renderResult(r: GrepResult, opts: GrepOpts): string {
|
|
@@ -340,32 +955,41 @@ function renderResult(r: GrepResult, opts: GrepOpts): string {
|
|
|
340
955
|
const showRepoHeader = r.repos.length > 1;
|
|
341
956
|
for (const repo of r.repos) {
|
|
342
957
|
if (repo.matches.length === 0) continue;
|
|
958
|
+
const primary = repo.matches.filter((m) => m.kind === "match").length;
|
|
343
959
|
if (showRepoHeader) {
|
|
344
960
|
lines.push("");
|
|
345
|
-
lines.push(
|
|
346
|
-
`── ${repo.name} (${repo.matches.length} match${repo.matches.length === 1 ? "" : "es"}) ──`,
|
|
347
|
-
);
|
|
961
|
+
lines.push(`── ${repo.name} (${primary} match${primary === 1 ? "" : "es"}) ──`);
|
|
348
962
|
}
|
|
963
|
+
// Context mode: `--` between disjoint windows (file change or line gap).
|
|
964
|
+
const contextMode = repo.matches.some((x) => x.kind === "context");
|
|
965
|
+
let prev: Match | undefined;
|
|
349
966
|
for (const m of repo.matches) {
|
|
350
|
-
if (
|
|
967
|
+
if (contextMode && prev && (prev.file !== m.file || m.line !== prev.line + 1)) {
|
|
968
|
+
lines.push("--");
|
|
969
|
+
}
|
|
970
|
+
if (opts.filesOnly || opts.files) {
|
|
351
971
|
lines.push(m.file);
|
|
352
972
|
} else if (opts.count) {
|
|
353
973
|
lines.push(`${m.file}:${m.line}`);
|
|
974
|
+
} else if (m.kind === "context") {
|
|
975
|
+
lines.push(`${m.file}-${m.line}-${m.text}`);
|
|
354
976
|
} else {
|
|
355
977
|
lines.push(`${m.file}:${m.line}:${m.text}`);
|
|
356
978
|
}
|
|
979
|
+
prev = m;
|
|
357
980
|
}
|
|
358
981
|
if (repo.truncated) {
|
|
359
|
-
lines.push(" (truncated;
|
|
982
|
+
lines.push(" (truncated; raise --limit)");
|
|
360
983
|
}
|
|
361
984
|
}
|
|
362
985
|
if (r.total_matches === 0) {
|
|
363
|
-
|
|
986
|
+
const modeTag = r.mode === "literal" ? " (literal)" : r.mode === "files" ? " (files)" : "";
|
|
987
|
+
lines.push(`(no matches for /${r.pattern}/${modeTag})`);
|
|
364
988
|
} else if (r.repos.length > 1 || r.truncated) {
|
|
365
989
|
lines.push("");
|
|
366
990
|
const summary = `${r.total_matches} match${r.total_matches === 1 ? "" : "es"} across ${r.total_files} file${r.total_files === 1 ? "" : "s"}`;
|
|
367
|
-
const tail = r.truncated ? " (truncated;
|
|
368
|
-
lines.push(`${summary}${tail} (${r.elapsed_ms}ms)`);
|
|
991
|
+
const tail = r.truncated ? " (truncated; raise --limit)" : "";
|
|
992
|
+
lines.push(`${summary}${tail} (${r.elapsed_ms}ms, ${r.engine})`);
|
|
369
993
|
}
|
|
370
994
|
return lines.join("\n");
|
|
371
995
|
}
|