@opengeni/jev 0.1.0-canary.36199476632001

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,648 @@
1
+ /**
2
+ * recall.ts - wide, deterministic lexical recall with ripgrep.
3
+ *
4
+ * One ripgrep pass over the union of every keyword's identifier variants, searched case-insensitively
5
+ * (camelCase <-> snake_case <-> kebab <-> spaced phrase; SCREAMING and Pascal forms are covered by -i);
6
+ * a union longer than CODE_SEARCH_MAX_PATTERN_CHARS is split into several passes and merged. Lines are
7
+ * attributed to keywords in JS. Short plain words get a leading word boundary so `turn` does not
8
+ * match `return`. Per-file score = sum over DISTINCT matched keywords of IDF + path bonus + a small
9
+ * hit-count tie breaker; tests are down-weighted unless the question is about tests.
10
+ */
11
+ import type { CodeSearchConfig } from "./config";
12
+ import {
13
+ mapLimit,
14
+ READ_CONCURRENCY,
15
+ RIPGREP_SPLIT_CONCURRENCY,
16
+ type WorkspaceSession,
17
+ } from "./session";
18
+ import {
19
+ compoundFragments,
20
+ escapeRegex,
21
+ isChangelogPath,
22
+ isDocPath,
23
+ isShortPlainWord,
24
+ isTestPath,
25
+ keywordVariants,
26
+ questionMentionsHistory,
27
+ questionMentionsTests,
28
+ splitWords,
29
+ } from "./text";
30
+ import { CODE_SEARCH_MAX_PATTERN_CHARS } from "./workspace";
31
+
32
+ export const BUILTIN_EXCLUDES = [
33
+ "!**/node_modules/**",
34
+ "!**/dist/**",
35
+ "!**/build/**",
36
+ "!**/.next/**",
37
+ "!**/coverage/**",
38
+ "!**/target/**",
39
+ "!**/vendor/**",
40
+ "!**/.git/**",
41
+ // a linked worktree has a `.git` FILE (gitdir pointer), which --hidden would otherwise search
42
+ "!**/.git",
43
+ "!**/.turbo/**",
44
+ "!*.lock",
45
+ "!**/package-lock.json",
46
+ "!**/pnpm-lock.yaml",
47
+ "!**/yarn.lock",
48
+ "!**/bun.lockb",
49
+ "!*.min.js",
50
+ "!*.min.css",
51
+ "!*.map",
52
+ "!*.snap",
53
+ "!**/*.gen.*",
54
+ "!**/*.generated.*",
55
+ "!**/gen/**",
56
+ "!**/*_pb.*",
57
+ "!**/*.pb.go",
58
+ "!*.{svg,png,jpg,jpeg,gif,ico,webp,woff,woff2,ttf,otf,eot,wasm,pdf,zip,gz,tgz,mp4,mp3,mov}",
59
+ ];
60
+
61
+ export interface KeywordInfo {
62
+ index: number;
63
+ raw: string;
64
+ variants: string[];
65
+ /** "phrase" = any variant as a substring; "word-prefix" = `\bword`. */
66
+ mode: "phrase" | "word-prefix";
67
+ /** JS-dialect pattern (attribution, window scoring, line cutting). */
68
+ pattern: string;
69
+ /** ripgrep-dialect pattern: same language, but word boundaries are ASCII `(?-u:\b)`. */
70
+ rgPattern: string;
71
+ /** Number of files with at least one matching line (content). */
72
+ df: number;
73
+ /** Files whose path matches (no content needed). */
74
+ pathDf: number;
75
+ hitLines: number;
76
+ idf: number;
77
+ /** Fragments used because the full keyword had zero hits. */
78
+ fragments: string[];
79
+ }
80
+
81
+ export interface HitLine {
82
+ line: number;
83
+ text: string;
84
+ kws: number[];
85
+ }
86
+
87
+ export interface FileCandidate {
88
+ path: string;
89
+ lexScore: number;
90
+ /** keyword index -> matching line count */
91
+ kwHits: Record<number, number>;
92
+ pathKws: number[];
93
+ hitLines: Map<number, HitLine>;
94
+ isTest: boolean;
95
+ isDoc: boolean;
96
+ }
97
+
98
+ export interface RecallResult {
99
+ keywords: KeywordInfo[];
100
+ totalFiles: number;
101
+ candidates: FileCandidate[];
102
+ scoredFiles: number;
103
+ searchPaths: string[];
104
+ widened: boolean;
105
+ /** Path prefixes that do not exist in the workspace (or are not workspace-relative); ignored. */
106
+ missingPrefixes: string[];
107
+ /** The path prefixes that exist, workspace-relative. */
108
+ validPrefixes: string[];
109
+ /** Path-only candidates dropped because their content is binary (NUL in the first 8 KB). */
110
+ binaryDropped: number;
111
+ ms: number;
112
+ }
113
+
114
+ export interface RecallInput {
115
+ session: WorkspaceSession;
116
+ question: string;
117
+ keywords: string[];
118
+ pathPrefixes: string[];
119
+ config: CodeSearchConfig;
120
+ }
121
+
122
+ export function excludeArgs(cfg: CodeSearchConfig): string[] {
123
+ return [
124
+ ...BUILTIN_EXCLUDES,
125
+ ...cfg.recall.extraExcludes.map((g) => (g.startsWith("!") ? g : `!${g}`)),
126
+ ].flatMap((g) => ["-g", g]);
127
+ }
128
+
129
+ /** `--hidden` (rg skips dot-dirs such as .github/ by default); `.git/` stays excluded by BUILTIN_EXCLUDES. */
130
+ export function hiddenArgs(cfg: CodeSearchConfig): string[] {
131
+ return cfg.recall.searchHidden ? ["--hidden"] : [];
132
+ }
133
+
134
+ export async function listFiles(
135
+ session: WorkspaceSession,
136
+ paths: string[],
137
+ cfg: CodeSearchConfig,
138
+ ): Promise<string[]> {
139
+ // exit 2 without output means nothing could be listed (for example only unreadable directories)
140
+ const out = await session.ripgrep(
141
+ [
142
+ "--files",
143
+ "--no-require-git",
144
+ ...hiddenArgs(cfg),
145
+ "--max-filesize",
146
+ String(cfg.recall.maxFileBytes),
147
+ ...excludeArgs(cfg),
148
+ "--",
149
+ ...paths,
150
+ ],
151
+ { allowFailure: true },
152
+ );
153
+ return out
154
+ .split("\n")
155
+ .filter(Boolean)
156
+ .map((p) => p.replace(/^\.\//, ""))
157
+ .sort();
158
+ }
159
+
160
+ /**
161
+ * Search pattern for one keyword. `pattern` is JS-dialect; `rgPattern` is the same for ripgrep except that the
162
+ * word boundary is ASCII: a Unicode `\b` under -i disables ripgrep's fast engines (measured on repo-snap: 5.6 s
163
+ * user CPU for one pass vs 0.2 s with `(?-u:\b)`), and JS `\b` is ASCII anyway.
164
+ */
165
+ export function buildPattern(
166
+ kw: string,
167
+ cfg: CodeSearchConfig,
168
+ ): { pattern: string; rgPattern: string; mode: "phrase" | "word-prefix"; variants: string[] } {
169
+ const variants = keywordVariants(kw);
170
+ if (variants.length === 1 && isShortPlainWord(variants[0]!, cfg.recall.shortWordMaxLen)) {
171
+ const lit = escapeRegex(variants[0]!);
172
+ return { pattern: `\\b${lit}`, rgPattern: `(?-u:\\b)${lit}`, mode: "word-prefix", variants };
173
+ }
174
+ // spaced-phrase variants match any run of whitespace
175
+ const alts = variants.map((v) => escapeRegex(v).replace(/ /g, "\\s+"));
176
+ return { pattern: alts.join("|"), rgPattern: alts.join("|"), mode: "phrase", variants };
177
+ }
178
+
179
+ /** Pattern for the 2-word fragments of a compound keyword (used only if the full keyword has zero hits). */
180
+ export function fragmentPattern(frags: string[]): string {
181
+ return frags
182
+ .flatMap((f) => keywordVariants(f))
183
+ .map((v) => escapeRegex(v).replace(/ /g, "\\s+"))
184
+ .join("|");
185
+ }
186
+
187
+ export interface RgMatch {
188
+ path: string;
189
+ line: number;
190
+ text: string;
191
+ }
192
+
193
+ /** Parse `rg --null --line-number --with-filename --no-heading` output. Omitted long lines are skipped. */
194
+ export function parseRgOutput(out: string): RgMatch[] {
195
+ const matches: RgMatch[] = [];
196
+ let pos = 0;
197
+ while (pos < out.length) {
198
+ let nl = out.indexOf("\n", pos);
199
+ if (nl < 0) nl = out.length;
200
+ const z = out.indexOf("\0", pos);
201
+ if (z > pos && z < nl) {
202
+ const colon = out.indexOf(":", z + 1);
203
+ if (colon > z && colon < nl) {
204
+ const line = Number(out.slice(z + 1, colon));
205
+ const text = out.slice(colon + 1, nl).replace(/\r$/, "");
206
+ if (Number.isFinite(line) && !text.startsWith("[Omitted long line")) {
207
+ matches.push({ path: out.slice(pos, z).replace(/^\.\//, ""), line, text });
208
+ }
209
+ }
210
+ }
211
+ pos = nl + 1;
212
+ }
213
+ return matches;
214
+ }
215
+
216
+ const byPathLine = (a: RgMatch, b: RgMatch) =>
217
+ a.path < b.path ? -1 : a.path > b.path ? 1 : a.line - b.line;
218
+
219
+ /**
220
+ * Whether a pattern built by this engine compiles: checked as a JS regex after mapping ripgrep's inline flag
221
+ * groups (`(?-u:`, `(?i:`) to plain groups. It tells an invalid pattern apart from ripgrep's exit 2 for an
222
+ * unreadable path.
223
+ */
224
+ export function ripgrepPatternCompiles(pattern: string): boolean {
225
+ return jsRegex(pattern.replace(/\(\?[A-Za-z-]+:/g, "(?:")) !== null;
226
+ }
227
+
228
+ /** One rg pass for a pattern (any number of alternatives), all matching lines, sorted by path then line. */
229
+ export async function searchPattern(
230
+ session: WorkspaceSession,
231
+ pattern: string,
232
+ paths: string[],
233
+ cfg: CodeSearchConfig,
234
+ opts: { caseInsensitive?: boolean; word?: boolean; maxPerFile?: number } = {},
235
+ ): Promise<RgMatch[]> {
236
+ const args = [
237
+ "--null",
238
+ "--line-number",
239
+ "--with-filename",
240
+ "--no-heading",
241
+ "--color",
242
+ "never",
243
+ ...(opts.caseInsensitive === false ? [] : ["-i"]),
244
+ ...(opts.word ? ["-w"] : []),
245
+ "--no-require-git",
246
+ ...hiddenArgs(cfg),
247
+ ...(opts.maxPerFile ? ["-m", String(opts.maxPerFile)] : []),
248
+ // very long lines (minified / generated / data) are omitted by rg and skipped by the parser
249
+ "--max-columns",
250
+ String(cfg.recall.maxLineColumns),
251
+ "--max-filesize",
252
+ String(cfg.recall.maxFileBytes),
253
+ ...excludeArgs(cfg),
254
+ "-e",
255
+ pattern,
256
+ "--",
257
+ ...paths,
258
+ ];
259
+ // rg also exits 2 without output when nothing matched and some path was unreadable; that is no match
260
+ const out = await session.ripgrep(args, { allowFailure: ripgrepPatternCompiles(pattern) });
261
+ const matches = parseRgOutput(out);
262
+ // rg searches in parallel; sort for determinism (same effect as --sort path, without serializing the search)
263
+ matches.sort(byPathLine);
264
+ return matches;
265
+ }
266
+
267
+ /** Split a regex at its top-level `|` (outside groups, character classes and escapes). */
268
+ export function splitAlternation(pattern: string): string[] {
269
+ const out: string[] = [];
270
+ let depth = 0;
271
+ let inClass = false;
272
+ let start = 0;
273
+ for (let i = 0; i < pattern.length; i++) {
274
+ const ch = pattern[i];
275
+ if (ch === "\\") i++;
276
+ else if (inClass) inClass = ch !== "]";
277
+ else if (ch === "[") inClass = true;
278
+ else if (ch === "(") depth++;
279
+ else if (ch === ")") depth--;
280
+ else if (ch === "|" && depth === 0) {
281
+ out.push(pattern.slice(start, i));
282
+ start = i + 1;
283
+ }
284
+ }
285
+ out.push(pattern.slice(start));
286
+ return out;
287
+ }
288
+
289
+ /**
290
+ * Join alternatives with `|`, in order, into as few patterns of at most maxChars as possible. An alternative
291
+ * longer than maxChars on its own gets a pattern of its own.
292
+ */
293
+ export function packAlternatives(alternatives: readonly string[], maxChars: number): string[] {
294
+ const out: string[] = [];
295
+ let cur: string | null = null;
296
+ for (const alt of alternatives) {
297
+ if (cur !== null && cur.length + 1 + alt.length <= maxChars) {
298
+ cur += `|${alt}`;
299
+ } else {
300
+ if (cur !== null) out.push(cur);
301
+ cur = alt;
302
+ }
303
+ }
304
+ if (cur !== null) out.push(cur);
305
+ return out;
306
+ }
307
+
308
+ /** Matches of several searches, as one search over their union returns them: by path then line, each line once. */
309
+ export function mergeMatches(lists: readonly RgMatch[][]): RgMatch[] {
310
+ const all = lists.flat().sort(byPathLine);
311
+ return all.filter(
312
+ (m, i) => i === 0 || m.path !== all[i - 1]!.path || m.line !== all[i - 1]!.line,
313
+ );
314
+ }
315
+
316
+ /**
317
+ * The lines matching `(?:a)|(?:b)|...` over the alternatives. A union longer than maxPatternChars is split
318
+ * into several rg passes (an alternative too long on its own is split at its own top-level `|`), run a few
319
+ * at a time and merged, so the result is the same as one pass. A part that still does not fit (a keyword
320
+ * of thousands of characters) is not searched, so it reports zero hits.
321
+ */
322
+ export async function searchAlternatives(
323
+ session: WorkspaceSession,
324
+ alternatives: readonly string[],
325
+ paths: string[],
326
+ cfg: CodeSearchConfig,
327
+ maxPatternChars = CODE_SEARCH_MAX_PATTERN_CHARS,
328
+ ): Promise<RgMatch[]> {
329
+ const groups = alternatives
330
+ .flatMap((p) =>
331
+ p.length + 4 <= maxPatternChars ? [`(?:${p})`] : splitAlternation(p).map((a) => `(?:${a})`),
332
+ )
333
+ .filter((g) => g.length <= maxPatternChars);
334
+ const patterns = packAlternatives(groups, maxPatternChars);
335
+ if (patterns.length <= 1) {
336
+ return patterns.length ? searchPattern(session, patterns[0]!, paths, cfg) : [];
337
+ }
338
+ const found = await mapLimit(patterns, RIPGREP_SPLIT_CONCURRENCY, (p) =>
339
+ searchPattern(session, p, paths, cfg),
340
+ );
341
+ return mergeMatches(found);
342
+ }
343
+
344
+ export function idfOf(df: number, n: number): number {
345
+ return Math.log(1 + n / (1 + df));
346
+ }
347
+
348
+ /** Does the path mention the keyword (any variant, separators ignored)? */
349
+ export function pathMatches(path: string, variants: string[]): boolean {
350
+ const flatPath = path.toLowerCase().replace(/[-_ .]/g, "");
351
+ return variants.some((v) => {
352
+ const flat = v.toLowerCase().replace(/[-_ ]/g, "");
353
+ return flat.length >= 4 && flatPath.includes(flat);
354
+ });
355
+ }
356
+
357
+ /** JS regex for an rg pattern built by buildPattern/fragmentPattern (the escapes used are valid in both dialects). */
358
+ export function jsRegex(pattern: string): RegExp | null {
359
+ try {
360
+ return new RegExp(pattern, "i");
361
+ } catch {
362
+ return null;
363
+ }
364
+ }
365
+
366
+ export interface Slot {
367
+ /** keyword index */
368
+ ki: number;
369
+ fragment: boolean;
370
+ re: RegExp | null;
371
+ literals: string[];
372
+ }
373
+
374
+ /**
375
+ * Attribute rg lines to keyword slots. Returns per slot the matches (all lines; the caller caps what it stores).
376
+ * A JS regex that fails to compile falls back to a case-insensitive literal test on the variants.
377
+ */
378
+ export function attribute(matches: RgMatch[], slots: Slot[]): RgMatch[][] {
379
+ const out: RgMatch[][] = slots.map(() => []);
380
+ for (const m of matches) {
381
+ const lower = m.text.toLowerCase();
382
+ slots.forEach((s, i) => {
383
+ const hit = s.re ? s.re.test(m.text) : s.literals.some((l) => lower.includes(l));
384
+ if (hit) out[i]!.push(m);
385
+ });
386
+ }
387
+ return out;
388
+ }
389
+
390
+ /**
391
+ * Normalize path prefixes to workspace-relative form ("./src/" -> "src"); prefixes that do not exist, are
392
+ * absolute or climb out with ".." are reported as missing instead of crashing ripgrep (exit 2, no output).
393
+ */
394
+ export async function normalizePrefixes(
395
+ session: WorkspaceSession,
396
+ prefixes: string[],
397
+ ): Promise<{ ok: string[]; missing: string[] }> {
398
+ const cleaned = prefixes.map((raw) => ({ raw, rel: cleanPrefix(raw) }));
399
+ const lookup = [
400
+ ...new Set(cleaned.map((c) => c.rel).filter((r): r is string => r !== null && r !== ".")),
401
+ ];
402
+ const kinds = lookup.length ? await session.pathKinds(lookup) : {};
403
+ const ok: string[] = [];
404
+ const missing: string[] = [];
405
+ for (const { raw, rel } of cleaned) {
406
+ if (rel === ".") ok.push(".");
407
+ else if (rel !== null && (kinds[rel] === "file" || kinds[rel] === "directory")) ok.push(rel);
408
+ else missing.push(raw);
409
+ }
410
+ return { ok: [...new Set(ok)], missing };
411
+ }
412
+
413
+ /** Workspace-relative form of a prefix, or null when it is absolute, climbs out, or starts with "-". */
414
+ export function cleanPrefix(p: string): string | null {
415
+ let s = p.trim().replace(/\\/g, "/");
416
+ if (!s || s.startsWith("/") || s.startsWith("~") || /^[A-Za-z]:\//.test(s)) return null;
417
+ while (s.startsWith("./")) s = s.slice(2);
418
+ s = s.replace(/\/+$/, "").replace(/\/{2,}/g, "/");
419
+ if (s === "" || s === ".") return ".";
420
+ if (s.startsWith("-") || s.split("/").some((seg) => seg === "..")) return null;
421
+ return s;
422
+ }
423
+
424
+ /** Binary (NUL in the first 8 KB, ripgrep's own heuristic) or unreadable. */
425
+ async function looksBinary(
426
+ session: WorkspaceSession,
427
+ path: string,
428
+ bytes = 8192,
429
+ ): Promise<boolean> {
430
+ return (await session.readText(path, bytes)) === null;
431
+ }
432
+
433
+ export async function recall(input: RecallInput): Promise<RecallResult> {
434
+ const t0 = performance.now();
435
+ const cfg = input.config;
436
+ const session = input.session;
437
+ // any whitespace run (newline, tab) becomes one space: a literal newline is not allowed in an rg pattern
438
+ const kwsRaw = [
439
+ ...new Set(
440
+ input.keywords.map((k) => k.replace(/\s+/g, " ").trim()).filter((k) => k.length >= 2),
441
+ ),
442
+ ];
443
+ const prefixes = await normalizePrefixes(session, input.pathPrefixes);
444
+ let searchPaths = prefixes.ok.length ? prefixes.ok : ["."];
445
+ // every -p prefix missing: search the whole repository (reported as widened)
446
+ let widened = input.pathPrefixes.length > 0 && prefixes.ok.length === 0;
447
+
448
+ const keywordsFresh = (): KeywordInfo[] =>
449
+ kwsRaw.map((raw, index) => {
450
+ const { pattern, rgPattern, mode, variants } = buildPattern(raw, cfg);
451
+ return {
452
+ index,
453
+ raw,
454
+ variants,
455
+ mode,
456
+ pattern,
457
+ rgPattern,
458
+ df: 0,
459
+ pathDf: 0,
460
+ hitLines: 0,
461
+ idf: 0,
462
+ fragments: [],
463
+ };
464
+ });
465
+
466
+ /**
467
+ * ONE rg pass over the union of every keyword pattern (and, speculatively, the 2-word fragments of
468
+ * compound keywords); lines are then attributed to keywords in JS. ~0.2 s on a 6k-file repo versus
469
+ * ~1-2 s for one rg process per keyword. A union over the pattern cap takes a few passes instead.
470
+ */
471
+ const attempt = async (paths: string[]) => {
472
+ const keywords = keywordsFresh();
473
+ const slots: Slot[] = [];
474
+ const fragsOf = new Map<number, { frags: string[]; pattern: string }>();
475
+ for (const k of keywords) {
476
+ slots.push({
477
+ ki: k.index,
478
+ fragment: false,
479
+ re: jsRegex(k.pattern),
480
+ literals: k.variants.map((v) => v.toLowerCase()),
481
+ });
482
+ const frags = cfg.recall.fragmentFallback ? compoundFragments(k.raw) : [];
483
+ if (frags.length) {
484
+ const pattern = fragmentPattern(frags);
485
+ fragsOf.set(k.index, { frags, pattern });
486
+ slots.push({
487
+ ki: k.index,
488
+ fragment: true,
489
+ re: jsRegex(pattern),
490
+ literals: frags.flatMap((f) => keywordVariants(f)).map((v) => v.toLowerCase()),
491
+ });
492
+ }
493
+ }
494
+ const alternatives = [
495
+ ...keywords.map((k) => k.rgPattern),
496
+ ...[...fragsOf.values()].map((f) => f.pattern),
497
+ ];
498
+ const [files, matches] = await Promise.all([
499
+ listFiles(session, paths, cfg),
500
+ searchAlternatives(session, alternatives, paths, cfg),
501
+ ]);
502
+ const bySlot = attribute(matches, slots);
503
+ const results: RgMatch[][] = keywords.map(() => []);
504
+ slots.forEach((s, i) => {
505
+ if (!s.fragment) results[s.ki] = bySlot[i]!;
506
+ });
507
+ // zero-hit compound keywords use their fragments (the model guessed a name that does not exist)
508
+ slots.forEach((s, i) => {
509
+ if (s.fragment && results[s.ki]!.length === 0 && bySlot[i]!.length > 0) {
510
+ results[s.ki] = bySlot[i]!;
511
+ keywords[s.ki]!.fragments = fragsOf.get(s.ki)!.frags;
512
+ }
513
+ });
514
+ return { files, keywords, results };
515
+ };
516
+
517
+ let { files, keywords, results } = await attempt(searchPaths);
518
+ const countFiles = (rs: RgMatch[][]) => new Set(rs.flat().map((m) => m.path)).size;
519
+ if (prefixes.ok.length && countFiles(results) < cfg.recall.minCandidatesBeforeWiden) {
520
+ searchPaths = ["."];
521
+ widened = true;
522
+ ({ files, keywords, results } = await attempt(searchPaths));
523
+ }
524
+
525
+ const n = Math.max(files.length, 1);
526
+ const byPath = new Map<string, FileCandidate>();
527
+ const get = (path: string) => {
528
+ let c = byPath.get(path);
529
+ if (!c) {
530
+ c = {
531
+ path,
532
+ lexScore: 0,
533
+ kwHits: {},
534
+ pathKws: [],
535
+ hitLines: new Map(),
536
+ isTest: isTestPath(path),
537
+ isDoc: isDocPath(path),
538
+ };
539
+ byPath.set(path, c);
540
+ }
541
+ return c;
542
+ };
543
+ const kwFiles = keywords.map(() => new Set<string>());
544
+ const maxStored = cfg.recall.maxMatchesPerFile;
545
+ results.forEach((matches, ki) => {
546
+ const kw = keywords[ki]!;
547
+ const seen = kwFiles[ki]!;
548
+ for (const m of matches) {
549
+ seen.add(m.path);
550
+ const c = get(m.path);
551
+ const cnt = (c.kwHits[ki] = (c.kwHits[ki] ?? 0) + 1);
552
+ // every matching line counts for scoring; only the first maxMatchesPerFile per keyword are kept as hit lines
553
+ if (cnt > maxStored) continue;
554
+ let h = c.hitLines.get(m.line);
555
+ if (!h) {
556
+ h = { line: m.line, text: m.text, kws: [] };
557
+ c.hitLines.set(m.line, h);
558
+ }
559
+ if (!h.kws.includes(ki)) h.kws.push(ki);
560
+ }
561
+ kw.df = seen.size;
562
+ kw.hitLines = matches.length;
563
+ });
564
+ // path-name matches (also for files without content hits)
565
+ for (const kw of keywords) {
566
+ const variants = kw.fragments.length
567
+ ? kw.fragments.flatMap((f) => keywordVariants(f))
568
+ : kw.variants;
569
+ for (const f of files) {
570
+ if (pathMatches(f, variants)) {
571
+ kw.pathDf++;
572
+ kwFiles[kw.index]!.add(f);
573
+ const c = get(f);
574
+ if (!c.pathKws.includes(kw.index)) c.pathKws.push(kw.index);
575
+ }
576
+ }
577
+ }
578
+ // IDF over files matching by content OR path
579
+ for (const kw of keywords)
580
+ kw.idf = kwFiles[kw.index]!.size > 0 ? idfOf(kwFiles[kw.index]!.size, n) : 0;
581
+
582
+ const testsOk = questionMentionsTests(input.question);
583
+ const historyOk = questionMentionsHistory(input.question);
584
+ for (const c of byPath.values()) {
585
+ let s = 0;
586
+ let lines = 0;
587
+ for (const [ki, cnt] of Object.entries(c.kwHits)) {
588
+ s += keywords[Number(ki)]!.idf;
589
+ lines += cnt;
590
+ }
591
+ for (const ki of c.pathKws) s += cfg.recall.pathBonus * keywords[ki]!.idf;
592
+ s += cfg.recall.hitCountWeight * Math.log(1 + lines);
593
+ if (c.isTest && !testsOk) s *= cfg.recall.testWeight;
594
+ if (!historyOk && isChangelogPath(c.path)) s *= cfg.recall.changelogWeight;
595
+ c.lexScore = s;
596
+ }
597
+ const scored = [...byPath.values()].filter((c) => c.lexScore > 0);
598
+ scored.sort((a, b) => b.lexScore - a.lexScore || (a.path < b.path ? -1 : 1));
599
+ // `rg --files` lists binary files too; one that matches only by path (no content hit) would be windowed as text.
600
+ // Same result as a sequential scan: each batch holds at most the number of slots still open.
601
+ const candidates: FileCandidate[] = [];
602
+ let binaryDropped = 0;
603
+ let next = 0;
604
+ while (next < scored.length && candidates.length < cfg.recall.maxCandidates) {
605
+ const batch = scored.slice(next, next + cfg.recall.maxCandidates - candidates.length);
606
+ next += batch.length;
607
+ const pathOnly = batch.filter((c) => c.hitLines.size === 0);
608
+ const binary = new Set<FileCandidate>();
609
+ const flags = await mapLimit(pathOnly, READ_CONCURRENCY, (c) => looksBinary(session, c.path));
610
+ pathOnly.forEach((c, i) => {
611
+ if (flags[i]) binary.add(c);
612
+ });
613
+ for (const c of batch) {
614
+ if (binary.has(c)) binaryDropped++;
615
+ else candidates.push(c);
616
+ }
617
+ }
618
+ return {
619
+ keywords,
620
+ totalFiles: files.length,
621
+ candidates,
622
+ scoredFiles: scored.length,
623
+ searchPaths,
624
+ widened,
625
+ missingPrefixes: prefixes.missing,
626
+ validPrefixes: prefixes.ok,
627
+ binaryDropped,
628
+ ms: Math.round(performance.now() - t0),
629
+ };
630
+ }
631
+
632
+ /** Weight of one hit line: sum of IDF of the distinct keywords on it. */
633
+ export function lineWeight(h: HitLine, keywords: KeywordInfo[]): number {
634
+ return h.kws.reduce((s, k) => s + (keywords[k]?.idf ?? 0), 0);
635
+ }
636
+
637
+ /** Up to n best hit lines of a file (by distinct-keyword IDF), returned in line order (long lines are trimmed by the caller). */
638
+ export function bestHitLines(c: FileCandidate, keywords: KeywordInfo[], n: number): HitLine[] {
639
+ return [...c.hitLines.values()]
640
+ .sort((a, b) => lineWeight(b, keywords) - lineWeight(a, keywords) || a.line - b.line)
641
+ .slice(0, n)
642
+ .sort((a, b) => a.line - b.line);
643
+ }
644
+
645
+ /** Words of all keywords (for overlap scoring). */
646
+ export function keywordWords(keywords: KeywordInfo[]): Set<string> {
647
+ return new Set(keywords.flatMap((k) => splitWords(k.raw)));
648
+ }