@opengeni/jev 0.1.0-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +190 -0
- package/README.md +140 -0
- package/dist/circuit-breaker.d.ts +58 -0
- package/dist/client.d.ts +173 -0
- package/dist/code-search/config.d.ts +142 -0
- package/dist/code-search/judge.d.ts +169 -0
- package/dist/code-search/leads.d.ts +96 -0
- package/dist/code-search/pack.d.ts +106 -0
- package/dist/code-search/recall.d.ts +155 -0
- package/dist/code-search/search.d.ts +93 -0
- package/dist/code-search/session.d.ts +33 -0
- package/dist/code-search/text.d.ts +35 -0
- package/dist/code-search/tool.d.ts +34 -0
- package/dist/code-search/windows.d.ts +85 -0
- package/dist/code-search/workspace.d.ts +51 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +3110 -0
- package/dist/index.js.map +1 -0
- package/package.json +39 -0
- package/src/circuit-breaker.ts +135 -0
- package/src/client.ts +577 -0
- package/src/code-search/config.ts +282 -0
- package/src/code-search/judge.ts +413 -0
- package/src/code-search/leads.ts +442 -0
- package/src/code-search/pack.ts +354 -0
- package/src/code-search/recall.ts +648 -0
- package/src/code-search/search.ts +773 -0
- package/src/code-search/session.ts +89 -0
- package/src/code-search/text.ts +159 -0
- package/src/code-search/tool.ts +209 -0
- package/src/code-search/windows.ts +617 -0
- package/src/code-search/workspace.ts +55 -0
- package/src/index.ts +71 -0
|
@@ -0,0 +1,773 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* search.ts - the code_search pipeline (scout-0.3.1 with the Jev judge):
|
|
3
|
+
*
|
|
4
|
+
* 1. recall (ripgrep) -> top maxCandidates files by distinct-keyword IDF
|
|
5
|
+
* 2. wave 1 (file triage, 1 Noul per file) -> up to maxFiles files (+ lexical guard)
|
|
6
|
+
* 3. wave 2 (passage verification) -> relevance + per-sub-question coverage per passage
|
|
7
|
+
* 4. wave 3 (leads, exactly one round) -> <= maxLeadsFollowed definitions, verified like wave 2
|
|
8
|
+
* 5. pack (bounded, verbatim, line-numbered) + status (1 request over the packed evidence)
|
|
9
|
+
*
|
|
10
|
+
* Every file and process access goes through the injected CodeSearchWorkspace. A Jev failure fails the
|
|
11
|
+
* search (JevUnavailableError / JevRequestError propagate); only a failed final status check keeps the
|
|
12
|
+
* fully Jev-scored pack and reports the status as unknown.
|
|
13
|
+
*/
|
|
14
|
+
import { JevRequestError, JevUnavailableError, type JevClient } from "../client";
|
|
15
|
+
import { DEFAULT_CODE_SEARCH_CONFIG, type CodeSearchConfig } from "./config";
|
|
16
|
+
import {
|
|
17
|
+
JevJudge,
|
|
18
|
+
type FileItem,
|
|
19
|
+
type JudgeContext,
|
|
20
|
+
type LeadItem,
|
|
21
|
+
type PassageItem,
|
|
22
|
+
} from "./judge";
|
|
23
|
+
import { chooseDefinitions, extractLeads, locateDefinitions, type LeadCandidate } from "./leads";
|
|
24
|
+
import { packBody, renderFooter, type EvidencePassage } from "./pack";
|
|
25
|
+
import {
|
|
26
|
+
bestHitLines,
|
|
27
|
+
excludeArgs,
|
|
28
|
+
idfOf,
|
|
29
|
+
recall,
|
|
30
|
+
type FileCandidate,
|
|
31
|
+
type KeywordInfo,
|
|
32
|
+
type RecallResult,
|
|
33
|
+
} from "./recall";
|
|
34
|
+
import { mapLimit, READ_CONCURRENCY, WorkspaceSession } from "./session";
|
|
35
|
+
import {
|
|
36
|
+
contentTerms,
|
|
37
|
+
escapeRegex,
|
|
38
|
+
estTokens,
|
|
39
|
+
fmtK,
|
|
40
|
+
isDocPath,
|
|
41
|
+
keywordVariants,
|
|
42
|
+
overlap,
|
|
43
|
+
questionMentionsHistory,
|
|
44
|
+
questionMentionsTests,
|
|
45
|
+
splitWords,
|
|
46
|
+
STOPWORDS,
|
|
47
|
+
trimAround,
|
|
48
|
+
} from "./text";
|
|
49
|
+
import {
|
|
50
|
+
buildFileWindows,
|
|
51
|
+
definitionWindow,
|
|
52
|
+
keywordRegex,
|
|
53
|
+
keywordsInRange,
|
|
54
|
+
langOf,
|
|
55
|
+
renderLines,
|
|
56
|
+
splitLines,
|
|
57
|
+
type RenderOpts,
|
|
58
|
+
type Window,
|
|
59
|
+
} from "./windows";
|
|
60
|
+
import type { CodeSearchWorkspace } from "./workspace";
|
|
61
|
+
|
|
62
|
+
/** The validated research version this engine ports (ranking, thresholds and defaults are unchanged). */
|
|
63
|
+
export const CODE_SEARCH_ENGINE_VERSION = "scout-0.3.1";
|
|
64
|
+
export const CODE_SEARCH_DEFAULT_BUDGET_TOKENS = 12_000;
|
|
65
|
+
|
|
66
|
+
export interface CodeSearchInput {
|
|
67
|
+
question: string;
|
|
68
|
+
keywords: string[];
|
|
69
|
+
subQuestions?: string[] | undefined;
|
|
70
|
+
/** Workspace-relative path prefixes to search; default the whole workspace. */
|
|
71
|
+
paths?: string[] | undefined;
|
|
72
|
+
workspace: CodeSearchWorkspace;
|
|
73
|
+
jev: JevClient;
|
|
74
|
+
signal?: AbortSignal | undefined;
|
|
75
|
+
/** Max pack size in tokens (default 12000). */
|
|
76
|
+
budgetTokens?: number | undefined;
|
|
77
|
+
/** Tuning and tests only; defaults to DEFAULT_CODE_SEARCH_CONFIG. */
|
|
78
|
+
config?: CodeSearchConfig | undefined;
|
|
79
|
+
/** In-memory stage trace (recall, wave1, wave2, leads, pack, summary, and one "jev" event per Jev request). */
|
|
80
|
+
onStage?: ((stage: string, data: Record<string, unknown>) => void) | undefined;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export interface CodeSearchStatus {
|
|
84
|
+
label: "sufficient" | "partial" | "insufficient" | "unknown";
|
|
85
|
+
overall: number | null;
|
|
86
|
+
subs: Array<number | null>;
|
|
87
|
+
/** Set when the final Jev sufficiency check failed (the pack itself is fully Jev-scored). */
|
|
88
|
+
error?: string;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
export interface CodeSearchStats {
|
|
92
|
+
wallMs: number;
|
|
93
|
+
stageMs: Record<string, number>;
|
|
94
|
+
candidates: number;
|
|
95
|
+
filesSelected: number;
|
|
96
|
+
passagesVerified: number;
|
|
97
|
+
passagesIncluded: number;
|
|
98
|
+
packChars: number;
|
|
99
|
+
packTokensEst: number;
|
|
100
|
+
workspaceCalls: number;
|
|
101
|
+
/** A ripgrep call returned partial output (byte cap or time limit); the pack header says so. */
|
|
102
|
+
ripgrepTruncated: boolean;
|
|
103
|
+
jev: { requests: number; inputTokens: number; costUsd: number; model: string | null };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
export interface CodeSearchResult {
|
|
107
|
+
version: string;
|
|
108
|
+
/** The rendered pack: a status header line, verbatim line-numbered passages, then the footer. */
|
|
109
|
+
text: string;
|
|
110
|
+
status: CodeSearchStatus;
|
|
111
|
+
stats: CodeSearchStats;
|
|
112
|
+
/** The error of a failed final status check, for a circuit breaker (the search itself succeeded). */
|
|
113
|
+
statusCheckError?: JevUnavailableError | JevRequestError;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
const r2 = (x: number | null | undefined) =>
|
|
117
|
+
x === null || x === undefined || !Number.isFinite(x) ? "?" : x.toFixed(2);
|
|
118
|
+
|
|
119
|
+
/** Lexical passage relevance in [0,1]: keyword IDF mass present, question-term overlap, file prior. */
|
|
120
|
+
export function lexicalPassageScore(
|
|
121
|
+
text: string,
|
|
122
|
+
kwPresent: number[],
|
|
123
|
+
keywords: KeywordInfo[],
|
|
124
|
+
qTerms: string[],
|
|
125
|
+
fileNorm: number,
|
|
126
|
+
): number {
|
|
127
|
+
const total = keywords.reduce((s, k) => s + k.idf, 0) || 1;
|
|
128
|
+
const mass = kwPresent.reduce((s, k) => s + keywords[k]!.idf, 0) / total;
|
|
129
|
+
const qov = overlap(qTerms, new Set(contentTerms(text)));
|
|
130
|
+
return Math.max(0, Math.min(1, 0.55 * mass + 0.3 * qov + 0.15 * fileNorm));
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
export function selectFiles(
|
|
134
|
+
cands: FileCandidate[],
|
|
135
|
+
scores: Map<string, number>,
|
|
136
|
+
ids: string[],
|
|
137
|
+
T1: number,
|
|
138
|
+
cfg: CodeSearchConfig,
|
|
139
|
+
): { selected: number[]; ranked: number[] } {
|
|
140
|
+
// cands are in lexical order; ids[i] is the judge id of cands[i]
|
|
141
|
+
const ranked = cands
|
|
142
|
+
.map((_, i) => i)
|
|
143
|
+
.sort(
|
|
144
|
+
(a, b) =>
|
|
145
|
+
(scores.get(ids[b]!) ?? 0) - (scores.get(ids[a]!) ?? 0) ||
|
|
146
|
+
cands[b]!.lexScore - cands[a]!.lexScore ||
|
|
147
|
+
a - b,
|
|
148
|
+
);
|
|
149
|
+
const selected: number[] = [];
|
|
150
|
+
for (let i = 0; i < Math.min(cfg.wave1.lexicalGuard, cands.length); i++) selected.push(i);
|
|
151
|
+
ranked.forEach((i, rank) => {
|
|
152
|
+
if (selected.length >= cfg.wave1.maxFiles || selected.includes(i)) return;
|
|
153
|
+
if ((scores.get(ids[i]!) ?? 0) >= T1 || rank < cfg.wave1.minFiles) selected.push(i);
|
|
154
|
+
});
|
|
155
|
+
// priority order for windowing = judge rank order
|
|
156
|
+
selected.sort((a, b) => ranked.indexOf(a) - ranked.indexOf(b));
|
|
157
|
+
return { selected, ranked };
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/** Round-robin across files (priority order), each file's windows by score, up to max. */
|
|
161
|
+
export function capPassages<T extends { score: number }>(perFile: T[][], max: number): T[][] {
|
|
162
|
+
const sorted = perFile.map((ws) => [...ws].sort((a, b) => b.score - a.score));
|
|
163
|
+
const out: T[][] = perFile.map(() => []);
|
|
164
|
+
let taken = 0;
|
|
165
|
+
for (let pass = 0; taken < max; pass++) {
|
|
166
|
+
let any = false;
|
|
167
|
+
for (let f = 0; f < sorted.length && taken < max; f++) {
|
|
168
|
+
const w = sorted[f]![pass];
|
|
169
|
+
if (w) {
|
|
170
|
+
out[f]!.push(w);
|
|
171
|
+
taken++;
|
|
172
|
+
any = true;
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
if (!any) break;
|
|
176
|
+
}
|
|
177
|
+
return out;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** Evidence text for the status check: included passages, best first, up to maxChars (whole blocks only). */
|
|
181
|
+
export function statusEvidence(
|
|
182
|
+
included: Array<{ block: string; rel: number }>,
|
|
183
|
+
maxChars: number,
|
|
184
|
+
): string {
|
|
185
|
+
const out: string[] = [];
|
|
186
|
+
let used = 0;
|
|
187
|
+
for (const p of [...included].sort((a, b) => b.rel - a.rel)) {
|
|
188
|
+
if (used + p.block.length + 2 > maxChars) continue;
|
|
189
|
+
out.push(p.block);
|
|
190
|
+
used += p.block.length + 2;
|
|
191
|
+
}
|
|
192
|
+
return out.join("\n\n");
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
export function statusLabel(
|
|
196
|
+
s: { overall: number; subs: number[] } | null,
|
|
197
|
+
cfg: CodeSearchConfig,
|
|
198
|
+
): CodeSearchStatus {
|
|
199
|
+
if (!s || !Number.isFinite(s.overall)) return { label: "unknown", overall: null, subs: [] };
|
|
200
|
+
const { hi, lo } = cfg.status;
|
|
201
|
+
const subsOk = s.subs.every((x) => x >= hi);
|
|
202
|
+
const anySub = s.subs.some((x) => x >= hi);
|
|
203
|
+
const label =
|
|
204
|
+
s.overall >= hi && subsOk
|
|
205
|
+
? "sufficient"
|
|
206
|
+
: s.overall >= lo || anySub
|
|
207
|
+
? "partial"
|
|
208
|
+
: "insufficient";
|
|
209
|
+
return { label, overall: s.overall, subs: s.subs };
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** Footer note about paths that were missing, or a search that had to widen to the whole workspace. */
|
|
213
|
+
export function prefixNote(
|
|
214
|
+
rec: Pick<RecallResult, "missingPrefixes" | "widened" | "validPrefixes">,
|
|
215
|
+
): string | undefined {
|
|
216
|
+
const notes: string[] = [];
|
|
217
|
+
if (rec.missingPrefixes.length) {
|
|
218
|
+
notes.push(
|
|
219
|
+
`Note: paths not found in the workspace (ignored): ${rec.missingPrefixes.join(", ")}.`,
|
|
220
|
+
);
|
|
221
|
+
}
|
|
222
|
+
if (rec.widened) {
|
|
223
|
+
notes.push(
|
|
224
|
+
!rec.validPrefixes.length
|
|
225
|
+
? "Note: the whole workspace was searched."
|
|
226
|
+
: "Note: nothing matched under the given paths; the whole workspace was searched.",
|
|
227
|
+
);
|
|
228
|
+
}
|
|
229
|
+
return notes.length ? notes.join(" ") : undefined;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/** Header label for partial ripgrep output, in the same position scout used for its other partial states. */
|
|
233
|
+
function partialLabel(session: WorkspaceSession): string {
|
|
234
|
+
if (!session.partial) return "";
|
|
235
|
+
const why = session.timedOut
|
|
236
|
+
? "a ripgrep search hit its time limit"
|
|
237
|
+
: "ripgrep output was cut at its size limit";
|
|
238
|
+
return `(partial search: ${why}; some matches may be missing)`;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export async function runCodeSearch(input: CodeSearchInput): Promise<CodeSearchResult> {
|
|
242
|
+
const outer = input.signal;
|
|
243
|
+
outer?.throwIfAborted();
|
|
244
|
+
// One controller per search: an error in one parallel branch cancels the others.
|
|
245
|
+
const controller = new AbortController();
|
|
246
|
+
const onOuterAbort = () => controller.abort(outer?.reason);
|
|
247
|
+
outer?.addEventListener("abort", onOuterAbort, { once: true });
|
|
248
|
+
try {
|
|
249
|
+
return await pipeline(input, controller.signal);
|
|
250
|
+
} catch (error) {
|
|
251
|
+
controller.abort(error);
|
|
252
|
+
if (outer?.aborted) throw outer.reason;
|
|
253
|
+
throw error;
|
|
254
|
+
} finally {
|
|
255
|
+
outer?.removeEventListener("abort", onOuterAbort);
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
async function pipeline(o: CodeSearchInput, signal: AbortSignal): Promise<CodeSearchResult> {
|
|
260
|
+
const cfg = o.config ?? DEFAULT_CODE_SEARCH_CONFIG;
|
|
261
|
+
const t0 = performance.now();
|
|
262
|
+
const stageMs: Record<string, number> = {};
|
|
263
|
+
const mark = (stage: string, since: number) => {
|
|
264
|
+
stageMs[stage] = Math.round(performance.now() - since);
|
|
265
|
+
};
|
|
266
|
+
const emit = (stage: string, data: Record<string, unknown>) => {
|
|
267
|
+
try {
|
|
268
|
+
o.onStage?.(stage, data);
|
|
269
|
+
} catch {
|
|
270
|
+
// tracing never breaks a search
|
|
271
|
+
}
|
|
272
|
+
};
|
|
273
|
+
const session = new WorkspaceSession(o.workspace, signal, cfg.recall.ripgrepTimeoutMs);
|
|
274
|
+
const judge = new JevJudge({ client: o.jev, config: cfg, signal, onEvent: emit });
|
|
275
|
+
const subQuestions = (o.subQuestions ?? [])
|
|
276
|
+
.map((s) => s.trim())
|
|
277
|
+
.filter(Boolean)
|
|
278
|
+
.slice(0, 4);
|
|
279
|
+
const ctx: JudgeContext = { question: o.question.trim(), subQuestions };
|
|
280
|
+
const budgetTokens = o.budgetTokens ?? CODE_SEARCH_DEFAULT_BUDGET_TOKENS;
|
|
281
|
+
const thr = cfg.thresholds;
|
|
282
|
+
emit("start", {
|
|
283
|
+
version: CODE_SEARCH_ENGINE_VERSION,
|
|
284
|
+
question: ctx.question,
|
|
285
|
+
subQuestions,
|
|
286
|
+
keywords: o.keywords,
|
|
287
|
+
paths: o.paths ?? [],
|
|
288
|
+
budgetTokens,
|
|
289
|
+
});
|
|
290
|
+
// open keep-alive connections while recall runs
|
|
291
|
+
o.jev.warmUp(cfg.jev.warmConnections);
|
|
292
|
+
|
|
293
|
+
// ---- 1. recall
|
|
294
|
+
let ts = performance.now();
|
|
295
|
+
const rec: RecallResult = await recall({
|
|
296
|
+
session,
|
|
297
|
+
question: ctx.question + " " + subQuestions.join(" "),
|
|
298
|
+
keywords: o.keywords,
|
|
299
|
+
pathPrefixes: o.paths ?? [],
|
|
300
|
+
config: cfg,
|
|
301
|
+
});
|
|
302
|
+
mark("recall", ts);
|
|
303
|
+
const kws = rec.keywords;
|
|
304
|
+
emit("recall", {
|
|
305
|
+
ms: rec.ms,
|
|
306
|
+
totalFiles: rec.totalFiles,
|
|
307
|
+
scoredFiles: rec.scoredFiles,
|
|
308
|
+
searchPaths: rec.searchPaths,
|
|
309
|
+
widened: rec.widened,
|
|
310
|
+
missingPrefixes: rec.missingPrefixes,
|
|
311
|
+
binaryDropped: rec.binaryDropped,
|
|
312
|
+
partial: session.partial,
|
|
313
|
+
keywords: kws.map((k) => ({
|
|
314
|
+
raw: k.raw,
|
|
315
|
+
mode: k.mode,
|
|
316
|
+
pattern: k.pattern,
|
|
317
|
+
df: k.df,
|
|
318
|
+
pathDf: k.pathDf,
|
|
319
|
+
hitLines: k.hitLines,
|
|
320
|
+
idf: round(k.idf),
|
|
321
|
+
fragments: k.fragments,
|
|
322
|
+
})),
|
|
323
|
+
candidates: rec.candidates.map((c) => ({
|
|
324
|
+
path: c.path,
|
|
325
|
+
lex: round(c.lexScore),
|
|
326
|
+
kws: Object.keys(c.kwHits).map(Number),
|
|
327
|
+
pathKws: c.pathKws,
|
|
328
|
+
lines: c.hitLines.size,
|
|
329
|
+
test: c.isTest,
|
|
330
|
+
})),
|
|
331
|
+
});
|
|
332
|
+
|
|
333
|
+
// ---- 2. wave 1: file triage
|
|
334
|
+
ts = performance.now();
|
|
335
|
+
const maxLex = rec.candidates[0]?.lexScore || 1;
|
|
336
|
+
const fileIds = rec.candidates.map((_, i) => `f${String(i).padStart(3, "0")}`);
|
|
337
|
+
const fileItems: FileItem[] = rec.candidates.map((c, i) => {
|
|
338
|
+
const hits = bestHitLines(c, kws, cfg.wave1.hitLinesPerFile);
|
|
339
|
+
const needles = Object.keys(c.kwHits).flatMap((k) => kws[Number(k)]!.variants);
|
|
340
|
+
const lines = hits.map(
|
|
341
|
+
(h) => ` ${h.line}: ${trimAround(h.text, needles, cfg.wave1.hitLineChars)}`,
|
|
342
|
+
);
|
|
343
|
+
return {
|
|
344
|
+
id: fileIds[i]!,
|
|
345
|
+
path: c.path,
|
|
346
|
+
descriptor: [c.path, ...lines].join("\n"),
|
|
347
|
+
lex: c.lexScore / maxLex,
|
|
348
|
+
};
|
|
349
|
+
});
|
|
350
|
+
const fileScores = await judge.scoreFiles(fileItems, ctx);
|
|
351
|
+
const { selected, ranked } = selectFiles(rec.candidates, fileScores, fileIds, thr.T1, cfg);
|
|
352
|
+
mark("wave1", ts);
|
|
353
|
+
emit("wave1", {
|
|
354
|
+
T1: thr.T1,
|
|
355
|
+
scores: rec.candidates.map((c, i) => ({
|
|
356
|
+
path: c.path,
|
|
357
|
+
p: round(fileScores.get(fileIds[i]!) ?? Number.NaN),
|
|
358
|
+
lex: round(fileItems[i]!.lex),
|
|
359
|
+
})),
|
|
360
|
+
selected: selected.map((i) => rec.candidates[i]!.path),
|
|
361
|
+
});
|
|
362
|
+
|
|
363
|
+
// ---- 3. wave 2: windows + verification
|
|
364
|
+
ts = performance.now();
|
|
365
|
+
// file contents are read in parallel up front; the windowing below is synchronous
|
|
366
|
+
const fileLines = new Map<string, string[]>();
|
|
367
|
+
const loadLines = async (paths: string[]) => {
|
|
368
|
+
const missing = [...new Set(paths)].filter((p) => !fileLines.has(p));
|
|
369
|
+
const texts = await mapLimit(missing, READ_CONCURRENCY, (p) =>
|
|
370
|
+
session.readText(p, cfg.recall.maxFileBytes),
|
|
371
|
+
);
|
|
372
|
+
missing.forEach((p, i) => {
|
|
373
|
+
const text = texts[i];
|
|
374
|
+
// binary content (NUL bytes) is never rendered as a passage
|
|
375
|
+
fileLines.set(p, text === null || text === undefined ? [] : splitLines(text));
|
|
376
|
+
});
|
|
377
|
+
};
|
|
378
|
+
const readLines = (path: string): string[] => fileLines.get(path) ?? [];
|
|
379
|
+
await loadLines(selected.map((i) => rec.candidates[i]!.path));
|
|
380
|
+
const kwNeedles = kws
|
|
381
|
+
.filter((k) => k.idf > 0)
|
|
382
|
+
.map((k) => keywordRegex(k))
|
|
383
|
+
.filter((re): re is RegExp => re !== null);
|
|
384
|
+
const renderFor = (path: string, needles: RegExp[] = kwNeedles): RenderOpts => ({
|
|
385
|
+
maxLineChars: isDocPath(path) ? cfg.wave2.maxProseLineChars : cfg.wave2.maxLineChars,
|
|
386
|
+
needles,
|
|
387
|
+
});
|
|
388
|
+
const perFileWindows: Window[][] = selected.map((i) => {
|
|
389
|
+
const c = rec.candidates[i]!;
|
|
390
|
+
const lines = readLines(c.path);
|
|
391
|
+
const hits = [...c.hitLines.values()];
|
|
392
|
+
return buildFileWindows(lines, hits, kws, langOf(c.path), cfg, renderFor(c.path));
|
|
393
|
+
});
|
|
394
|
+
const capped = capPassages(perFileWindows, cfg.wave2.maxPassages);
|
|
395
|
+
const qTerms = contentTerms(ctx.question);
|
|
396
|
+
const subTerms = subQuestions.map((s) => contentTerms(s));
|
|
397
|
+
const evidence: EvidencePassage[] = [];
|
|
398
|
+
const passageItems: PassageItem[] = [];
|
|
399
|
+
let pid = 0;
|
|
400
|
+
const makePassage = (path: string, w: Window, fileNorm: number, lead?: string) => {
|
|
401
|
+
const lines = readLines(path);
|
|
402
|
+
const render = renderFor(path, lead ? [new RegExp(`\\b${escapeRegex(lead)}\\b`)] : kwNeedles);
|
|
403
|
+
const text = renderLines(lines, w.start, w.end, render);
|
|
404
|
+
const raw = lines.slice(w.start - 1, w.end).join("\n");
|
|
405
|
+
const present = keywordsInRange(lines, w.start, w.end, kws);
|
|
406
|
+
const rawTerms = new Set(contentTerms(raw));
|
|
407
|
+
const id = `p${String(pid++).padStart(3, "0")}`;
|
|
408
|
+
const lex = lexicalPassageScore(raw, present, kws, qTerms, fileNorm);
|
|
409
|
+
const lexCov = subTerms.map((t) => overlap(t, rawTerms));
|
|
410
|
+
const label =
|
|
411
|
+
w.label && w.label.line < w.start ? `L${w.label.line}: ${w.label.text}` : undefined;
|
|
412
|
+
passageItems.push({ id, path, start: w.start, end: w.end, text, label, lex, lexCov });
|
|
413
|
+
evidence.push({
|
|
414
|
+
id,
|
|
415
|
+
path,
|
|
416
|
+
start: w.start,
|
|
417
|
+
end: w.end,
|
|
418
|
+
fileLines: lines,
|
|
419
|
+
hits: w.hits,
|
|
420
|
+
label: w.label,
|
|
421
|
+
kind: w.kind,
|
|
422
|
+
rel: lex,
|
|
423
|
+
cov: lexCov,
|
|
424
|
+
lex,
|
|
425
|
+
lead,
|
|
426
|
+
render,
|
|
427
|
+
});
|
|
428
|
+
};
|
|
429
|
+
selected.forEach((ci, f) => {
|
|
430
|
+
const c = rec.candidates[ci]!;
|
|
431
|
+
for (const w of [...capped[f]!].sort((a, b) => a.start - b.start))
|
|
432
|
+
makePassage(c.path, w, c.lexScore / maxLex);
|
|
433
|
+
});
|
|
434
|
+
const wave2Scores = await judge.scorePassages(passageItems, ctx, "wave2");
|
|
435
|
+
for (const e of evidence) {
|
|
436
|
+
const s = wave2Scores.get(e.id);
|
|
437
|
+
if (s) {
|
|
438
|
+
e.rel = s.rel;
|
|
439
|
+
e.cov = s.cov;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
mark("wave2", ts);
|
|
443
|
+
emit("wave2", {
|
|
444
|
+
T2: thr.T2,
|
|
445
|
+
passages: evidence.map((e) => ({
|
|
446
|
+
id: e.id,
|
|
447
|
+
path: e.path,
|
|
448
|
+
start: e.start,
|
|
449
|
+
end: e.end,
|
|
450
|
+
kind: e.kind,
|
|
451
|
+
hits: e.hits.length,
|
|
452
|
+
rel: round(e.rel),
|
|
453
|
+
cov: e.cov.map(round),
|
|
454
|
+
lex: round(e.lex ?? Number.NaN),
|
|
455
|
+
})),
|
|
456
|
+
});
|
|
457
|
+
|
|
458
|
+
// ---- 4. wave 3: leads (exactly one round)
|
|
459
|
+
ts = performance.now();
|
|
460
|
+
let leadCands: LeadCandidate[] = [];
|
|
461
|
+
let leadScores = new Map<string, number>();
|
|
462
|
+
const chosenNames = new Set<string>();
|
|
463
|
+
if (cfg.wave3.enabled) {
|
|
464
|
+
const leadsFollowed: Array<{ name: string; score: number; def: string | null }> = [];
|
|
465
|
+
const seeds = evidence
|
|
466
|
+
.filter((e) => e.rel >= thr.T2)
|
|
467
|
+
.sort((a, b) => b.rel - a.rel)
|
|
468
|
+
.slice(0, cfg.wave3.seedPassages)
|
|
469
|
+
.map((e) => ({
|
|
470
|
+
path: e.path,
|
|
471
|
+
start: e.start,
|
|
472
|
+
lines: e.fileLines.slice(e.start - 1, e.end),
|
|
473
|
+
rel: e.rel,
|
|
474
|
+
}));
|
|
475
|
+
const searched = new Set<string>();
|
|
476
|
+
for (const k of kws) {
|
|
477
|
+
for (const v of [...k.variants, ...k.fragments.flatMap((f) => keywordVariants(f))]) {
|
|
478
|
+
searched.add(v.toLowerCase());
|
|
479
|
+
searched.add(splitWords(v).join(" "));
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
const qWords = new Set(
|
|
483
|
+
[
|
|
484
|
+
...splitWords(ctx.question + " " + subQuestions.join(" ")),
|
|
485
|
+
...kws.flatMap((k) => splitWords(k.raw)),
|
|
486
|
+
].filter((w) => w.length >= 3 && !STOPWORDS.has(w)),
|
|
487
|
+
);
|
|
488
|
+
// extract generously, keep only leads whose definition exists and is not already in the evidence
|
|
489
|
+
const extracted = extractLeads(
|
|
490
|
+
seeds,
|
|
491
|
+
searched,
|
|
492
|
+
qWords,
|
|
493
|
+
Math.ceil(cfg.wave3.maxLeadCandidates * 1.5),
|
|
494
|
+
);
|
|
495
|
+
const selectedPaths = new Set(selected.map((i) => rec.candidates[i]!.path));
|
|
496
|
+
const allowTests = questionMentionsTests(ctx.question);
|
|
497
|
+
const defSearch = await locateDefinitions(
|
|
498
|
+
session,
|
|
499
|
+
extracted.map((l) => l.name),
|
|
500
|
+
cfg,
|
|
501
|
+
excludeArgs(cfg),
|
|
502
|
+
12,
|
|
503
|
+
allowTests,
|
|
504
|
+
);
|
|
505
|
+
const seenAt = new Map(extracted.map((l) => [l.name, l.seenAt.path]));
|
|
506
|
+
const defs = chooseDefinitions(
|
|
507
|
+
defSearch.hits,
|
|
508
|
+
selectedPaths,
|
|
509
|
+
cfg.wave3.defsPerLead,
|
|
510
|
+
allowTests,
|
|
511
|
+
seenAt,
|
|
512
|
+
);
|
|
513
|
+
const inEvidence = (path: string, line: number) =>
|
|
514
|
+
evidence.some((e) => e.path === path && line >= e.start && line <= e.end);
|
|
515
|
+
const leadDrops: Record<string, string> = {};
|
|
516
|
+
// genericity penalty: a name defined/used as a key in many files (sessionId, workspaceId, isRecord) is rarely
|
|
517
|
+
// what the answer hinges on
|
|
518
|
+
const nFiles = Math.max(rec.totalFiles, 1);
|
|
519
|
+
const genericity = (name: string) =>
|
|
520
|
+
idfOf(defSearch.fileCounts.get(name) ?? 0, nFiles) / idfOf(0, nFiles);
|
|
521
|
+
leadCands = extracted
|
|
522
|
+
.filter((l) => {
|
|
523
|
+
const ds = defs.get(l.name) ?? [];
|
|
524
|
+
if (!ds.length) leadDrops[l.name] = "no definition found";
|
|
525
|
+
else if (ds.every((d) => inEvidence(d.path, d.line)))
|
|
526
|
+
leadDrops[l.name] = "definition already in evidence";
|
|
527
|
+
return ds.length > 0 && !ds.every((d) => inEvidence(d.path, d.line));
|
|
528
|
+
})
|
|
529
|
+
.map((l) => ({ ...l, weight: Math.round(l.weight * genericity(l.name) * 1000) / 1000 }))
|
|
530
|
+
.sort((a, b) => b.weight - a.weight || (a.name < b.name ? -1 : 1))
|
|
531
|
+
.slice(0, cfg.wave3.maxLeadCandidates);
|
|
532
|
+
const maxW = Math.max(...leadCands.map((l) => l.weight), 1e-9);
|
|
533
|
+
const leadItems: LeadItem[] = leadCands.map((l, i) => ({
|
|
534
|
+
id: `l${String(i).padStart(3, "0")}`,
|
|
535
|
+
name: l.name,
|
|
536
|
+
seenAt: `${l.seenAt.path}:${l.seenAt.line}`,
|
|
537
|
+
context: l.context,
|
|
538
|
+
lex: l.weight / maxW,
|
|
539
|
+
}));
|
|
540
|
+
leadScores = await judge.scoreLeads(leadItems, ctx);
|
|
541
|
+
const chosen = leadItems
|
|
542
|
+
.map((l) => ({ l, p: leadScores.get(l.id) ?? 0 }))
|
|
543
|
+
.filter((x) => x.p >= thr.T3)
|
|
544
|
+
.sort((a, b) => b.p - a.p || (a.l.id < b.l.id ? -1 : 1))
|
|
545
|
+
.slice(0, cfg.wave3.maxLeadsFollowed);
|
|
546
|
+
for (const x of chosen) chosenNames.add(x.l.name);
|
|
547
|
+
await loadLines(chosen.flatMap((x) => (defs.get(x.l.name) ?? []).map((d) => d.path)));
|
|
548
|
+
const defItemsStart = passageItems.length;
|
|
549
|
+
for (const x of chosen) {
|
|
550
|
+
for (const d of defs.get(x.l.name) ?? []) {
|
|
551
|
+
const lines = readLines(d.path);
|
|
552
|
+
// the file could not be read as text (missing, binary, UTF-16) or changed after ripgrep saw it
|
|
553
|
+
if (d.line > lines.length) {
|
|
554
|
+
leadsFollowed.push({
|
|
555
|
+
name: x.l.name,
|
|
556
|
+
score: x.p,
|
|
557
|
+
def: `${d.path}:${d.line} (not in the file as read)`,
|
|
558
|
+
});
|
|
559
|
+
continue;
|
|
560
|
+
}
|
|
561
|
+
const w = definitionWindow(
|
|
562
|
+
lines,
|
|
563
|
+
d.line,
|
|
564
|
+
langOf(d.path),
|
|
565
|
+
cfg,
|
|
566
|
+
renderFor(d.path, [new RegExp(`\\b${escapeRegex(x.l.name)}\\b`)]),
|
|
567
|
+
);
|
|
568
|
+
const overlapsExisting = evidence.some(
|
|
569
|
+
(e) => e.path === d.path && !(w.end < e.start || w.start > e.end),
|
|
570
|
+
);
|
|
571
|
+
leadsFollowed.push({
|
|
572
|
+
name: x.l.name,
|
|
573
|
+
score: x.p,
|
|
574
|
+
def: `${d.path}:${d.line}${overlapsExisting ? " (overlaps evidence)" : ""}`,
|
|
575
|
+
});
|
|
576
|
+
if (overlapsExisting) continue;
|
|
577
|
+
makePassage(d.path, w, 0, x.l.name);
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
const defItems = passageItems.slice(defItemsStart);
|
|
581
|
+
if (defItems.length) {
|
|
582
|
+
const defScores = await judge.scorePassages(defItems, ctx, "lead_defs");
|
|
583
|
+
for (const e of evidence) {
|
|
584
|
+
const s = defScores.get(e.id);
|
|
585
|
+
if (s) {
|
|
586
|
+
e.rel = s.rel;
|
|
587
|
+
e.cov = s.cov;
|
|
588
|
+
}
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
emit("leads", {
|
|
592
|
+
T3: thr.T3,
|
|
593
|
+
seeds: seeds.map((s) => `${s.path}:${s.start}`),
|
|
594
|
+
candidates: leadItems.map((l) => ({
|
|
595
|
+
id: l.id,
|
|
596
|
+
name: l.name,
|
|
597
|
+
seenAt: l.seenAt,
|
|
598
|
+
lex: round(l.lex),
|
|
599
|
+
p: round(leadScores.get(l.id) ?? Number.NaN),
|
|
600
|
+
})),
|
|
601
|
+
followed: leadsFollowed,
|
|
602
|
+
defsFound: defSearch.hits.length,
|
|
603
|
+
defSearchMs: defSearch.ms,
|
|
604
|
+
extracted: extracted.length,
|
|
605
|
+
dropped: leadDrops,
|
|
606
|
+
defPassages: evidence
|
|
607
|
+
.filter((e) => e.kind === "def")
|
|
608
|
+
.map((e) => ({
|
|
609
|
+
id: e.id,
|
|
610
|
+
path: e.path,
|
|
611
|
+
start: e.start,
|
|
612
|
+
end: e.end,
|
|
613
|
+
lead: e.lead,
|
|
614
|
+
rel: round(e.rel),
|
|
615
|
+
})),
|
|
616
|
+
});
|
|
617
|
+
}
|
|
618
|
+
mark("wave3", ts);
|
|
619
|
+
|
|
620
|
+
// ---- 5. pack + status
|
|
621
|
+
ts = performance.now();
|
|
622
|
+
const cpt = cfg.pack.charsPerToken;
|
|
623
|
+
const totalChars = Math.floor(budgetTokens * cpt);
|
|
624
|
+
const headerReserve = 420 + subQuestions.reduce((s, q) => s + q.length + 8, 0);
|
|
625
|
+
const footerReserve = Math.min(1800, Math.floor(totalChars * 0.12));
|
|
626
|
+
const body = packBody(evidence, {
|
|
627
|
+
subQuestions,
|
|
628
|
+
T2: thr.T2,
|
|
629
|
+
bodyChars: Math.max(0, totalChars - headerReserve - footerReserve),
|
|
630
|
+
cfg,
|
|
631
|
+
downweightChangelogs: !questionMentionsHistory(ctx.question + " " + subQuestions.join(" ")),
|
|
632
|
+
downweightTests: !questionMentionsTests(ctx.question),
|
|
633
|
+
});
|
|
634
|
+
const includedFiles = new Set(body.included.map((p) => p.path));
|
|
635
|
+
const windowedFiles = new Set(evidence.map((e) => e.path));
|
|
636
|
+
const otherFiles = ranked
|
|
637
|
+
.filter((i) => !windowedFiles.has(rec.candidates[i]!.path))
|
|
638
|
+
.map((i) => ({ path: rec.candidates[i]!.path, score: fileScores.get(fileIds[i]!) ?? 0 }))
|
|
639
|
+
.filter((x) => x.score > 0);
|
|
640
|
+
const leadIdByName = new Map(leadCands.map((l, i) => [l.name, `l${String(i).padStart(3, "0")}`]));
|
|
641
|
+
const notFollowed = leadCands
|
|
642
|
+
.filter((l) => !chosenNames.has(l.name))
|
|
643
|
+
.map((l) => ({
|
|
644
|
+
name: l.name,
|
|
645
|
+
score: leadScores.get(leadIdByName.get(l.name)!) ?? 0,
|
|
646
|
+
seenAt: `${l.seenAt.path}:${l.seenAt.line}`,
|
|
647
|
+
}))
|
|
648
|
+
.sort((a, b) => b.score - a.score);
|
|
649
|
+
const zero = kws.filter((k) => k.df === 0 && k.pathDf === 0);
|
|
650
|
+
const fragmentOnly = kws.filter((k) => k.fragments.length > 0 && (k.df > 0 || k.pathDf > 0));
|
|
651
|
+
const zeroHitKeywords = [
|
|
652
|
+
...zero.map((k) => ({ raw: k.raw, fragments: [] as string[] })),
|
|
653
|
+
...fragmentOnly.map((k) => ({ raw: k.raw, fragments: k.fragments })),
|
|
654
|
+
];
|
|
655
|
+
const widenedNote = prefixNote(rec);
|
|
656
|
+
const footer = renderFooter({
|
|
657
|
+
excluded: body.excluded,
|
|
658
|
+
otherFiles,
|
|
659
|
+
leadsNotFollowed: notFollowed,
|
|
660
|
+
zeroHitKeywords,
|
|
661
|
+
...(widenedNote ? { widenedNote } : {}),
|
|
662
|
+
cfg,
|
|
663
|
+
maxChars: footerReserve,
|
|
664
|
+
});
|
|
665
|
+
mark("pack", ts);
|
|
666
|
+
|
|
667
|
+
// A failure of the status request alone (after the pack) keeps the Jev pack with status unknown.
|
|
668
|
+
ts = performance.now();
|
|
669
|
+
let status: CodeSearchStatus = { label: "unknown", overall: null, subs: [] };
|
|
670
|
+
let statusCheckError: JevUnavailableError | JevRequestError | undefined;
|
|
671
|
+
let statusEvidenceChars = 0;
|
|
672
|
+
if (cfg.status.enabled && body.included.length) {
|
|
673
|
+
const ev = statusEvidence(body.included, cfg.status.maxEvidenceChars);
|
|
674
|
+
statusEvidenceChars = ev.length;
|
|
675
|
+
try {
|
|
676
|
+
status = statusLabel(await judge.status(ev, ctx), cfg);
|
|
677
|
+
} catch (error) {
|
|
678
|
+
if (
|
|
679
|
+
signal.aborted ||
|
|
680
|
+
!(error instanceof JevUnavailableError || error instanceof JevRequestError)
|
|
681
|
+
)
|
|
682
|
+
throw error;
|
|
683
|
+
statusCheckError = error;
|
|
684
|
+
status = { label: "unknown", overall: null, subs: [], error: error.message.slice(0, 200) };
|
|
685
|
+
}
|
|
686
|
+
} else if (!body.included.length) {
|
|
687
|
+
status = { label: "insufficient", overall: null, subs: [] };
|
|
688
|
+
}
|
|
689
|
+
mark("status", ts);
|
|
690
|
+
emit("pack", {
|
|
691
|
+
T2: thr.T2,
|
|
692
|
+
budgetTokens,
|
|
693
|
+
totalChars,
|
|
694
|
+
priority: body.priority,
|
|
695
|
+
included: body.included.map((p) => ({
|
|
696
|
+
id: p.id,
|
|
697
|
+
path: p.path,
|
|
698
|
+
start: p.start,
|
|
699
|
+
end: p.end,
|
|
700
|
+
trimmed: p.trimmed,
|
|
701
|
+
rel: round(p.rel),
|
|
702
|
+
cov: p.cov.map(round),
|
|
703
|
+
chars: p.block.length,
|
|
704
|
+
})),
|
|
705
|
+
excluded: body.excluded.map((e) => ({
|
|
706
|
+
id: e.id,
|
|
707
|
+
path: e.path,
|
|
708
|
+
start: e.start,
|
|
709
|
+
end: e.end,
|
|
710
|
+
rel: round(e.rel),
|
|
711
|
+
})),
|
|
712
|
+
status,
|
|
713
|
+
statusEvidenceChars,
|
|
714
|
+
});
|
|
715
|
+
|
|
716
|
+
// ---- render
|
|
717
|
+
const jevTotals = Object.values(judge.stats()).reduce(
|
|
718
|
+
(a, s) => ({
|
|
719
|
+
requests: a.requests + s.requests,
|
|
720
|
+
inputTokens: a.inputTokens + s.inputTokens,
|
|
721
|
+
costUsd: a.costUsd + s.costUsd,
|
|
722
|
+
}),
|
|
723
|
+
{ requests: 0, inputTokens: 0, costUsd: 0 },
|
|
724
|
+
);
|
|
725
|
+
const wallMs = Math.round(performance.now() - t0);
|
|
726
|
+
const label = partialLabel(session);
|
|
727
|
+
// A number, not a verdict: "sufficient" read as permission to stop, but the
|
|
728
|
+
// check only sees the passages the search returned.
|
|
729
|
+
const statusText =
|
|
730
|
+
status.label === "unknown" || status.overall === null
|
|
731
|
+
? status.error
|
|
732
|
+
? "evidence rating unknown (check failed)"
|
|
733
|
+
: "evidence rating unknown (no check)"
|
|
734
|
+
: `evidence rating ${r2(status.overall)}` +
|
|
735
|
+
(status.subs.length
|
|
736
|
+
? ` (${status.subs.map((x, j) => `s${j + 1} ${r2(x)}`).join(", ")})`
|
|
737
|
+
: "");
|
|
738
|
+
const buildText = (packTok: number) => {
|
|
739
|
+
const head = [
|
|
740
|
+
`code_search${label ? " " + label : ""}: ${statusText} | ${body.included.length} passages from ${includedFiles.size} files, ~${fmtK(packTok)} tokens | ${(wallMs / 1000).toFixed(1)}s`,
|
|
741
|
+
];
|
|
742
|
+
if (subQuestions.length) head.push(subQuestions.map((s, j) => `s${j + 1}: ${s}`).join("\n"));
|
|
743
|
+
head.push(
|
|
744
|
+
body.included.length
|
|
745
|
+
? "Passages are verbatim with original line numbers (N| text), grouped by file, best first; rel = relevance, [sN] = covers sub-question N. The rating covers only these passages; it cannot see other entry points, defaults, flags or exceptions the search did not return."
|
|
746
|
+
: "No passage passed verification. Try other keywords (exact identifiers, config keys, error strings) or read the candidates below.",
|
|
747
|
+
);
|
|
748
|
+
return [head.join("\n"), body.body, footer].filter(Boolean).join("\n\n") + "\n";
|
|
749
|
+
};
|
|
750
|
+
let text = buildText(0);
|
|
751
|
+
text = buildText(estTokens(text.length, cpt));
|
|
752
|
+
const stats: CodeSearchStats = {
|
|
753
|
+
wallMs,
|
|
754
|
+
stageMs,
|
|
755
|
+
candidates: rec.candidates.length,
|
|
756
|
+
filesSelected: selected.length,
|
|
757
|
+
passagesVerified: evidence.length,
|
|
758
|
+
passagesIncluded: body.included.length,
|
|
759
|
+
packChars: text.length,
|
|
760
|
+
packTokensEst: estTokens(text.length, cpt),
|
|
761
|
+
workspaceCalls: session.calls,
|
|
762
|
+
ripgrepTruncated: session.partial,
|
|
763
|
+
jev: { ...jevTotals, model: judge.model() },
|
|
764
|
+
};
|
|
765
|
+
emit("summary", { wallMs, stageMs, stats, status, jevByStage: judge.stats() });
|
|
766
|
+
const result: CodeSearchResult = { version: CODE_SEARCH_ENGINE_VERSION, text, status, stats };
|
|
767
|
+
if (statusCheckError) result.statusCheckError = statusCheckError;
|
|
768
|
+
return result;
|
|
769
|
+
}
|
|
770
|
+
|
|
771
|
+
function round(x: number): number {
|
|
772
|
+
return Number.isFinite(x) ? Math.round(x * 1000) / 1000 : x;
|
|
773
|
+
}
|