@opengeni/jev 0.1.0-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +190 -0
- package/README.md +140 -0
- package/dist/circuit-breaker.d.ts +58 -0
- package/dist/client.d.ts +173 -0
- package/dist/code-search/config.d.ts +142 -0
- package/dist/code-search/judge.d.ts +169 -0
- package/dist/code-search/leads.d.ts +96 -0
- package/dist/code-search/pack.d.ts +106 -0
- package/dist/code-search/recall.d.ts +155 -0
- package/dist/code-search/search.d.ts +93 -0
- package/dist/code-search/session.d.ts +33 -0
- package/dist/code-search/text.d.ts +35 -0
- package/dist/code-search/tool.d.ts +34 -0
- package/dist/code-search/windows.d.ts +85 -0
- package/dist/code-search/workspace.d.ts +51 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +3110 -0
- package/dist/index.js.map +1 -0
- package/package.json +39 -0
- package/src/circuit-breaker.ts +135 -0
- package/src/client.ts +577 -0
- package/src/code-search/config.ts +282 -0
- package/src/code-search/judge.ts +413 -0
- package/src/code-search/leads.ts +442 -0
- package/src/code-search/pack.ts +354 -0
- package/src/code-search/recall.ts +648 -0
- package/src/code-search/search.ts +773 -0
- package/src/code-search/session.ts +89 -0
- package/src/code-search/text.ts +159 -0
- package/src/code-search/tool.ts +209 -0
- package/src/code-search/windows.ts +617 -0
- package/src/code-search/workspace.ts +55 -0
- package/src/index.ts +71 -0
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Every code_search threshold, budget and cap in one place. The defaults are scout-0.3.1's Jev
|
|
3
|
+
* configuration, tuned on the E1 DEV split (2026-09-24); change them only with a new evaluation.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
export interface CodeSearchConfig {
|
|
7
|
+
recall: {
|
|
8
|
+
/** Candidate files passed to wave 1. */
|
|
9
|
+
maxCandidates: number;
|
|
10
|
+
/** Matching lines kept per file per keyword. */
|
|
11
|
+
maxMatchesPerFile: number;
|
|
12
|
+
/** rg --max-filesize (bytes); also the read cap for candidate files. */
|
|
13
|
+
maxFileBytes: number;
|
|
14
|
+
/** rg --max-columns: lines longer than this (minified / generated / data blobs) are ignored entirely. */
|
|
15
|
+
maxLineColumns: number;
|
|
16
|
+
/** Multiplier for test files unless the question is about tests. */
|
|
17
|
+
testWeight: number;
|
|
18
|
+
/** Path-name match bonus as a fraction of the keyword's IDF. */
|
|
19
|
+
pathBonus: number;
|
|
20
|
+
/** Extra weight per log(1 + matching lines) (tie breaker). */
|
|
21
|
+
hitCountWeight: number;
|
|
22
|
+
/** Single plain words up to this length are searched with a leading word boundary (`\bturn`, not `return`). */
|
|
23
|
+
shortWordMaxLen: number;
|
|
24
|
+
/** Compound keywords with zero hits retry their 2-part fragments (e.g. shouldCompactContext -> compactContext). */
|
|
25
|
+
fragmentFallback: boolean;
|
|
26
|
+
/** If paths restrict the search and fewer candidates than this are found, widen to the whole workspace. */
|
|
27
|
+
minCandidatesBeforeWiden: number;
|
|
28
|
+
/** Extra rg exclude globs (on top of the built-in list). */
|
|
29
|
+
extraExcludes: string[];
|
|
30
|
+
/** Also search hidden files/dirs (.github/workflows, .env.example, .changeset); .git is always excluded. */
|
|
31
|
+
searchHidden: boolean;
|
|
32
|
+
/** Multiplier for release notes (CHANGELOG*, .changeset/) unless the question is about history/releases. */
|
|
33
|
+
changelogWeight: number;
|
|
34
|
+
/** Time limit per ripgrep call; a timed-out search continues with partial output. */
|
|
35
|
+
ripgrepTimeoutMs: number;
|
|
36
|
+
};
|
|
37
|
+
wave1: {
|
|
38
|
+
filesPerRequest: number;
|
|
39
|
+
hitLinesPerFile: number;
|
|
40
|
+
hitLineChars: number;
|
|
41
|
+
/** Hard cap on files selected for windowing. */
|
|
42
|
+
maxFiles: number;
|
|
43
|
+
/** Always take at least this many files by judge score (guards against an over-strict T1). */
|
|
44
|
+
minFiles: number;
|
|
45
|
+
/** Always keep the top-N lexical files (fusion guard against judge misses). */
|
|
46
|
+
lexicalGuard: number;
|
|
47
|
+
};
|
|
48
|
+
wave2: {
|
|
49
|
+
windowsPerFile: number;
|
|
50
|
+
/** How many of a file's strongest hit lines seed windows (before merging/splitting). */
|
|
51
|
+
seedHitsPerFile: number;
|
|
52
|
+
/** Max lines to walk up from a hit looking for the enclosing declaration. */
|
|
53
|
+
maxUp: number;
|
|
54
|
+
/** Max lines below a hit kept in its window. */
|
|
55
|
+
maxDown: number;
|
|
56
|
+
/** Windows longer than this are split around hit clusters. */
|
|
57
|
+
maxWindowLines: number;
|
|
58
|
+
/** Fallback context when no enclosing declaration is found within maxUp. */
|
|
59
|
+
fallbackBefore: number;
|
|
60
|
+
fallbackAfter: number;
|
|
61
|
+
/** Merge windows whose gap is at most this many lines. */
|
|
62
|
+
mergeGap: number;
|
|
63
|
+
/** Windows shorter than this get surrounding context. */
|
|
64
|
+
minWindowLines: number;
|
|
65
|
+
passagesPerRequest: number;
|
|
66
|
+
/** Max passage text chars per Jev request. */
|
|
67
|
+
maxRequestChars: number;
|
|
68
|
+
/** Max passages verified in wave 2. */
|
|
69
|
+
maxPassages: number;
|
|
70
|
+
/** Code lines longer than this are cut with an explicit marker. */
|
|
71
|
+
maxLineChars: number;
|
|
72
|
+
/** Same for prose files (md/mdx/txt/rst), which often have single-line paragraphs of 1-6k chars. */
|
|
73
|
+
maxProseLineChars: number;
|
|
74
|
+
/** Windows whose rendered text is longer are split into whole-line sub-windows around their hits. */
|
|
75
|
+
maxWindowChars: number;
|
|
76
|
+
/** Same for markdown (sections are long, paragraphs independent). */
|
|
77
|
+
maxProseWindowChars: number;
|
|
78
|
+
};
|
|
79
|
+
wave3: {
|
|
80
|
+
enabled: boolean;
|
|
81
|
+
maxLeadCandidates: number;
|
|
82
|
+
maxLeadsFollowed: number;
|
|
83
|
+
/** Leads are extracted from at most this many of the most relevant passages (all >= T2). */
|
|
84
|
+
seedPassages: number;
|
|
85
|
+
defsPerLead: number;
|
|
86
|
+
};
|
|
87
|
+
status: {
|
|
88
|
+
enabled: boolean;
|
|
89
|
+
/** Max evidence chars sent to the status check (best passages first); keeps the request under Jev's 32k-token limit. */
|
|
90
|
+
maxEvidenceChars: number;
|
|
91
|
+
/** sufficient: overall >= hi and every sub-question >= hi. */
|
|
92
|
+
hi: number;
|
|
93
|
+
/** partial: overall >= lo or any sub-question >= hi. */
|
|
94
|
+
lo: number;
|
|
95
|
+
};
|
|
96
|
+
pack: {
|
|
97
|
+
charsPerToken: number;
|
|
98
|
+
/** Include passages below T2 only if fewer than this many passed (top by score, above minRelevance). */
|
|
99
|
+
minPassages: number;
|
|
100
|
+
minRelevance: number;
|
|
101
|
+
/** Fill the remaining budget with passages below T2 (ranked). */
|
|
102
|
+
fillBudget: boolean;
|
|
103
|
+
moreCandidates: number;
|
|
104
|
+
leadsNotFollowed: number;
|
|
105
|
+
/** A trimmed passage must keep at least this many lines. */
|
|
106
|
+
minTrimLines: number;
|
|
107
|
+
/** File-diversity penalty: a second passage of one file must beat the first passage of another by this margin. */
|
|
108
|
+
filePenalty: number;
|
|
109
|
+
/** Passages whose rendered block is longer are trimmed around their hits to this many chars (0 = off). */
|
|
110
|
+
maxPassageChars: number;
|
|
111
|
+
/** Multiplier on the rel used for pack ORDERING of release-note passages; 1 = off. */
|
|
112
|
+
changelogPrior: number;
|
|
113
|
+
/** Same for prose docs (md/mdx/txt/rst, not release notes); 1 = off. */
|
|
114
|
+
docPrior: number;
|
|
115
|
+
/** Same for test files unless the question is about tests; 1 = off. */
|
|
116
|
+
testPrior: number;
|
|
117
|
+
/** Pack ordering score = (rel + lexWeight x lexical passage score) x prior. */
|
|
118
|
+
lexWeight: number;
|
|
119
|
+
};
|
|
120
|
+
jev: {
|
|
121
|
+
/** Inline the question text in every Jev question when it is at most this long; else reference `question`. */
|
|
122
|
+
inlineQuestionMaxChars: number;
|
|
123
|
+
/** Wave 1: put the true/false rubric on every file question instead of once in the state. */
|
|
124
|
+
fileCriteriaPerQuestion: boolean;
|
|
125
|
+
/** Keep-alive connections opened while recall runs (0 = off). */
|
|
126
|
+
warmConnections: number;
|
|
127
|
+
};
|
|
128
|
+
thresholds: {
|
|
129
|
+
/** Wave 1: file triage floor (p >= T1 selects, up to maxFiles). */
|
|
130
|
+
T1: number;
|
|
131
|
+
/** Wave 2: passage relevance floor for inclusion in the pack, lead seeding and sub-question coverage. */
|
|
132
|
+
T2: number;
|
|
133
|
+
/** Wave 3: lead floor (p >= T3 is followed, up to maxLeadsFollowed). */
|
|
134
|
+
T3: number;
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export type CodeSearchConfigOverride = {
|
|
139
|
+
[K in keyof CodeSearchConfig]?: Partial<CodeSearchConfig[K]>;
|
|
140
|
+
};
|
|
141
|
+
|
|
142
|
+
export const DEFAULT_CODE_SEARCH_CONFIG: Readonly<CodeSearchConfig> = deepFreeze({
|
|
143
|
+
recall: {
|
|
144
|
+
maxCandidates: 240,
|
|
145
|
+
maxMatchesPerFile: 25,
|
|
146
|
+
maxFileBytes: 4_000_000,
|
|
147
|
+
maxLineColumns: 8000,
|
|
148
|
+
testWeight: 0.6,
|
|
149
|
+
pathBonus: 0.5,
|
|
150
|
+
hitCountWeight: 0.1,
|
|
151
|
+
shortWordMaxLen: 5,
|
|
152
|
+
fragmentFallback: true,
|
|
153
|
+
minCandidatesBeforeWiden: 5,
|
|
154
|
+
extraExcludes: [],
|
|
155
|
+
searchHidden: true,
|
|
156
|
+
changelogWeight: 0.6,
|
|
157
|
+
ripgrepTimeoutMs: 30_000,
|
|
158
|
+
},
|
|
159
|
+
wave1: {
|
|
160
|
+
filesPerRequest: 60,
|
|
161
|
+
hitLinesPerFile: 3,
|
|
162
|
+
hitLineChars: 160,
|
|
163
|
+
maxFiles: 16,
|
|
164
|
+
minFiles: 8,
|
|
165
|
+
lexicalGuard: 5,
|
|
166
|
+
},
|
|
167
|
+
wave2: {
|
|
168
|
+
windowsPerFile: 5,
|
|
169
|
+
seedHitsPerFile: 24,
|
|
170
|
+
maxUp: 40,
|
|
171
|
+
maxDown: 120,
|
|
172
|
+
maxWindowLines: 150,
|
|
173
|
+
fallbackBefore: 10,
|
|
174
|
+
fallbackAfter: 30,
|
|
175
|
+
mergeGap: 2,
|
|
176
|
+
minWindowLines: 12,
|
|
177
|
+
passagesPerRequest: 4,
|
|
178
|
+
maxRequestChars: 24_000,
|
|
179
|
+
maxPassages: 80,
|
|
180
|
+
maxLineChars: 400,
|
|
181
|
+
maxProseLineChars: 1600,
|
|
182
|
+
maxWindowChars: 8000,
|
|
183
|
+
maxProseWindowChars: 4000,
|
|
184
|
+
},
|
|
185
|
+
wave3: {
|
|
186
|
+
enabled: true,
|
|
187
|
+
maxLeadCandidates: 60,
|
|
188
|
+
maxLeadsFollowed: 6,
|
|
189
|
+
seedPassages: 12,
|
|
190
|
+
defsPerLead: 1,
|
|
191
|
+
},
|
|
192
|
+
status: { enabled: true, maxEvidenceChars: 60_000, hi: 0.7, lo: 0.4 },
|
|
193
|
+
pack: {
|
|
194
|
+
charsPerToken: 3.2,
|
|
195
|
+
minPassages: 4,
|
|
196
|
+
minRelevance: 0.1,
|
|
197
|
+
fillBudget: false,
|
|
198
|
+
moreCandidates: 10,
|
|
199
|
+
leadsNotFollowed: 8,
|
|
200
|
+
minTrimLines: 15,
|
|
201
|
+
filePenalty: 0.1,
|
|
202
|
+
maxPassageChars: 3200,
|
|
203
|
+
changelogPrior: 0.5,
|
|
204
|
+
docPrior: 0.85,
|
|
205
|
+
testPrior: 1,
|
|
206
|
+
lexWeight: 0,
|
|
207
|
+
},
|
|
208
|
+
jev: {
|
|
209
|
+
inlineQuestionMaxChars: 600,
|
|
210
|
+
fileCriteriaPerQuestion: false,
|
|
211
|
+
warmConnections: 12,
|
|
212
|
+
},
|
|
213
|
+
// calibrated on the DEV split from the observed score distributions (relevant files mostly 0.6-0.9;
|
|
214
|
+
// minFiles and the lexical guard keep recall when few files pass)
|
|
215
|
+
thresholds: { T1: 0.6, T2: 0.5, T3: 0.5 },
|
|
216
|
+
});
|
|
217
|
+
|
|
218
|
+
/** Defaults with a partial override merged on top (unknown keys and wrong types are rejected), validated. */
|
|
219
|
+
export function codeSearchConfig(override: CodeSearchConfigOverride = {}): CodeSearchConfig {
|
|
220
|
+
const base = structuredClone(DEFAULT_CODE_SEARCH_CONFIG) as CodeSearchConfig;
|
|
221
|
+
const out = base as unknown as Record<string, Record<string, unknown>>;
|
|
222
|
+
for (const [section, values] of Object.entries(override as Record<string, unknown>)) {
|
|
223
|
+
const target = out[section];
|
|
224
|
+
if (!target) throw new Error(`unknown code_search config section: ${section}`);
|
|
225
|
+
if (!isPlainObject(values)) throw new Error(`code_search config ${section}: expected object`);
|
|
226
|
+
for (const [key, value] of Object.entries(values)) {
|
|
227
|
+
if (!(key in target)) throw new Error(`unknown code_search config key: ${section}.${key}`);
|
|
228
|
+
const current = target[key];
|
|
229
|
+
if (Array.isArray(current) ? !Array.isArray(value) : typeof current !== typeof value) {
|
|
230
|
+
throw new Error(
|
|
231
|
+
`code_search config ${section}.${key}: expected ${Array.isArray(current) ? "array" : typeof current}`,
|
|
232
|
+
);
|
|
233
|
+
}
|
|
234
|
+
target[key] = value;
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
validateCodeSearchConfig(base);
|
|
238
|
+
return base;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export function validateCodeSearchConfig(c: CodeSearchConfig): void {
|
|
242
|
+
const probs: string[] = [];
|
|
243
|
+
for (const [k, v] of Object.entries(c.thresholds))
|
|
244
|
+
if (!(v >= 0 && v <= 1)) probs.push(`thresholds.${k} must be in [0,1]`);
|
|
245
|
+
if (c.wave1.maxFiles < 1) probs.push("wave1.maxFiles must be >= 1");
|
|
246
|
+
if (c.wave1.minFiles > c.wave1.maxFiles) probs.push("wave1.minFiles must be <= wave1.maxFiles");
|
|
247
|
+
if (c.wave1.filesPerRequest < 1 || c.wave1.filesPerRequest > 250)
|
|
248
|
+
probs.push("wave1.filesPerRequest must be 1..250");
|
|
249
|
+
if (c.wave2.passagesPerRequest < 1) probs.push("wave2.passagesPerRequest must be >= 1");
|
|
250
|
+
if (c.wave2.maxWindowLines < 10) probs.push("wave2.maxWindowLines must be >= 10");
|
|
251
|
+
if (c.wave2.maxWindowChars < 4 * c.wave2.maxLineChars)
|
|
252
|
+
probs.push("wave2.maxWindowChars must be >= 4 x wave2.maxLineChars");
|
|
253
|
+
if (c.wave2.maxProseWindowChars < 2 * c.wave2.maxProseLineChars + 200) {
|
|
254
|
+
probs.push("wave2.maxProseWindowChars must be >= 2 x wave2.maxProseLineChars + 200");
|
|
255
|
+
}
|
|
256
|
+
if (c.wave3.maxLeadCandidates > 250) probs.push("wave3.maxLeadCandidates must be <= 250");
|
|
257
|
+
if (!(c.status.lo <= c.status.hi)) probs.push("status.lo must be <= status.hi");
|
|
258
|
+
if (c.pack.charsPerToken <= 0) probs.push("pack.charsPerToken must be > 0");
|
|
259
|
+
if (c.pack.filePenalty < 0) probs.push("pack.filePenalty must be >= 0");
|
|
260
|
+
if (c.pack.maxPassageChars !== 0 && c.pack.maxPassageChars < 600)
|
|
261
|
+
probs.push("pack.maxPassageChars must be 0 or >= 600");
|
|
262
|
+
for (const k of ["changelogPrior", "docPrior", "testPrior"] as const) {
|
|
263
|
+
if (!(c.pack[k] > 0 && c.pack[k] <= 1)) probs.push(`pack.${k} must be in (0,1]`);
|
|
264
|
+
}
|
|
265
|
+
if (!(c.pack.lexWeight >= 0)) probs.push("pack.lexWeight must be >= 0");
|
|
266
|
+
if (!(c.recall.changelogWeight > 0 && c.recall.changelogWeight <= 1))
|
|
267
|
+
probs.push("recall.changelogWeight must be in (0,1]");
|
|
268
|
+
if (!(c.recall.ripgrepTimeoutMs > 0)) probs.push("recall.ripgrepTimeoutMs must be > 0");
|
|
269
|
+
if (probs.length) throw new Error(`invalid code_search config: ${probs.join("; ")}`);
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
function isPlainObject(x: unknown): x is Record<string, unknown> {
|
|
273
|
+
return typeof x === "object" && x !== null && !Array.isArray(x);
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
function deepFreeze<T>(value: T): T {
|
|
277
|
+
if (typeof value === "object" && value !== null) {
|
|
278
|
+
for (const v of Object.values(value)) deepFreeze(v);
|
|
279
|
+
Object.freeze(value);
|
|
280
|
+
}
|
|
281
|
+
return value;
|
|
282
|
+
}
|
|
@@ -0,0 +1,413 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* judge.ts - the only place where scores come from: Jev answers one probability per judged item.
|
|
3
|
+
*
|
|
4
|
+
* All Jev wording lives in PROMPTS below (one place to review). Design rules followed:
|
|
5
|
+
* - one state + many narrow Nouls per request (fan-out); one Noul per candidate (multi-select)
|
|
6
|
+
* - criteria text is explicit (in the state for 60-file triage batches, per question for passages)
|
|
7
|
+
* - the literal question text is inlined in every question (no `collections[i]`-style indirection)
|
|
8
|
+
* - state holds only the candidates being judged (filtered by code first)
|
|
9
|
+
* - no counting, math or dates asked of Jev; code combines answers
|
|
10
|
+
* A Jev failure is not replaced by lexical scores: it propagates and the whole search fails.
|
|
11
|
+
*/
|
|
12
|
+
import { noul, type JevAnswer, type JevClient, type JevNoulQuestion } from "../client";
|
|
13
|
+
import type { CodeSearchConfig } from "./config";
|
|
14
|
+
|
|
15
|
+
export interface JudgeContext {
|
|
16
|
+
question: string;
|
|
17
|
+
subQuestions: string[];
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export interface FileItem {
|
|
21
|
+
id: string;
|
|
22
|
+
path: string;
|
|
23
|
+
/** path + best hit lines, as shown to the judge */
|
|
24
|
+
descriptor: string;
|
|
25
|
+
/** Normalized lexical score; used only if Jev returns no usable number for this item. */
|
|
26
|
+
lex: number;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export interface PassageItem {
|
|
30
|
+
id: string;
|
|
31
|
+
path: string;
|
|
32
|
+
start: number;
|
|
33
|
+
end: number;
|
|
34
|
+
/** rendered `N| text` */
|
|
35
|
+
text: string;
|
|
36
|
+
label?: string | undefined;
|
|
37
|
+
lex: number;
|
|
38
|
+
lexCov: number[];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface LeadItem {
|
|
42
|
+
id: string;
|
|
43
|
+
name: string;
|
|
44
|
+
seenAt: string;
|
|
45
|
+
context: string;
|
|
46
|
+
lex: number;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface PassageScore {
|
|
50
|
+
rel: number;
|
|
51
|
+
cov: number[];
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface StatusScore {
|
|
55
|
+
overall: number;
|
|
56
|
+
subs: number[];
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export interface StageStats {
|
|
60
|
+
requests: number;
|
|
61
|
+
inputTokens: number;
|
|
62
|
+
costUsd: number;
|
|
63
|
+
ms: number;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export type JudgeEvent = (stage: string, data: Record<string, unknown>) => void;
|
|
67
|
+
|
|
68
|
+
// ---------------------------------------------------------------------------
|
|
69
|
+
// Prompts (all Jev wording; identical to scout-0.3.1)
|
|
70
|
+
// ---------------------------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
export const PROMPTS = {
|
|
73
|
+
fileTask:
|
|
74
|
+
"Code search triage for a software repository. A developer asked `question`. `files` lists candidate files found by keyword search; each entry is the file path followed by up to 3 matching lines as `line: text`. Judge each file independently by its path and matching lines.",
|
|
75
|
+
fileCriteria: {
|
|
76
|
+
yes: "Reading this file would likely help answer the question: it probably implements, decides, computes, configures, enforces or documents the asked behavior. A doc or design note that explains the behavior counts.",
|
|
77
|
+
no: "The file probably only mentions the same words, imports or passes values through, belongs to an unrelated feature that shares keywords, or is a test that is not about the asked behavior.",
|
|
78
|
+
},
|
|
79
|
+
fileQuestion: (id: string, path: string, q: string) =>
|
|
80
|
+
`Would reading file \`files.${id}\` (${path}) help answer the question${q}? Apply \`criteria\`.`,
|
|
81
|
+
fileQuestionPerQ: (id: string, path: string, q: string) =>
|
|
82
|
+
`Would reading file \`files.${id}\` (${path}) help answer the question${q}?`,
|
|
83
|
+
|
|
84
|
+
passageTask:
|
|
85
|
+
"Evidence check for a question about a software repository. Each entry in `passages` is a verbatim excerpt of one repository file with its original line numbers (`N| text`). `in` names the enclosing declaration when the excerpt starts inside one. Judge each passage on its own content only.",
|
|
86
|
+
passageQuestion: (id: string, path: string, a: number, b: number, q: string) =>
|
|
87
|
+
`Does \`passages.${id}\` (${path} lines ${a}-${b}) contain code or text needed to answer the question${q}?`,
|
|
88
|
+
passageCriteria: {
|
|
89
|
+
true: "The passage implements, decides, computes, configures, enforces or documents part of the asked behavior. A helper that performs one asked step counts, and so does a doc passage that states the behavior.",
|
|
90
|
+
false:
|
|
91
|
+
"The passage only mentions, imports, calls or passes through names related to the question, is a test or type declaration that does not show the behavior, or is unrelated code that shares keywords.",
|
|
92
|
+
},
|
|
93
|
+
coverageQuestion: (id: string, path: string, a: number, b: number, sub: string) =>
|
|
94
|
+
`Does \`passages.${id}\` (${path} lines ${a}-${b}) show the answer to this part of the question: "${sub}"?`,
|
|
95
|
+
coverageCriteria: {
|
|
96
|
+
true: "The passage itself states or implements the answer to this part.",
|
|
97
|
+
false:
|
|
98
|
+
"The passage does not address this part, or only mentions it without showing the answer.",
|
|
99
|
+
},
|
|
100
|
+
|
|
101
|
+
leadTask:
|
|
102
|
+
"Follow-up selection for a question about a software repository. Evidence passages already found reference the identifiers in `leads`, whose definitions have not been read yet. Each lead shows the identifier, where it was seen, and the line it appears on.",
|
|
103
|
+
leadCriteria: {
|
|
104
|
+
yes: "The definition likely computes, decides, configures or enforces part of the asked behavior, or holds a constant, setting or rule that the answer depends on.",
|
|
105
|
+
no: "It is a generic helper (logging, formatting, errors, type plumbing, database or HTTP plumbing) or it is unrelated to the question.",
|
|
106
|
+
},
|
|
107
|
+
leadQuestion: (id: string, name: string, q: string) =>
|
|
108
|
+
`Would reading where \`${name}\` (\`leads.${id}\`) is defined help answer the question${q}? Apply \`criteria\`.`,
|
|
109
|
+
|
|
110
|
+
statusTask:
|
|
111
|
+
"Sufficiency check. `evidence` is a set of verbatim excerpts of repository files (with original line numbers) collected to answer `question`.",
|
|
112
|
+
statusQuestion: (q: string) =>
|
|
113
|
+
`Do the \`evidence\` passages show the answer to the question${q}?`,
|
|
114
|
+
statusSubQuestion: (sub: string) =>
|
|
115
|
+
`Do the \`evidence\` passages show the answer to this part of the question: "${sub}"?`,
|
|
116
|
+
statusCriteria: {
|
|
117
|
+
true: "Together the passages state or implement the answer concretely enough to cite file and line.",
|
|
118
|
+
false:
|
|
119
|
+
"Some part of the answer is missing: the passages only mention the topic, or the deciding code is elsewhere.",
|
|
120
|
+
},
|
|
121
|
+
};
|
|
122
|
+
|
|
123
|
+
function inlineQ(question: string, maxChars: number): string {
|
|
124
|
+
return question.length <= maxChars ? `: "${question}"` : " in `question`";
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function withSubs(ctx: JudgeContext): Record<string, unknown> {
|
|
128
|
+
return ctx.subQuestions.length
|
|
129
|
+
? { sub_questions: Object.fromEntries(ctx.subQuestions.map((s, i) => [`s${i + 1}`, s])) }
|
|
130
|
+
: {};
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// ---------------------------------------------------------------------------
|
|
134
|
+
// Request builders (exported for tests)
|
|
135
|
+
// ---------------------------------------------------------------------------
|
|
136
|
+
|
|
137
|
+
export function buildFileRequest(items: FileItem[], ctx: JudgeContext, cfg: CodeSearchConfig) {
|
|
138
|
+
const q = inlineQ(ctx.question, cfg.jev.inlineQuestionMaxChars);
|
|
139
|
+
// default: the rubric is stated once in the state and referenced (~65% fewer question tokens);
|
|
140
|
+
// fileCriteriaPerQuestion puts it on every Noul instead
|
|
141
|
+
const perQ = cfg.jev.fileCriteriaPerQuestion;
|
|
142
|
+
const state = {
|
|
143
|
+
task: PROMPTS.fileTask,
|
|
144
|
+
question: ctx.question,
|
|
145
|
+
...withSubs(ctx),
|
|
146
|
+
...(perQ ? {} : { criteria: PROMPTS.fileCriteria }),
|
|
147
|
+
files: Object.fromEntries(items.map((f) => [f.id, f.descriptor])),
|
|
148
|
+
};
|
|
149
|
+
const questions: Record<string, JevNoulQuestion> = Object.fromEntries(
|
|
150
|
+
items.map((f) => [
|
|
151
|
+
f.id,
|
|
152
|
+
perQ
|
|
153
|
+
? noul(PROMPTS.fileQuestionPerQ(f.id, f.path, q), {
|
|
154
|
+
true: PROMPTS.fileCriteria.yes,
|
|
155
|
+
false: PROMPTS.fileCriteria.no,
|
|
156
|
+
})
|
|
157
|
+
: noul(PROMPTS.fileQuestion(f.id, f.path, q)),
|
|
158
|
+
]),
|
|
159
|
+
);
|
|
160
|
+
return { state, questions };
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
export function buildPassageRequest(
|
|
164
|
+
items: PassageItem[],
|
|
165
|
+
ctx: JudgeContext,
|
|
166
|
+
cfg: CodeSearchConfig,
|
|
167
|
+
) {
|
|
168
|
+
const q = inlineQ(ctx.question, cfg.jev.inlineQuestionMaxChars);
|
|
169
|
+
const state = {
|
|
170
|
+
task: PROMPTS.passageTask,
|
|
171
|
+
question: ctx.question,
|
|
172
|
+
...withSubs(ctx),
|
|
173
|
+
passages: Object.fromEntries(
|
|
174
|
+
items.map((p) => [
|
|
175
|
+
p.id,
|
|
176
|
+
{
|
|
177
|
+
file: p.path,
|
|
178
|
+
lines: `${p.start}-${p.end}`,
|
|
179
|
+
...(p.label ? { in: p.label } : {}),
|
|
180
|
+
text: p.text,
|
|
181
|
+
},
|
|
182
|
+
]),
|
|
183
|
+
),
|
|
184
|
+
};
|
|
185
|
+
const questions: Record<string, JevNoulQuestion> = {};
|
|
186
|
+
for (const p of items) {
|
|
187
|
+
questions[`rel::${p.id}`] = noul(
|
|
188
|
+
PROMPTS.passageQuestion(p.id, p.path, p.start, p.end, q),
|
|
189
|
+
PROMPTS.passageCriteria,
|
|
190
|
+
);
|
|
191
|
+
ctx.subQuestions.forEach((sub, j) => {
|
|
192
|
+
questions[`cov::${p.id}::${j}`] = noul(
|
|
193
|
+
PROMPTS.coverageQuestion(p.id, p.path, p.start, p.end, sub),
|
|
194
|
+
PROMPTS.coverageCriteria,
|
|
195
|
+
);
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
return { state, questions };
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
export function buildLeadRequest(items: LeadItem[], ctx: JudgeContext, cfg: CodeSearchConfig) {
|
|
202
|
+
const q = inlineQ(ctx.question, cfg.jev.inlineQuestionMaxChars);
|
|
203
|
+
const state = {
|
|
204
|
+
task: PROMPTS.leadTask,
|
|
205
|
+
question: ctx.question,
|
|
206
|
+
...withSubs(ctx),
|
|
207
|
+
criteria: PROMPTS.leadCriteria,
|
|
208
|
+
leads: Object.fromEntries(
|
|
209
|
+
items.map((l) => [l.id, `${l.name} (seen at ${l.seenAt}) ${l.context}`]),
|
|
210
|
+
),
|
|
211
|
+
};
|
|
212
|
+
const questions: Record<string, JevNoulQuestion> = Object.fromEntries(
|
|
213
|
+
items.map((l) => [l.id, noul(PROMPTS.leadQuestion(l.id, l.name, q))]),
|
|
214
|
+
);
|
|
215
|
+
return { state, questions };
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
export function buildStatusRequest(evidence: string, ctx: JudgeContext, cfg: CodeSearchConfig) {
|
|
219
|
+
const q = inlineQ(ctx.question, cfg.jev.inlineQuestionMaxChars);
|
|
220
|
+
const state = { task: PROMPTS.statusTask, question: ctx.question, ...withSubs(ctx), evidence };
|
|
221
|
+
const questions: Record<string, JevNoulQuestion> = {
|
|
222
|
+
overall: noul(PROMPTS.statusQuestion(q), PROMPTS.statusCriteria),
|
|
223
|
+
};
|
|
224
|
+
ctx.subQuestions.forEach((sub, j) => {
|
|
225
|
+
questions[`sub::${j}`] = noul(PROMPTS.statusSubQuestion(sub), PROMPTS.statusCriteria);
|
|
226
|
+
});
|
|
227
|
+
return { state, questions };
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** Group passages into requests: consecutive (same-file first) up to perRequest items and maxChars of text. */
|
|
231
|
+
export function chunkPassages(
|
|
232
|
+
items: PassageItem[],
|
|
233
|
+
perRequest: number,
|
|
234
|
+
maxChars: number,
|
|
235
|
+
): PassageItem[][] {
|
|
236
|
+
const out: PassageItem[][] = [];
|
|
237
|
+
let cur: PassageItem[] = [];
|
|
238
|
+
let chars = 0;
|
|
239
|
+
for (const p of items) {
|
|
240
|
+
const c = p.text.length;
|
|
241
|
+
if (cur.length && (cur.length >= perRequest || chars + c > maxChars)) {
|
|
242
|
+
out.push(cur);
|
|
243
|
+
cur = [];
|
|
244
|
+
chars = 0;
|
|
245
|
+
}
|
|
246
|
+
cur.push(p);
|
|
247
|
+
chars += c;
|
|
248
|
+
}
|
|
249
|
+
if (cur.length) out.push(cur);
|
|
250
|
+
return out;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/** Split into the fewest groups of at most n items, as equal in size as possible (69 -> 35 + 34, not 60 + 9). */
|
|
254
|
+
export function chunkEven<T>(xs: T[], n: number): T[][] {
|
|
255
|
+
if (!xs.length) return [];
|
|
256
|
+
const groups = Math.ceil(xs.length / n);
|
|
257
|
+
const size = Math.ceil(xs.length / groups);
|
|
258
|
+
return chunk(xs, size);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
function chunk<T>(xs: T[], n: number): T[][] {
|
|
262
|
+
const out: T[][] = [];
|
|
263
|
+
for (let i = 0; i < xs.length; i += n) out.push(xs.slice(i, i + n));
|
|
264
|
+
return out;
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
const noulOf = (a: JevAnswer | undefined): number =>
|
|
268
|
+
a && a.type === "noul" && Number.isFinite(a.probability) ? a.probability : Number.NaN;
|
|
269
|
+
|
|
270
|
+
function orLex(p: number, lex: number): number {
|
|
271
|
+
return Number.isFinite(p) ? p : lex;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// ---------------------------------------------------------------------------
|
|
275
|
+
// Jev judge
|
|
276
|
+
// ---------------------------------------------------------------------------
|
|
277
|
+
|
|
278
|
+
export class JevJudge {
|
|
279
|
+
private readonly stageStats: Record<string, StageStats> = {};
|
|
280
|
+
private jevModel: string | null = null;
|
|
281
|
+
|
|
282
|
+
constructor(
|
|
283
|
+
private readonly o: {
|
|
284
|
+
client: JevClient;
|
|
285
|
+
config: CodeSearchConfig;
|
|
286
|
+
signal: AbortSignal;
|
|
287
|
+
onEvent?: JudgeEvent | undefined;
|
|
288
|
+
},
|
|
289
|
+
) {}
|
|
290
|
+
|
|
291
|
+
stats(): Record<string, StageStats> {
|
|
292
|
+
return this.stageStats;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
model(): string | null {
|
|
296
|
+
return this.jevModel;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
private stat(stage: string): StageStats {
|
|
300
|
+
return (this.stageStats[stage] ??= { requests: 0, inputTokens: 0, costUsd: 0, ms: 0 });
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/** One logical request (the client may split its questions into several HTTP requests). */
|
|
304
|
+
private async ask(
|
|
305
|
+
stage: string,
|
|
306
|
+
state: unknown,
|
|
307
|
+
questions: Record<string, JevNoulQuestion>,
|
|
308
|
+
): Promise<Record<string, JevAnswer>> {
|
|
309
|
+
const t0 = performance.now();
|
|
310
|
+
const r = await this.o.client.ask(state, questions, {
|
|
311
|
+
tag: stage,
|
|
312
|
+
signal: this.o.signal,
|
|
313
|
+
});
|
|
314
|
+
const st = this.stat(stage);
|
|
315
|
+
st.requests += r.requests;
|
|
316
|
+
st.inputTokens += r.usage.inputTokens;
|
|
317
|
+
st.costUsd += r.costUsd;
|
|
318
|
+
this.jevModel = r.model || this.jevModel;
|
|
319
|
+
this.o.onEvent?.("jev", {
|
|
320
|
+
jevStage: stage,
|
|
321
|
+
ms: Math.round(performance.now() - t0),
|
|
322
|
+
requests: r.requests,
|
|
323
|
+
model: r.model,
|
|
324
|
+
inputTokens: r.usage.inputTokens,
|
|
325
|
+
costUsd: r.costUsd,
|
|
326
|
+
state,
|
|
327
|
+
questions,
|
|
328
|
+
answers: r.answers,
|
|
329
|
+
});
|
|
330
|
+
return r.answers;
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
/** Run a stage's batches in parallel; the first failure rejects the stage. */
|
|
334
|
+
private async stage<I, R>(
|
|
335
|
+
stage: string,
|
|
336
|
+
batches: I[][],
|
|
337
|
+
build: (b: I[]) => { state: unknown; questions: Record<string, JevNoulQuestion> },
|
|
338
|
+
read: (b: I[], answers: Record<string, JevAnswer>) => Array<[string, R]>,
|
|
339
|
+
): Promise<Map<string, R>> {
|
|
340
|
+
const t0 = performance.now();
|
|
341
|
+
try {
|
|
342
|
+
const results = await Promise.all(
|
|
343
|
+
batches.map(async (b) => {
|
|
344
|
+
const req = build(b);
|
|
345
|
+
return read(b, await this.ask(stage, req.state, req.questions));
|
|
346
|
+
}),
|
|
347
|
+
);
|
|
348
|
+
return new Map(results.flat());
|
|
349
|
+
} finally {
|
|
350
|
+
this.stat(stage).ms += Math.round(performance.now() - t0);
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
async scoreFiles(items: FileItem[], ctx: JudgeContext): Promise<Map<string, number>> {
|
|
355
|
+
if (!items.length) return new Map();
|
|
356
|
+
return this.stage(
|
|
357
|
+
"wave1",
|
|
358
|
+
chunkEven(items, this.o.config.wave1.filesPerRequest),
|
|
359
|
+
(b) => buildFileRequest(b, ctx, this.o.config),
|
|
360
|
+
(b, a) => b.map((f) => [f.id, orLex(noulOf(a[f.id]), f.lex)]),
|
|
361
|
+
);
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
async scorePassages(
|
|
365
|
+
items: PassageItem[],
|
|
366
|
+
ctx: JudgeContext,
|
|
367
|
+
stage = "wave2",
|
|
368
|
+
): Promise<Map<string, PassageScore>> {
|
|
369
|
+
if (!items.length) return new Map();
|
|
370
|
+
const w = this.o.config.wave2;
|
|
371
|
+
return this.stage(
|
|
372
|
+
stage,
|
|
373
|
+
chunkPassages(items, w.passagesPerRequest, w.maxRequestChars),
|
|
374
|
+
(b) => buildPassageRequest(b, ctx, this.o.config),
|
|
375
|
+
(b, a) =>
|
|
376
|
+
b.map((p) => [
|
|
377
|
+
p.id,
|
|
378
|
+
{
|
|
379
|
+
rel: orLex(noulOf(a[`rel::${p.id}`]), p.lex),
|
|
380
|
+
cov: ctx.subQuestions.map((_, j) =>
|
|
381
|
+
orLex(noulOf(a[`cov::${p.id}::${j}`]), p.lexCov[j] ?? 0),
|
|
382
|
+
),
|
|
383
|
+
},
|
|
384
|
+
]),
|
|
385
|
+
);
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
async scoreLeads(items: LeadItem[], ctx: JudgeContext): Promise<Map<string, number>> {
|
|
389
|
+
if (!items.length) return new Map();
|
|
390
|
+
return this.stage(
|
|
391
|
+
"leads",
|
|
392
|
+
chunk(items, 250),
|
|
393
|
+
(b) => buildLeadRequest(b, ctx, this.o.config),
|
|
394
|
+
(b, a) => b.map((l) => [l.id, orLex(noulOf(a[l.id]), l.lex)]),
|
|
395
|
+
);
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/** null = nothing to check (empty evidence). */
|
|
399
|
+
async status(evidence: string, ctx: JudgeContext): Promise<StatusScore | null> {
|
|
400
|
+
if (!evidence.trim()) return null;
|
|
401
|
+
const req = buildStatusRequest(evidence, ctx, this.o.config);
|
|
402
|
+
const t0 = performance.now();
|
|
403
|
+
try {
|
|
404
|
+
const a = await this.ask("status", req.state, req.questions);
|
|
405
|
+
return {
|
|
406
|
+
overall: noulOf(a.overall),
|
|
407
|
+
subs: ctx.subQuestions.map((_, j) => noulOf(a[`sub::${j}`])),
|
|
408
|
+
};
|
|
409
|
+
} finally {
|
|
410
|
+
this.stat("status").ms += Math.round(performance.now() - t0);
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
}
|