@opengeni/jev 0.1.0-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +190 -0
- package/README.md +140 -0
- package/dist/circuit-breaker.d.ts +58 -0
- package/dist/client.d.ts +173 -0
- package/dist/code-search/config.d.ts +142 -0
- package/dist/code-search/judge.d.ts +169 -0
- package/dist/code-search/leads.d.ts +96 -0
- package/dist/code-search/pack.d.ts +106 -0
- package/dist/code-search/recall.d.ts +155 -0
- package/dist/code-search/search.d.ts +93 -0
- package/dist/code-search/session.d.ts +33 -0
- package/dist/code-search/text.d.ts +35 -0
- package/dist/code-search/tool.d.ts +34 -0
- package/dist/code-search/windows.d.ts +85 -0
- package/dist/code-search/workspace.d.ts +51 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +3110 -0
- package/dist/index.js.map +1 -0
- package/package.json +39 -0
- package/src/circuit-breaker.ts +135 -0
- package/src/client.ts +577 -0
- package/src/code-search/config.ts +282 -0
- package/src/code-search/judge.ts +413 -0
- package/src/code-search/leads.ts +442 -0
- package/src/code-search/pack.ts +354 -0
- package/src/code-search/recall.ts +648 -0
- package/src/code-search/search.ts +773 -0
- package/src/code-search/session.ts +89 -0
- package/src/code-search/text.ts +159 -0
- package/src/code-search/tool.ts +209 -0
- package/src/code-search/windows.ts +617 -0
- package/src/code-search/workspace.ts +55 -0
- package/src/index.ts +71 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Every code_search threshold, budget and cap in one place. The defaults are scout-0.3.1's Jev
|
|
3
|
+
* configuration, tuned on the E1 DEV split (2026-09-24); change them only with a new evaluation.
|
|
4
|
+
*/
|
|
5
|
+
export interface CodeSearchConfig {
|
|
6
|
+
recall: {
|
|
7
|
+
/** Candidate files passed to wave 1. */
|
|
8
|
+
maxCandidates: number;
|
|
9
|
+
/** Matching lines kept per file per keyword. */
|
|
10
|
+
maxMatchesPerFile: number;
|
|
11
|
+
/** rg --max-filesize (bytes); also the read cap for candidate files. */
|
|
12
|
+
maxFileBytes: number;
|
|
13
|
+
/** rg --max-columns: lines longer than this (minified / generated / data blobs) are ignored entirely. */
|
|
14
|
+
maxLineColumns: number;
|
|
15
|
+
/** Multiplier for test files unless the question is about tests. */
|
|
16
|
+
testWeight: number;
|
|
17
|
+
/** Path-name match bonus as a fraction of the keyword's IDF. */
|
|
18
|
+
pathBonus: number;
|
|
19
|
+
/** Extra weight per log(1 + matching lines) (tie breaker). */
|
|
20
|
+
hitCountWeight: number;
|
|
21
|
+
/** Single plain words up to this length are searched with a leading word boundary (`\bturn`, not `return`). */
|
|
22
|
+
shortWordMaxLen: number;
|
|
23
|
+
/** Compound keywords with zero hits retry their 2-part fragments (e.g. shouldCompactContext -> compactContext). */
|
|
24
|
+
fragmentFallback: boolean;
|
|
25
|
+
/** If paths restrict the search and fewer candidates than this are found, widen to the whole workspace. */
|
|
26
|
+
minCandidatesBeforeWiden: number;
|
|
27
|
+
/** Extra rg exclude globs (on top of the built-in list). */
|
|
28
|
+
extraExcludes: string[];
|
|
29
|
+
/** Also search hidden files/dirs (.github/workflows, .env.example, .changeset); .git is always excluded. */
|
|
30
|
+
searchHidden: boolean;
|
|
31
|
+
/** Multiplier for release notes (CHANGELOG*, .changeset/) unless the question is about history/releases. */
|
|
32
|
+
changelogWeight: number;
|
|
33
|
+
/** Time limit per ripgrep call; a timed-out search continues with partial output. */
|
|
34
|
+
ripgrepTimeoutMs: number;
|
|
35
|
+
};
|
|
36
|
+
wave1: {
|
|
37
|
+
filesPerRequest: number;
|
|
38
|
+
hitLinesPerFile: number;
|
|
39
|
+
hitLineChars: number;
|
|
40
|
+
/** Hard cap on files selected for windowing. */
|
|
41
|
+
maxFiles: number;
|
|
42
|
+
/** Always take at least this many files by judge score (guards against an over-strict T1). */
|
|
43
|
+
minFiles: number;
|
|
44
|
+
/** Always keep the top-N lexical files (fusion guard against judge misses). */
|
|
45
|
+
lexicalGuard: number;
|
|
46
|
+
};
|
|
47
|
+
wave2: {
|
|
48
|
+
windowsPerFile: number;
|
|
49
|
+
/** How many of a file's strongest hit lines seed windows (before merging/splitting). */
|
|
50
|
+
seedHitsPerFile: number;
|
|
51
|
+
/** Max lines to walk up from a hit looking for the enclosing declaration. */
|
|
52
|
+
maxUp: number;
|
|
53
|
+
/** Max lines below a hit kept in its window. */
|
|
54
|
+
maxDown: number;
|
|
55
|
+
/** Windows longer than this are split around hit clusters. */
|
|
56
|
+
maxWindowLines: number;
|
|
57
|
+
/** Fallback context when no enclosing declaration is found within maxUp. */
|
|
58
|
+
fallbackBefore: number;
|
|
59
|
+
fallbackAfter: number;
|
|
60
|
+
/** Merge windows whose gap is at most this many lines. */
|
|
61
|
+
mergeGap: number;
|
|
62
|
+
/** Windows shorter than this get surrounding context. */
|
|
63
|
+
minWindowLines: number;
|
|
64
|
+
passagesPerRequest: number;
|
|
65
|
+
/** Max passage text chars per Jev request. */
|
|
66
|
+
maxRequestChars: number;
|
|
67
|
+
/** Max passages verified in wave 2. */
|
|
68
|
+
maxPassages: number;
|
|
69
|
+
/** Code lines longer than this are cut with an explicit marker. */
|
|
70
|
+
maxLineChars: number;
|
|
71
|
+
/** Same for prose files (md/mdx/txt/rst), which often have single-line paragraphs of 1-6k chars. */
|
|
72
|
+
maxProseLineChars: number;
|
|
73
|
+
/** Windows whose rendered text is longer are split into whole-line sub-windows around their hits. */
|
|
74
|
+
maxWindowChars: number;
|
|
75
|
+
/** Same for markdown (sections are long, paragraphs independent). */
|
|
76
|
+
maxProseWindowChars: number;
|
|
77
|
+
};
|
|
78
|
+
wave3: {
|
|
79
|
+
enabled: boolean;
|
|
80
|
+
maxLeadCandidates: number;
|
|
81
|
+
maxLeadsFollowed: number;
|
|
82
|
+
/** Leads are extracted from at most this many of the most relevant passages (all >= T2). */
|
|
83
|
+
seedPassages: number;
|
|
84
|
+
defsPerLead: number;
|
|
85
|
+
};
|
|
86
|
+
status: {
|
|
87
|
+
enabled: boolean;
|
|
88
|
+
/** Max evidence chars sent to the status check (best passages first); keeps the request under Jev's 32k-token limit. */
|
|
89
|
+
maxEvidenceChars: number;
|
|
90
|
+
/** sufficient: overall >= hi and every sub-question >= hi. */
|
|
91
|
+
hi: number;
|
|
92
|
+
/** partial: overall >= lo or any sub-question >= hi. */
|
|
93
|
+
lo: number;
|
|
94
|
+
};
|
|
95
|
+
pack: {
|
|
96
|
+
charsPerToken: number;
|
|
97
|
+
/** Include passages below T2 only if fewer than this many passed (top by score, above minRelevance). */
|
|
98
|
+
minPassages: number;
|
|
99
|
+
minRelevance: number;
|
|
100
|
+
/** Fill the remaining budget with passages below T2 (ranked). */
|
|
101
|
+
fillBudget: boolean;
|
|
102
|
+
moreCandidates: number;
|
|
103
|
+
leadsNotFollowed: number;
|
|
104
|
+
/** A trimmed passage must keep at least this many lines. */
|
|
105
|
+
minTrimLines: number;
|
|
106
|
+
/** File-diversity penalty: a second passage of one file must beat the first passage of another by this margin. */
|
|
107
|
+
filePenalty: number;
|
|
108
|
+
/** Passages whose rendered block is longer are trimmed around their hits to this many chars (0 = off). */
|
|
109
|
+
maxPassageChars: number;
|
|
110
|
+
/** Multiplier on the rel used for pack ORDERING of release-note passages; 1 = off. */
|
|
111
|
+
changelogPrior: number;
|
|
112
|
+
/** Same for prose docs (md/mdx/txt/rst, not release notes); 1 = off. */
|
|
113
|
+
docPrior: number;
|
|
114
|
+
/** Same for test files unless the question is about tests; 1 = off. */
|
|
115
|
+
testPrior: number;
|
|
116
|
+
/** Pack ordering score = (rel + lexWeight x lexical passage score) x prior. */
|
|
117
|
+
lexWeight: number;
|
|
118
|
+
};
|
|
119
|
+
jev: {
|
|
120
|
+
/** Inline the question text in every Jev question when it is at most this long; else reference `question`. */
|
|
121
|
+
inlineQuestionMaxChars: number;
|
|
122
|
+
/** Wave 1: put the true/false rubric on every file question instead of once in the state. */
|
|
123
|
+
fileCriteriaPerQuestion: boolean;
|
|
124
|
+
/** Keep-alive connections opened while recall runs (0 = off). */
|
|
125
|
+
warmConnections: number;
|
|
126
|
+
};
|
|
127
|
+
thresholds: {
|
|
128
|
+
/** Wave 1: file triage floor (p >= T1 selects, up to maxFiles). */
|
|
129
|
+
T1: number;
|
|
130
|
+
/** Wave 2: passage relevance floor for inclusion in the pack, lead seeding and sub-question coverage. */
|
|
131
|
+
T2: number;
|
|
132
|
+
/** Wave 3: lead floor (p >= T3 is followed, up to maxLeadsFollowed). */
|
|
133
|
+
T3: number;
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
export type CodeSearchConfigOverride = {
|
|
137
|
+
[K in keyof CodeSearchConfig]?: Partial<CodeSearchConfig[K]>;
|
|
138
|
+
};
|
|
139
|
+
export declare const DEFAULT_CODE_SEARCH_CONFIG: Readonly<CodeSearchConfig>;
|
|
140
|
+
/** Defaults with a partial override merged on top (unknown keys and wrong types are rejected), validated. */
|
|
141
|
+
export declare function codeSearchConfig(override?: CodeSearchConfigOverride): CodeSearchConfig;
|
|
142
|
+
export declare function validateCodeSearchConfig(c: CodeSearchConfig): void;
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* judge.ts - the only place where scores come from: Jev answers one probability per judged item.
|
|
3
|
+
*
|
|
4
|
+
* All Jev wording lives in PROMPTS below (one place to review). Design rules followed:
|
|
5
|
+
* - one state + many narrow Nouls per request (fan-out); one Noul per candidate (multi-select)
|
|
6
|
+
* - criteria text is explicit (in the state for 60-file triage batches, per question for passages)
|
|
7
|
+
* - the literal question text is inlined in every question (no `collections[i]`-style indirection)
|
|
8
|
+
* - state holds only the candidates being judged (filtered by code first)
|
|
9
|
+
* - no counting, math or dates asked of Jev; code combines answers
|
|
10
|
+
* A Jev failure is not replaced by lexical scores: it propagates and the whole search fails.
|
|
11
|
+
*/
|
|
12
|
+
import { type JevClient, type JevNoulQuestion } from "../client.js";
|
|
13
|
+
import type { CodeSearchConfig } from "./config.js";
|
|
14
|
+
export interface JudgeContext {
|
|
15
|
+
question: string;
|
|
16
|
+
subQuestions: string[];
|
|
17
|
+
}
|
|
18
|
+
export interface FileItem {
|
|
19
|
+
id: string;
|
|
20
|
+
path: string;
|
|
21
|
+
/** path + best hit lines, as shown to the judge */
|
|
22
|
+
descriptor: string;
|
|
23
|
+
/** Normalized lexical score; used only if Jev returns no usable number for this item. */
|
|
24
|
+
lex: number;
|
|
25
|
+
}
|
|
26
|
+
export interface PassageItem {
|
|
27
|
+
id: string;
|
|
28
|
+
path: string;
|
|
29
|
+
start: number;
|
|
30
|
+
end: number;
|
|
31
|
+
/** rendered `N| text` */
|
|
32
|
+
text: string;
|
|
33
|
+
label?: string | undefined;
|
|
34
|
+
lex: number;
|
|
35
|
+
lexCov: number[];
|
|
36
|
+
}
|
|
37
|
+
export interface LeadItem {
|
|
38
|
+
id: string;
|
|
39
|
+
name: string;
|
|
40
|
+
seenAt: string;
|
|
41
|
+
context: string;
|
|
42
|
+
lex: number;
|
|
43
|
+
}
|
|
44
|
+
export interface PassageScore {
|
|
45
|
+
rel: number;
|
|
46
|
+
cov: number[];
|
|
47
|
+
}
|
|
48
|
+
export interface StatusScore {
|
|
49
|
+
overall: number;
|
|
50
|
+
subs: number[];
|
|
51
|
+
}
|
|
52
|
+
export interface StageStats {
|
|
53
|
+
requests: number;
|
|
54
|
+
inputTokens: number;
|
|
55
|
+
costUsd: number;
|
|
56
|
+
ms: number;
|
|
57
|
+
}
|
|
58
|
+
export type JudgeEvent = (stage: string, data: Record<string, unknown>) => void;
|
|
59
|
+
export declare const PROMPTS: {
|
|
60
|
+
fileTask: string;
|
|
61
|
+
fileCriteria: {
|
|
62
|
+
yes: string;
|
|
63
|
+
no: string;
|
|
64
|
+
};
|
|
65
|
+
fileQuestion: (id: string, path: string, q: string) => string;
|
|
66
|
+
fileQuestionPerQ: (id: string, path: string, q: string) => string;
|
|
67
|
+
passageTask: string;
|
|
68
|
+
passageQuestion: (id: string, path: string, a: number, b: number, q: string) => string;
|
|
69
|
+
passageCriteria: {
|
|
70
|
+
true: string;
|
|
71
|
+
false: string;
|
|
72
|
+
};
|
|
73
|
+
coverageQuestion: (id: string, path: string, a: number, b: number, sub: string) => string;
|
|
74
|
+
coverageCriteria: {
|
|
75
|
+
true: string;
|
|
76
|
+
false: string;
|
|
77
|
+
};
|
|
78
|
+
leadTask: string;
|
|
79
|
+
leadCriteria: {
|
|
80
|
+
yes: string;
|
|
81
|
+
no: string;
|
|
82
|
+
};
|
|
83
|
+
leadQuestion: (id: string, name: string, q: string) => string;
|
|
84
|
+
statusTask: string;
|
|
85
|
+
statusQuestion: (q: string) => string;
|
|
86
|
+
statusSubQuestion: (sub: string) => string;
|
|
87
|
+
statusCriteria: {
|
|
88
|
+
true: string;
|
|
89
|
+
false: string;
|
|
90
|
+
};
|
|
91
|
+
};
|
|
92
|
+
export declare function buildFileRequest(items: FileItem[], ctx: JudgeContext, cfg: CodeSearchConfig): {
|
|
93
|
+
state: {
|
|
94
|
+
task: string;
|
|
95
|
+
question: string;
|
|
96
|
+
criteria?: {
|
|
97
|
+
yes: string;
|
|
98
|
+
no: string;
|
|
99
|
+
};
|
|
100
|
+
files: {
|
|
101
|
+
[k: string]: string;
|
|
102
|
+
};
|
|
103
|
+
};
|
|
104
|
+
questions: Record<string, JevNoulQuestion>;
|
|
105
|
+
};
|
|
106
|
+
export declare function buildPassageRequest(items: PassageItem[], ctx: JudgeContext, cfg: CodeSearchConfig): {
|
|
107
|
+
state: {
|
|
108
|
+
task: string;
|
|
109
|
+
question: string;
|
|
110
|
+
passages: {
|
|
111
|
+
[k: string]: {
|
|
112
|
+
file: string;
|
|
113
|
+
lines: string;
|
|
114
|
+
in?: string;
|
|
115
|
+
text: string;
|
|
116
|
+
};
|
|
117
|
+
};
|
|
118
|
+
};
|
|
119
|
+
questions: Record<string, JevNoulQuestion>;
|
|
120
|
+
};
|
|
121
|
+
export declare function buildLeadRequest(items: LeadItem[], ctx: JudgeContext, cfg: CodeSearchConfig): {
|
|
122
|
+
state: {
|
|
123
|
+
task: string;
|
|
124
|
+
question: string;
|
|
125
|
+
criteria: {
|
|
126
|
+
yes: string;
|
|
127
|
+
no: string;
|
|
128
|
+
};
|
|
129
|
+
leads: {
|
|
130
|
+
[k: string]: string;
|
|
131
|
+
};
|
|
132
|
+
};
|
|
133
|
+
questions: Record<string, JevNoulQuestion>;
|
|
134
|
+
};
|
|
135
|
+
export declare function buildStatusRequest(evidence: string, ctx: JudgeContext, cfg: CodeSearchConfig): {
|
|
136
|
+
state: {
|
|
137
|
+
task: string;
|
|
138
|
+
question: string;
|
|
139
|
+
evidence: string;
|
|
140
|
+
};
|
|
141
|
+
questions: Record<string, JevNoulQuestion>;
|
|
142
|
+
};
|
|
143
|
+
/** Group passages into requests: consecutive (same-file first) up to perRequest items and maxChars of text. */
|
|
144
|
+
export declare function chunkPassages(items: PassageItem[], perRequest: number, maxChars: number): PassageItem[][];
|
|
145
|
+
/** Split into the fewest groups of at most n items, as equal in size as possible (69 -> 35 + 34, not 60 + 9). */
|
|
146
|
+
export declare function chunkEven<T>(xs: T[], n: number): T[][];
|
|
147
|
+
export declare class JevJudge {
|
|
148
|
+
private readonly o;
|
|
149
|
+
private readonly stageStats;
|
|
150
|
+
private jevModel;
|
|
151
|
+
constructor(o: {
|
|
152
|
+
client: JevClient;
|
|
153
|
+
config: CodeSearchConfig;
|
|
154
|
+
signal: AbortSignal;
|
|
155
|
+
onEvent?: JudgeEvent | undefined;
|
|
156
|
+
});
|
|
157
|
+
stats(): Record<string, StageStats>;
|
|
158
|
+
model(): string | null;
|
|
159
|
+
private stat;
|
|
160
|
+
/** One logical request (the client may split its questions into several HTTP requests). */
|
|
161
|
+
private ask;
|
|
162
|
+
/** Run a stage's batches in parallel; the first failure rejects the stage. */
|
|
163
|
+
private stage;
|
|
164
|
+
scoreFiles(items: FileItem[], ctx: JudgeContext): Promise<Map<string, number>>;
|
|
165
|
+
scorePassages(items: PassageItem[], ctx: JudgeContext, stage?: string): Promise<Map<string, PassageScore>>;
|
|
166
|
+
scoreLeads(items: LeadItem[], ctx: JudgeContext): Promise<Map<string, number>>;
|
|
167
|
+
/** null = nothing to check (empty evidence). */
|
|
168
|
+
status(evidence: string, ctx: JudgeContext): Promise<StatusScore | null>;
|
|
169
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* leads.ts - wave 3: identifiers referenced by relevant passages whose definitions were not read.
|
|
3
|
+
*
|
|
4
|
+
* Extraction (code only): called functions, imported names, PascalCase types, UPPER_CASE constants,
|
|
5
|
+
* env/config keys and camelCase member accesses. Excluded: names already searched (keywords and their
|
|
6
|
+
* variants), names defined inside the selected passages, a stoplist of builtins/common helpers, and
|
|
7
|
+
* names shorter than 4 chars. Definitions are located with one ripgrep regex call (a few when the names do
|
|
8
|
+
* not fit one pattern under CODE_SEARCH_MAX_PATTERN_CHARS).
|
|
9
|
+
*/
|
|
10
|
+
import type { CodeSearchConfig } from "./config.js";
|
|
11
|
+
import { type WorkspaceSession } from "./session.js";
|
|
12
|
+
export interface SeedPassage {
|
|
13
|
+
path: string;
|
|
14
|
+
start: number;
|
|
15
|
+
/** raw file lines of the passage (no `N|` prefixes) */
|
|
16
|
+
lines: string[];
|
|
17
|
+
rel: number;
|
|
18
|
+
}
|
|
19
|
+
export interface LeadCandidate {
|
|
20
|
+
name: string;
|
|
21
|
+
/** weighted frequency: sum of relevance of seed passages mentioning it + occurrence bonus */
|
|
22
|
+
weight: number;
|
|
23
|
+
occurrences: number;
|
|
24
|
+
called: boolean;
|
|
25
|
+
seenAt: {
|
|
26
|
+
path: string;
|
|
27
|
+
line: number;
|
|
28
|
+
};
|
|
29
|
+
context: string;
|
|
30
|
+
kinds: string[];
|
|
31
|
+
}
|
|
32
|
+
export declare const LEAD_STOPLIST: Set<string>;
|
|
33
|
+
/** Names declared inside a passage (so they are not leads). */
|
|
34
|
+
export declare function definedNames(text: string): Set<string>;
|
|
35
|
+
/** Raw identifier occurrences (name, kind) in one line. */
|
|
36
|
+
export declare function identifiersInLine(line: string): Array<{
|
|
37
|
+
name: string;
|
|
38
|
+
kind: string;
|
|
39
|
+
}>;
|
|
40
|
+
/**
|
|
41
|
+
* Extract lead candidates from seed passages (most relevant first), ranked by weighted frequency.
|
|
42
|
+
* `searched` = lowercased keywords + variants (already searched; not leads).
|
|
43
|
+
*/
|
|
44
|
+
export declare function extractLeads(seeds: SeedPassage[], searched: Set<string>, questionWords: Set<string>, max: number): LeadCandidate[];
|
|
45
|
+
export interface DefinitionHit {
|
|
46
|
+
name: string;
|
|
47
|
+
path: string;
|
|
48
|
+
line: number;
|
|
49
|
+
kind: "decl" | "sqlfn" | "method" | "key";
|
|
50
|
+
text: string;
|
|
51
|
+
}
|
|
52
|
+
/** The JS regexes used to classify an rg match line for one name (priority order). */
|
|
53
|
+
export declare function definitionKinds(name: string): Array<{
|
|
54
|
+
kind: DefinitionHit["kind"];
|
|
55
|
+
re: RegExp;
|
|
56
|
+
}>;
|
|
57
|
+
/**
|
|
58
|
+
* rg pattern (ASCII mode) for definition-shaped lines of any of the names: declarations, SQL functions,
|
|
59
|
+
* methods and object keys / fields / config keys. Classified per name in JS with definitionKinds().
|
|
60
|
+
*/
|
|
61
|
+
export declare function definitionPattern(names: readonly string[]): string;
|
|
62
|
+
/**
|
|
63
|
+
* definitionPattern over consecutive groups of names, each pattern at most maxChars. A name whose own pattern
|
|
64
|
+
* is longer is left out: it gets no definition, as after a failed search.
|
|
65
|
+
*/
|
|
66
|
+
export declare function definitionPatterns(names: readonly string[], maxChars?: number): string[];
|
|
67
|
+
/** Package root of a path: the first two segments for apps/ and packages/ (apps/worker, packages/runtime), else the first. */
|
|
68
|
+
export declare function packageOf(path: string): string;
|
|
69
|
+
/**
|
|
70
|
+
* Choose the best definitions per name. A name can be defined in many places (`search`, `isRecord`), so locality
|
|
71
|
+
* dominates: the file where the lead was seen, then its package; then kind rank (declaration > method > key),
|
|
72
|
+
* non-test, an already-selected file, a config-ish path for keys, path, line.
|
|
73
|
+
*/
|
|
74
|
+
export declare function chooseDefinitions(hits: DefinitionHit[], preferPaths: Set<string>, perName: number, allowTests?: boolean, seenAt?: Map<string, string>): Map<string, DefinitionHit[]>;
|
|
75
|
+
export interface DefinitionSearch {
|
|
76
|
+
hits: DefinitionHit[];
|
|
77
|
+
/** name -> number of repository files that mention it as a word (for an IDF-style genericity penalty) */
|
|
78
|
+
fileCounts: Map<string, number>;
|
|
79
|
+
ms: number;
|
|
80
|
+
}
|
|
81
|
+
/** Files a definition search never needs: prose/data (definitions of code identifiers live in code or SQL). */
|
|
82
|
+
export declare const DEF_SCAN_EXCLUDES: string[];
|
|
83
|
+
/** Test files, excluded from the definition scan unless the question is about tests. */
|
|
84
|
+
export declare const TEST_EXCLUDES: string[];
|
|
85
|
+
/**
|
|
86
|
+
* ONE rg pass (ASCII-mode regex, full lines) for definition-shaped lines of all names over code and SQL files
|
|
87
|
+
* (tests only when the question is about tests), classified per name in JS: declarations/SQL functions/methods
|
|
88
|
+
* first; object keys / fields / config keys (`name:` / `name =`) only for names without a stronger definition,
|
|
89
|
+
* and a name whose only definitions are more than maxKeyOnlyDefs keys is a generic field (sessionId, accountId)
|
|
90
|
+
* and gets none. fileCounts = files with any definition-shaped line for the name (generic fields and helpers
|
|
91
|
+
* redefined everywhere score high), used as a genericity penalty. About 0.4 CPU-s on a 6k-file repository; a
|
|
92
|
+
* second pass that counted every mention cost another ~0.7 CPU-s and full-line mention output was 80k lines.
|
|
93
|
+
* Names that do not fit one pattern under the cap are searched in a few passes, merged line by line.
|
|
94
|
+
* A failed definition search yields no hits (the leads are then dropped), as in scout.
|
|
95
|
+
*/
|
|
96
|
+
export declare function locateDefinitions(session: WorkspaceSession, names: string[], cfg: CodeSearchConfig, excludeArgs: string[], maxKeyOnlyDefs?: number, allowTests?: boolean): Promise<DefinitionSearch>;
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* pack.ts - bounded evidence pack.
|
|
3
|
+
*
|
|
4
|
+
* Priority: (1) for each sub-question, the best passage covering it (coverage >= T2), (2) passages with
|
|
5
|
+
* relevance >= T2 by relevance, (3) if fewer than minPassages passed, the next best above minRelevance,
|
|
6
|
+
* (4) optionally (fillBudget) the rest. Greedy under a char budget (tokens * charsPerToken); a passage that
|
|
7
|
+
* does not fit is trimmed to whole lines around its hits (>= minTrimLines) and marked, never cut mid-line.
|
|
8
|
+
* Output groups passages by file (files by best relevance, passages by line).
|
|
9
|
+
*/
|
|
10
|
+
import type { CodeSearchConfig } from "./config.js";
|
|
11
|
+
import { type RenderOpts } from "./windows.js";
|
|
12
|
+
export interface EvidencePassage {
|
|
13
|
+
id: string;
|
|
14
|
+
path: string;
|
|
15
|
+
start: number;
|
|
16
|
+
end: number;
|
|
17
|
+
fileLines: string[];
|
|
18
|
+
hits: number[];
|
|
19
|
+
label?: {
|
|
20
|
+
line: number;
|
|
21
|
+
text: string;
|
|
22
|
+
} | undefined;
|
|
23
|
+
kind: "hit" | "header" | "def";
|
|
24
|
+
rel: number;
|
|
25
|
+
cov: number[];
|
|
26
|
+
/** Lexical passage score (used by pack.lexWeight). */
|
|
27
|
+
lex?: number | undefined;
|
|
28
|
+
lead?: string | undefined;
|
|
29
|
+
/** How lines are rendered (line cap + needles for cutting very long lines around a hit). */
|
|
30
|
+
render?: RenderOpts | undefined;
|
|
31
|
+
}
|
|
32
|
+
export interface PackedPassage {
|
|
33
|
+
id: string;
|
|
34
|
+
path: string;
|
|
35
|
+
start: number;
|
|
36
|
+
end: number;
|
|
37
|
+
origStart: number;
|
|
38
|
+
origEnd: number;
|
|
39
|
+
trimmed: boolean;
|
|
40
|
+
rel: number;
|
|
41
|
+
cov: number[];
|
|
42
|
+
kind: EvidencePassage["kind"];
|
|
43
|
+
lead?: string | undefined;
|
|
44
|
+
label?: {
|
|
45
|
+
line: number;
|
|
46
|
+
text: string;
|
|
47
|
+
} | undefined;
|
|
48
|
+
/** rendered block including its `==` header line */
|
|
49
|
+
block: string;
|
|
50
|
+
}
|
|
51
|
+
export interface PackOptions {
|
|
52
|
+
subQuestions: string[];
|
|
53
|
+
T2: number;
|
|
54
|
+
/** Chars available for passage blocks (header/footer excluded). */
|
|
55
|
+
bodyChars: number;
|
|
56
|
+
cfg: CodeSearchConfig;
|
|
57
|
+
/** Apply pack.changelogPrior (false when the question is about history / releases). */
|
|
58
|
+
downweightChangelogs?: boolean;
|
|
59
|
+
/** Apply pack.testPrior (false when the question is about tests). */
|
|
60
|
+
downweightTests?: boolean;
|
|
61
|
+
}
|
|
62
|
+
export interface PackBody {
|
|
63
|
+
included: PackedPassage[];
|
|
64
|
+
excluded: EvidencePassage[];
|
|
65
|
+
body: string;
|
|
66
|
+
priority: string[];
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Greedy file-diverse order: repeatedly take the passage with the highest eff(x) - penalty x (passages already
|
|
70
|
+
* taken from its file). penalty 0 = plain descending eff.
|
|
71
|
+
*/
|
|
72
|
+
export declare function diverseOrder(xs: EvidencePassage[], eff: (x: EvidencePassage) => number, penalty: number, taken: Map<string, number>): EvidencePassage[];
|
|
73
|
+
/** Path prior for pack ordering (release notes, prose docs, tests); 1 for code. */
|
|
74
|
+
export declare function pathPrior(path: string, o: Pick<PackOptions, "cfg" | "downweightChangelogs" | "downweightTests">): number;
|
|
75
|
+
/** Pack ordering score: (rel + lexWeight x lex) x path prior. Inclusion floors (T2, minRelevance) still use raw rel. */
|
|
76
|
+
export declare function packScore(x: EvidencePassage, o: Pick<PackOptions, "cfg" | "downweightChangelogs" | "downweightTests">): number;
|
|
77
|
+
export declare function priorityOrder(passages: EvidencePassage[], o: Pick<PackOptions, "subQuestions" | "T2" | "cfg" | "downweightChangelogs" | "downweightTests">): EvidencePassage[];
|
|
78
|
+
export declare function renderBlock(x: EvidencePassage, start: number, end: number, o: PackOptions): string;
|
|
79
|
+
/** Largest contiguous whole-line sub-range around the passage's hits whose block fits in maxChars. */
|
|
80
|
+
export declare function trimToFit(x: EvidencePassage, maxChars: number, o: PackOptions): {
|
|
81
|
+
start: number;
|
|
82
|
+
end: number;
|
|
83
|
+
} | null;
|
|
84
|
+
export declare function packBody(passages: EvidencePassage[], o: PackOptions): PackBody;
|
|
85
|
+
export interface FooterInput {
|
|
86
|
+
excluded: EvidencePassage[];
|
|
87
|
+
otherFiles: Array<{
|
|
88
|
+
path: string;
|
|
89
|
+
score: number;
|
|
90
|
+
}>;
|
|
91
|
+
leadsNotFollowed: Array<{
|
|
92
|
+
name: string;
|
|
93
|
+
score: number;
|
|
94
|
+
seenAt: string;
|
|
95
|
+
note?: string;
|
|
96
|
+
}>;
|
|
97
|
+
zeroHitKeywords: Array<{
|
|
98
|
+
raw: string;
|
|
99
|
+
fragments: string[];
|
|
100
|
+
}>;
|
|
101
|
+
widenedNote?: string;
|
|
102
|
+
cfg: CodeSearchConfig;
|
|
103
|
+
maxChars: number;
|
|
104
|
+
}
|
|
105
|
+
/** Footer lines, dropping list entries from the end until it fits maxChars. */
|
|
106
|
+
export declare function renderFooter(f: FooterInput): string;
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* recall.ts - wide, deterministic lexical recall with ripgrep.
|
|
3
|
+
*
|
|
4
|
+
* One ripgrep pass over the union of every keyword's identifier variants, searched case-insensitively
|
|
5
|
+
* (camelCase <-> snake_case <-> kebab <-> spaced phrase; SCREAMING and Pascal forms are covered by -i);
|
|
6
|
+
* a union longer than CODE_SEARCH_MAX_PATTERN_CHARS is split into several passes and merged. Lines are
|
|
7
|
+
* attributed to keywords in JS. Short plain words get a leading word boundary so `turn` does not
|
|
8
|
+
* match `return`. Per-file score = sum over DISTINCT matched keywords of IDF + path bonus + a small
|
|
9
|
+
* hit-count tie breaker; tests are down-weighted unless the question is about tests.
|
|
10
|
+
*/
|
|
11
|
+
import type { CodeSearchConfig } from "./config.js";
|
|
12
|
+
import { type WorkspaceSession } from "./session.js";
|
|
13
|
+
export declare const BUILTIN_EXCLUDES: string[];
|
|
14
|
+
export interface KeywordInfo {
|
|
15
|
+
index: number;
|
|
16
|
+
raw: string;
|
|
17
|
+
variants: string[];
|
|
18
|
+
/** "phrase" = any variant as a substring; "word-prefix" = `\bword`. */
|
|
19
|
+
mode: "phrase" | "word-prefix";
|
|
20
|
+
/** JS-dialect pattern (attribution, window scoring, line cutting). */
|
|
21
|
+
pattern: string;
|
|
22
|
+
/** ripgrep-dialect pattern: same language, but word boundaries are ASCII `(?-u:\b)`. */
|
|
23
|
+
rgPattern: string;
|
|
24
|
+
/** Number of files with at least one matching line (content). */
|
|
25
|
+
df: number;
|
|
26
|
+
/** Files whose path matches (no content needed). */
|
|
27
|
+
pathDf: number;
|
|
28
|
+
hitLines: number;
|
|
29
|
+
idf: number;
|
|
30
|
+
/** Fragments used because the full keyword had zero hits. */
|
|
31
|
+
fragments: string[];
|
|
32
|
+
}
|
|
33
|
+
export interface HitLine {
|
|
34
|
+
line: number;
|
|
35
|
+
text: string;
|
|
36
|
+
kws: number[];
|
|
37
|
+
}
|
|
38
|
+
export interface FileCandidate {
|
|
39
|
+
path: string;
|
|
40
|
+
lexScore: number;
|
|
41
|
+
/** keyword index -> matching line count */
|
|
42
|
+
kwHits: Record<number, number>;
|
|
43
|
+
pathKws: number[];
|
|
44
|
+
hitLines: Map<number, HitLine>;
|
|
45
|
+
isTest: boolean;
|
|
46
|
+
isDoc: boolean;
|
|
47
|
+
}
|
|
48
|
+
export interface RecallResult {
|
|
49
|
+
keywords: KeywordInfo[];
|
|
50
|
+
totalFiles: number;
|
|
51
|
+
candidates: FileCandidate[];
|
|
52
|
+
scoredFiles: number;
|
|
53
|
+
searchPaths: string[];
|
|
54
|
+
widened: boolean;
|
|
55
|
+
/** Path prefixes that do not exist in the workspace (or are not workspace-relative); ignored. */
|
|
56
|
+
missingPrefixes: string[];
|
|
57
|
+
/** The path prefixes that exist, workspace-relative. */
|
|
58
|
+
validPrefixes: string[];
|
|
59
|
+
/** Path-only candidates dropped because their content is binary (NUL in the first 8 KB). */
|
|
60
|
+
binaryDropped: number;
|
|
61
|
+
ms: number;
|
|
62
|
+
}
|
|
63
|
+
export interface RecallInput {
|
|
64
|
+
session: WorkspaceSession;
|
|
65
|
+
question: string;
|
|
66
|
+
keywords: string[];
|
|
67
|
+
pathPrefixes: string[];
|
|
68
|
+
config: CodeSearchConfig;
|
|
69
|
+
}
|
|
70
|
+
export declare function excludeArgs(cfg: CodeSearchConfig): string[];
|
|
71
|
+
/** `--hidden` (rg skips dot-dirs such as .github/ by default); `.git/` stays excluded by BUILTIN_EXCLUDES. */
|
|
72
|
+
export declare function hiddenArgs(cfg: CodeSearchConfig): string[];
|
|
73
|
+
export declare function listFiles(session: WorkspaceSession, paths: string[], cfg: CodeSearchConfig): Promise<string[]>;
|
|
74
|
+
/**
|
|
75
|
+
* Search pattern for one keyword. `pattern` is JS-dialect; `rgPattern` is the same for ripgrep except that the
|
|
76
|
+
* word boundary is ASCII: a Unicode `\b` under -i disables ripgrep's fast engines (measured on repo-snap: 5.6 s
|
|
77
|
+
* user CPU for one pass vs 0.2 s with `(?-u:\b)`), and JS `\b` is ASCII anyway.
|
|
78
|
+
*/
|
|
79
|
+
export declare function buildPattern(kw: string, cfg: CodeSearchConfig): {
|
|
80
|
+
pattern: string;
|
|
81
|
+
rgPattern: string;
|
|
82
|
+
mode: "phrase" | "word-prefix";
|
|
83
|
+
variants: string[];
|
|
84
|
+
};
|
|
85
|
+
/** Pattern for the 2-word fragments of a compound keyword (used only if the full keyword has zero hits). */
|
|
86
|
+
export declare function fragmentPattern(frags: string[]): string;
|
|
87
|
+
export interface RgMatch {
|
|
88
|
+
path: string;
|
|
89
|
+
line: number;
|
|
90
|
+
text: string;
|
|
91
|
+
}
|
|
92
|
+
/** Parse `rg --null --line-number --with-filename --no-heading` output. Omitted long lines are skipped. */
|
|
93
|
+
export declare function parseRgOutput(out: string): RgMatch[];
|
|
94
|
+
/**
|
|
95
|
+
* Whether a pattern built by this engine compiles: checked as a JS regex after mapping ripgrep's inline flag
|
|
96
|
+
* groups (`(?-u:`, `(?i:`) to plain groups. It tells an invalid pattern apart from ripgrep's exit 2 for an
|
|
97
|
+
* unreadable path.
|
|
98
|
+
*/
|
|
99
|
+
export declare function ripgrepPatternCompiles(pattern: string): boolean;
|
|
100
|
+
/** One rg pass for a pattern (any number of alternatives), all matching lines, sorted by path then line. */
|
|
101
|
+
export declare function searchPattern(session: WorkspaceSession, pattern: string, paths: string[], cfg: CodeSearchConfig, opts?: {
|
|
102
|
+
caseInsensitive?: boolean;
|
|
103
|
+
word?: boolean;
|
|
104
|
+
maxPerFile?: number;
|
|
105
|
+
}): Promise<RgMatch[]>;
|
|
106
|
+
/** Split a regex at its top-level `|` (outside groups, character classes and escapes). */
|
|
107
|
+
export declare function splitAlternation(pattern: string): string[];
|
|
108
|
+
/**
|
|
109
|
+
* Join alternatives with `|`, in order, into as few patterns of at most maxChars as possible. An alternative
|
|
110
|
+
* longer than maxChars on its own gets a pattern of its own.
|
|
111
|
+
*/
|
|
112
|
+
export declare function packAlternatives(alternatives: readonly string[], maxChars: number): string[];
|
|
113
|
+
/** Matches of several searches, as one search over their union returns them: by path then line, each line once. */
|
|
114
|
+
export declare function mergeMatches(lists: readonly RgMatch[][]): RgMatch[];
|
|
115
|
+
/**
|
|
116
|
+
* The lines matching `(?:a)|(?:b)|...` over the alternatives. A union longer than maxPatternChars is split
|
|
117
|
+
* into several rg passes (an alternative too long on its own is split at its own top-level `|`), run a few
|
|
118
|
+
* at a time and merged, so the result is the same as one pass. A part that still does not fit (a keyword
|
|
119
|
+
* of thousands of characters) is not searched, so it reports zero hits.
|
|
120
|
+
*/
|
|
121
|
+
export declare function searchAlternatives(session: WorkspaceSession, alternatives: readonly string[], paths: string[], cfg: CodeSearchConfig, maxPatternChars?: number): Promise<RgMatch[]>;
|
|
122
|
+
export declare function idfOf(df: number, n: number): number;
|
|
123
|
+
/** Does the path mention the keyword (any variant, separators ignored)? */
|
|
124
|
+
export declare function pathMatches(path: string, variants: string[]): boolean;
|
|
125
|
+
/** JS regex for an rg pattern built by buildPattern/fragmentPattern (the escapes used are valid in both dialects). */
|
|
126
|
+
export declare function jsRegex(pattern: string): RegExp | null;
|
|
127
|
+
export interface Slot {
|
|
128
|
+
/** keyword index */
|
|
129
|
+
ki: number;
|
|
130
|
+
fragment: boolean;
|
|
131
|
+
re: RegExp | null;
|
|
132
|
+
literals: string[];
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* Attribute rg lines to keyword slots. Returns per slot the matches (all lines; the caller caps what it stores).
|
|
136
|
+
* A JS regex that fails to compile falls back to a case-insensitive literal test on the variants.
|
|
137
|
+
*/
|
|
138
|
+
export declare function attribute(matches: RgMatch[], slots: Slot[]): RgMatch[][];
|
|
139
|
+
/**
|
|
140
|
+
* Normalize path prefixes to workspace-relative form ("./src/" -> "src"); prefixes that do not exist, are
|
|
141
|
+
* absolute or climb out with ".." are reported as missing instead of crashing ripgrep (exit 2, no output).
|
|
142
|
+
*/
|
|
143
|
+
export declare function normalizePrefixes(session: WorkspaceSession, prefixes: string[]): Promise<{
|
|
144
|
+
ok: string[];
|
|
145
|
+
missing: string[];
|
|
146
|
+
}>;
|
|
147
|
+
/** Workspace-relative form of a prefix, or null when it is absolute, climbs out, or starts with "-". */
|
|
148
|
+
export declare function cleanPrefix(p: string): string | null;
|
|
149
|
+
export declare function recall(input: RecallInput): Promise<RecallResult>;
|
|
150
|
+
/** Weight of one hit line: sum of IDF of the distinct keywords on it. */
|
|
151
|
+
export declare function lineWeight(h: HitLine, keywords: KeywordInfo[]): number;
|
|
152
|
+
/** Up to n best hit lines of a file (by distinct-keyword IDF), returned in line order (long lines are trimmed by the caller). */
|
|
153
|
+
export declare function bestHitLines(c: FileCandidate, keywords: KeywordInfo[], n: number): HitLine[];
|
|
154
|
+
/** Words of all keywords (for overlap scoring). */
|
|
155
|
+
export declare function keywordWords(keywords: KeywordInfo[]): Set<string>;
|