@opengeni/jev 0.1.0-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +190 -0
- package/README.md +140 -0
- package/dist/circuit-breaker.d.ts +58 -0
- package/dist/client.d.ts +173 -0
- package/dist/code-search/config.d.ts +142 -0
- package/dist/code-search/judge.d.ts +169 -0
- package/dist/code-search/leads.d.ts +96 -0
- package/dist/code-search/pack.d.ts +106 -0
- package/dist/code-search/recall.d.ts +155 -0
- package/dist/code-search/search.d.ts +93 -0
- package/dist/code-search/session.d.ts +33 -0
- package/dist/code-search/text.d.ts +35 -0
- package/dist/code-search/tool.d.ts +34 -0
- package/dist/code-search/windows.d.ts +85 -0
- package/dist/code-search/workspace.d.ts +51 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +3110 -0
- package/dist/index.js.map +1 -0
- package/package.json +39 -0
- package/src/circuit-breaker.ts +135 -0
- package/src/client.ts +577 -0
- package/src/code-search/config.ts +282 -0
- package/src/code-search/judge.ts +413 -0
- package/src/code-search/leads.ts +442 -0
- package/src/code-search/pack.ts +354 -0
- package/src/code-search/recall.ts +648 -0
- package/src/code-search/search.ts +773 -0
- package/src/code-search/session.ts +89 -0
- package/src/code-search/text.ts +159 -0
- package/src/code-search/tool.ts +209 -0
- package/src/code-search/windows.ts +617 -0
- package/src/code-search/workspace.ts +55 -0
- package/src/index.ts +71 -0
|
@@ -0,0 +1,442 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* leads.ts - wave 3: identifiers referenced by relevant passages whose definitions were not read.
|
|
3
|
+
*
|
|
4
|
+
* Extraction (code only): called functions, imported names, PascalCase types, UPPER_CASE constants,
|
|
5
|
+
* env/config keys and camelCase member accesses. Excluded: names already searched (keywords and their
|
|
6
|
+
* variants), names defined inside the selected passages, a stoplist of builtins/common helpers, and
|
|
7
|
+
* names shorter than 4 chars. Definitions are located with one ripgrep regex call (a few when the names do
|
|
8
|
+
* not fit one pattern under CODE_SEARCH_MAX_PATTERN_CHARS).
|
|
9
|
+
*/
|
|
10
|
+
import type { CodeSearchConfig } from "./config";
|
|
11
|
+
import { mapLimit, RIPGREP_SPLIT_CONCURRENCY, type WorkspaceSession } from "./session";
|
|
12
|
+
import { escapeRegex, isTestPath, splitWords } from "./text";
|
|
13
|
+
import { CODE_SEARCH_MAX_PATTERN_CHARS } from "./workspace";
|
|
14
|
+
|
|
15
|
+
export interface SeedPassage {
|
|
16
|
+
path: string;
|
|
17
|
+
start: number;
|
|
18
|
+
/** raw file lines of the passage (no `N|` prefixes) */
|
|
19
|
+
lines: string[];
|
|
20
|
+
rel: number;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export interface LeadCandidate {
|
|
24
|
+
name: string;
|
|
25
|
+
/** weighted frequency: sum of relevance of seed passages mentioning it + occurrence bonus */
|
|
26
|
+
weight: number;
|
|
27
|
+
occurrences: number;
|
|
28
|
+
called: boolean;
|
|
29
|
+
seenAt: { path: string; line: number };
|
|
30
|
+
context: string;
|
|
31
|
+
kinds: string[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export const LEAD_STOPLIST = new Set(
|
|
35
|
+
// JS/TS builtins, globals, utility types
|
|
36
|
+
(
|
|
37
|
+
"Promise Array Object String Number Boolean Map Set WeakMap WeakSet Symbol Error TypeError RangeError SyntaxError " +
|
|
38
|
+
"JSON Math Date RegExp BigInt Buffer URL URLSearchParams console require parseInt parseFloat setTimeout clearTimeout " +
|
|
39
|
+
"setInterval clearInterval setImmediate queueMicrotask isNaN isFinite encodeURIComponent decodeURIComponent " +
|
|
40
|
+
"structuredClone fetch Response Request Headers FormData Blob File AbortController AbortSignal TextEncoder TextDecoder " +
|
|
41
|
+
"Uint8Array Int32Array Float64Array ArrayBuffer DataView Record Partial Readonly ReadonlyArray Pick Omit Exclude Extract " +
|
|
42
|
+
"ReturnType Parameters Awaited NonNullable Required InstanceType Iterable AsyncIterable Generator AsyncGenerator " +
|
|
43
|
+
"PromiseLike Function Uppercase Lowercase keyof typeof instanceof undefined null true false this super " +
|
|
44
|
+
"process Bun Deno globalThis window document performance crypto " +
|
|
45
|
+
// test helpers
|
|
46
|
+
"expect describe test beforeEach afterEach beforeAll afterAll mock spyOn toBe toEqual toStrictEqual toMatch " +
|
|
47
|
+
"toContain toThrow toHaveBeenCalled toHaveBeenCalledWith toHaveLength toBeDefined toBeUndefined toBeNull toBeTruthy " +
|
|
48
|
+
"toBeFalsy toMatchObject toBeGreaterThan toBeLessThan resolves rejects " +
|
|
49
|
+
// very common methods
|
|
50
|
+
"push pop shift unshift map filter reduce forEach find findIndex some every includes join split slice splice concat " +
|
|
51
|
+
"indexOf lastIndexOf keys values entries from assign freeze stringify parse toString valueOf trim trimStart trimEnd " +
|
|
52
|
+
"replace replaceAll match matchAll exec startsWith endsWith toLowerCase toUpperCase padStart padEnd localeCompare " +
|
|
53
|
+
"then catch finally resolve reject allSettled race floor ceil round abs sqrt random sort reverse flat flatMap fill " +
|
|
54
|
+
"length size delete clear warn error info debug trace assert emit once listen close write read send next done " +
|
|
55
|
+
"toISOString getTime toFixed charAt charCodeAt codePointAt fromEntries isArray hasOwnProperty defineProperty " +
|
|
56
|
+
"getOwnPropertyNames create apply call bind " +
|
|
57
|
+
// drizzle / zod / sql helpers
|
|
58
|
+
"select insert update values returning where from innerJoin leftJoin orderBy groupBy limit offset execute " +
|
|
59
|
+
"inArray isNull isNotNull desc asc sql eq ne gt gte lt lte and or not like ilike between exists " +
|
|
60
|
+
"object string number boolean array optional nullable nullish literal union enum infer default describe " +
|
|
61
|
+
"safeParse parseAsync strict passthrough extend merge partial refine superRefine transform coerce positive " +
|
|
62
|
+
"nonnegative int min max email uuid regex " +
|
|
63
|
+
// SQL functions
|
|
64
|
+
"coalesce count now greatest least jsonb_build_object jsonb_set jsonb_agg array_agg format lower upper nullif " +
|
|
65
|
+
"current_setting set_config gen_random_uuid clock_timestamp " +
|
|
66
|
+
// Rust
|
|
67
|
+
"Ok Err Some None Box Vec Arc Mutex RwLock Option Result clone unwrap expect into iter collect map_err ok_or as_ref " +
|
|
68
|
+
"to_string to_owned println eprintln format vec"
|
|
69
|
+
).split(/\s+/),
|
|
70
|
+
);
|
|
71
|
+
|
|
72
|
+
const CONTROL = new Set([
|
|
73
|
+
"if",
|
|
74
|
+
"for",
|
|
75
|
+
"while",
|
|
76
|
+
"switch",
|
|
77
|
+
"catch",
|
|
78
|
+
"return",
|
|
79
|
+
"function",
|
|
80
|
+
"typeof",
|
|
81
|
+
"await",
|
|
82
|
+
"async",
|
|
83
|
+
"new",
|
|
84
|
+
"else",
|
|
85
|
+
"case",
|
|
86
|
+
"void",
|
|
87
|
+
"delete",
|
|
88
|
+
"yield",
|
|
89
|
+
"import",
|
|
90
|
+
"export",
|
|
91
|
+
"constructor",
|
|
92
|
+
"super",
|
|
93
|
+
"this",
|
|
94
|
+
"static",
|
|
95
|
+
"private",
|
|
96
|
+
"public",
|
|
97
|
+
"protected",
|
|
98
|
+
]);
|
|
99
|
+
|
|
100
|
+
/** Names declared inside a passage (so they are not leads). */
|
|
101
|
+
export function definedNames(text: string): Set<string> {
|
|
102
|
+
const out = new Set<string>();
|
|
103
|
+
const res = [
|
|
104
|
+
/\bfunction\*?\s+([A-Za-z_$][\w$]*)/g,
|
|
105
|
+
/\b(?:const|let|var)\s+([A-Za-z_$][\w$]*)/g,
|
|
106
|
+
/\b(?:class|interface|type|enum|namespace|struct|trait|fn|def|func|mod)\s+([A-Za-z_$][\w$]*)/g,
|
|
107
|
+
/(?:FUNCTION|function)\s+(?:[\w"]+\.)?"?([A-Za-z_][\w]*)"?\s*\(/g,
|
|
108
|
+
/^\s*(?:(?:public|private|protected|static|async|readonly|override|get|set)\s+)*([A-Za-z_$][\w$]*)\s*(?:<[^>()]*>)?\s*\([^;]*\)\s*(?::[^={;]+)?\{\s*$/gm,
|
|
109
|
+
];
|
|
110
|
+
for (const re of res) for (const m of text.matchAll(re)) if (!CONTROL.has(m[1]!)) out.add(m[1]!);
|
|
111
|
+
// destructured consts: const { a, b: c } = ...
|
|
112
|
+
for (const m of text.matchAll(/\b(?:const|let|var)\s*\{([^}]*)\}/g)) {
|
|
113
|
+
for (const part of m[1]!.split(",")) {
|
|
114
|
+
const name = part.split(":").pop()!.split("=")[0]!.trim();
|
|
115
|
+
if (/^[A-Za-z_$][\w$]*$/.test(name)) out.add(name);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return out;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Raw identifier occurrences (name, kind) in one line. */
|
|
122
|
+
export function identifiersInLine(line: string): Array<{ name: string; kind: string }> {
|
|
123
|
+
const out: Array<{ name: string; kind: string }> = [];
|
|
124
|
+
for (const m of line.matchAll(/\b([A-Za-z_$][\w$]*)\s*(?:<[^<>()]*>)?\s*\(/g))
|
|
125
|
+
out.push({ name: m[1]!, kind: "call" });
|
|
126
|
+
const imp = /import\s+(?:type\s+)?\{([^}]*)\}/.exec(line);
|
|
127
|
+
if (imp) {
|
|
128
|
+
for (const part of imp[1]!.split(",")) {
|
|
129
|
+
const name = part
|
|
130
|
+
.replace(/^\s*type\s+/, "")
|
|
131
|
+
.split(/\s+as\s+/)[0]!
|
|
132
|
+
.trim();
|
|
133
|
+
if (/^[A-Za-z_$][\w$]*$/.test(name)) out.push({ name, kind: "import" });
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
for (const m of line.matchAll(/\b([A-Z][a-z0-9]+(?:[A-Z][A-Za-z0-9]*)+)\b/g))
|
|
137
|
+
out.push({ name: m[1]!, kind: "type" });
|
|
138
|
+
for (const m of line.matchAll(/\b([A-Z][A-Z0-9]*(?:_[A-Z0-9]+)+)\b/g))
|
|
139
|
+
out.push({ name: m[1]!, kind: "constant" });
|
|
140
|
+
for (const m of line.matchAll(/\.([a-z][a-z0-9]*[A-Z][A-Za-z0-9]*)\b/g))
|
|
141
|
+
out.push({ name: m[1]!, kind: "member" });
|
|
142
|
+
return out;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function trimContext(line: string, name: string, max = 160): string {
|
|
146
|
+
const t = line.trim().replace(/\s+/g, " ");
|
|
147
|
+
if (t.length <= max) return t;
|
|
148
|
+
const i = Math.max(0, t.indexOf(name) - 50);
|
|
149
|
+
return (i > 0 ? "..." : "") + t.slice(i, i + max - 6) + "...";
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Extract lead candidates from seed passages (most relevant first), ranked by weighted frequency.
|
|
154
|
+
* `searched` = lowercased keywords + variants (already searched; not leads).
|
|
155
|
+
*/
|
|
156
|
+
export function extractLeads(
|
|
157
|
+
seeds: SeedPassage[],
|
|
158
|
+
searched: Set<string>,
|
|
159
|
+
questionWords: Set<string>,
|
|
160
|
+
max: number,
|
|
161
|
+
): LeadCandidate[] {
|
|
162
|
+
const defined = new Set<string>();
|
|
163
|
+
for (const s of seeds) for (const n of definedNames(s.lines.join("\n"))) defined.add(n);
|
|
164
|
+
const byName = new Map<string, LeadCandidate>();
|
|
165
|
+
for (const s of seeds) {
|
|
166
|
+
const inThis = new Set<string>();
|
|
167
|
+
const prose = /\.(md|mdx|txt|rst)$/i.test(s.path);
|
|
168
|
+
const sql = /\.sql$/i.test(s.path);
|
|
169
|
+
s.lines.forEach((line, i) => {
|
|
170
|
+
if (!prose && /^\s*(\/\/|\*|\/\*|#(?!\[)|--)/.test(line)) return; // code comments: skip
|
|
171
|
+
// prose: only identifiers inside `backticks`
|
|
172
|
+
const scan = prose ? (line.match(/`[^`]+`/g) ?? []).join(" ") : line;
|
|
173
|
+
if (!scan) return;
|
|
174
|
+
for (const { name, kind } of identifiersInLine(scan)) {
|
|
175
|
+
if (name.length < 4 || CONTROL.has(name) || LEAD_STOPLIST.has(name) || defined.has(name))
|
|
176
|
+
continue;
|
|
177
|
+
if (sql && /^[A-Z]+$/.test(name)) continue; // SQL keywords followed by "(" (CHECK, EXISTS, VALUES) are not calls
|
|
178
|
+
if (searched.has(name.toLowerCase()) || searched.has(splitWords(name).join(" "))) continue;
|
|
179
|
+
let c = byName.get(name);
|
|
180
|
+
if (!c) {
|
|
181
|
+
c = {
|
|
182
|
+
name,
|
|
183
|
+
weight: 0,
|
|
184
|
+
occurrences: 0,
|
|
185
|
+
called: false,
|
|
186
|
+
seenAt: { path: s.path, line: s.start + i },
|
|
187
|
+
context: trimContext(line, name),
|
|
188
|
+
kinds: [],
|
|
189
|
+
};
|
|
190
|
+
byName.set(name, c);
|
|
191
|
+
}
|
|
192
|
+
c.occurrences++;
|
|
193
|
+
if (kind === "call") c.called = true;
|
|
194
|
+
if (!c.kinds.includes(kind)) c.kinds.push(kind);
|
|
195
|
+
if (!inThis.has(name)) {
|
|
196
|
+
c.weight += s.rel;
|
|
197
|
+
inThis.add(name);
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
const score = (c: LeadCandidate) => {
|
|
203
|
+
const words = splitWords(c.name);
|
|
204
|
+
const ov = words.filter((w) => questionWords.has(w)).length / Math.max(1, words.length);
|
|
205
|
+
return c.weight + 0.1 * Math.min(5, c.occurrences - 1) + (c.called ? 0.2 : 0) + ov;
|
|
206
|
+
};
|
|
207
|
+
return [...byName.values()]
|
|
208
|
+
.sort((a, b) => score(b) - score(a) || (a.name < b.name ? -1 : 1))
|
|
209
|
+
.slice(0, max)
|
|
210
|
+
.map((c) => ({ ...c, weight: Math.round(score(c) * 1000) / 1000 }));
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
export interface DefinitionHit {
|
|
214
|
+
name: string;
|
|
215
|
+
path: string;
|
|
216
|
+
line: number;
|
|
217
|
+
kind: "decl" | "sqlfn" | "method" | "key";
|
|
218
|
+
text: string;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/** The JS regexes used to classify an rg match line for one name (priority order). */
|
|
222
|
+
export function definitionKinds(name: string): Array<{ kind: DefinitionHit["kind"]; re: RegExp }> {
|
|
223
|
+
const n = escapeRegex(name);
|
|
224
|
+
return [
|
|
225
|
+
{
|
|
226
|
+
kind: "decl",
|
|
227
|
+
re: new RegExp(
|
|
228
|
+
`\\b(?:function\\*?|const|let|var|class|interface|type|enum|namespace|struct|trait|fn|def|func|mod)\\s+${n}\\b`,
|
|
229
|
+
),
|
|
230
|
+
},
|
|
231
|
+
{ kind: "sqlfn", re: new RegExp(`\\bfunction\\s+(?:[\\w"]+\\.)?"?${n}"?\\s*\\(`, "i") },
|
|
232
|
+
{
|
|
233
|
+
kind: "method",
|
|
234
|
+
re: new RegExp(
|
|
235
|
+
`^\\s*(?:(?:public|private|protected|static|async|readonly|override|get|set)\\s+)*${n}\\s*(?:<[^>()]*>)?\\s*\\([^;]*(?:\\{|\\(|,)\\s*$`,
|
|
236
|
+
),
|
|
237
|
+
},
|
|
238
|
+
{ kind: "key", re: new RegExp(`^\\s*["']?${n}["']?\\??\\s*[:=]`) },
|
|
239
|
+
];
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/**
|
|
243
|
+
* rg pattern (ASCII mode) for definition-shaped lines of any of the names: declarations, SQL functions,
|
|
244
|
+
* methods and object keys / fields / config keys. Classified per name in JS with definitionKinds().
|
|
245
|
+
*/
|
|
246
|
+
export function definitionPattern(names: readonly string[]): string {
|
|
247
|
+
const alt = names.map(escapeRegex).join("|");
|
|
248
|
+
return (
|
|
249
|
+
"(?-u:" +
|
|
250
|
+
[
|
|
251
|
+
`\\b(?:function\\*?|const|let|var|class|interface|type|enum|namespace|struct|trait|fn|def|func|mod)\\s+(?:${alt})\\b`,
|
|
252
|
+
`(?i:function)\\s+(?:[\\w"]+\\.)?"?(?:${alt})"?\\s*\\(`,
|
|
253
|
+
`^\\s*(?:(?:public|private|protected|static|async|readonly|override|get|set)\\s+)*(?:${alt})\\s*(?:<[^>()]*>)?\\s*\\(`,
|
|
254
|
+
`^\\s*["']?(?:${alt})["']?\\??\\s*[:=]`,
|
|
255
|
+
].join("|") +
|
|
256
|
+
")"
|
|
257
|
+
);
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* definitionPattern over consecutive groups of names, each pattern at most maxChars. A name whose own pattern
|
|
262
|
+
* is longer is left out: it gets no definition, as after a failed search.
|
|
263
|
+
*/
|
|
264
|
+
export function definitionPatterns(
|
|
265
|
+
names: readonly string[],
|
|
266
|
+
maxChars = CODE_SEARCH_MAX_PATTERN_CHARS,
|
|
267
|
+
): string[] {
|
|
268
|
+
const out: string[] = [];
|
|
269
|
+
let group: string[] = [];
|
|
270
|
+
for (const name of names) {
|
|
271
|
+
if (definitionPattern([name]).length > maxChars) continue;
|
|
272
|
+
if (group.length && definitionPattern([...group, name]).length > maxChars) {
|
|
273
|
+
out.push(definitionPattern(group));
|
|
274
|
+
group = [];
|
|
275
|
+
}
|
|
276
|
+
group.push(name);
|
|
277
|
+
}
|
|
278
|
+
if (group.length) out.push(definitionPattern(group));
|
|
279
|
+
return out;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
const KIND_RANK: Record<DefinitionHit["kind"], number> = { decl: 3, sqlfn: 3, method: 2, key: 1 };
|
|
283
|
+
|
|
284
|
+
/** Package root of a path: the first two segments for apps/ and packages/ (apps/worker, packages/runtime), else the first. */
|
|
285
|
+
export function packageOf(path: string): string {
|
|
286
|
+
const parts = path.split("/");
|
|
287
|
+
return /^(apps|packages|services|libs|crates)$/.test(parts[0] ?? "") && parts.length > 2
|
|
288
|
+
? parts.slice(0, 2).join("/")
|
|
289
|
+
: (parts[0] ?? "");
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Choose the best definitions per name. A name can be defined in many places (`search`, `isRecord`), so locality
|
|
294
|
+
* dominates: the file where the lead was seen, then its package; then kind rank (declaration > method > key),
|
|
295
|
+
* non-test, an already-selected file, a config-ish path for keys, path, line.
|
|
296
|
+
*/
|
|
297
|
+
export function chooseDefinitions(
|
|
298
|
+
hits: DefinitionHit[],
|
|
299
|
+
preferPaths: Set<string>,
|
|
300
|
+
perName: number,
|
|
301
|
+
allowTests = false,
|
|
302
|
+
seenAt: Map<string, string> = new Map(),
|
|
303
|
+
): Map<string, DefinitionHit[]> {
|
|
304
|
+
const by = new Map<string, DefinitionHit[]>();
|
|
305
|
+
for (const h of hits) {
|
|
306
|
+
if (!allowTests && isTestPath(h.path)) continue;
|
|
307
|
+
const arr = by.get(h.name) ?? [];
|
|
308
|
+
arr.push(h);
|
|
309
|
+
by.set(h.name, arr);
|
|
310
|
+
}
|
|
311
|
+
const rank = (h: DefinitionHit) =>
|
|
312
|
+
(seenAt.get(h.name) === h.path ? 100 : 0) +
|
|
313
|
+
(seenAt.has(h.name) && packageOf(seenAt.get(h.name)!) === packageOf(h.path) ? 50 : 0) +
|
|
314
|
+
KIND_RANK[h.kind] * 10 +
|
|
315
|
+
(isTestPath(h.path) ? 0 : 4) +
|
|
316
|
+
(preferPaths.has(h.path) ? 2 : 0) +
|
|
317
|
+
(h.kind === "key" && /config|setting|schema|env/i.test(h.path) ? 1 : 0);
|
|
318
|
+
const out = new Map<string, DefinitionHit[]>();
|
|
319
|
+
for (const [name, arr] of by) {
|
|
320
|
+
arr.sort(
|
|
321
|
+
(a, b) => rank(b) - rank(a) || (a.path < b.path ? -1 : a.path > b.path ? 1 : a.line - b.line),
|
|
322
|
+
);
|
|
323
|
+
out.set(name, arr.slice(0, perName));
|
|
324
|
+
}
|
|
325
|
+
return out;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
export interface DefinitionSearch {
|
|
329
|
+
hits: DefinitionHit[];
|
|
330
|
+
/** name -> number of repository files that mention it as a word (for an IDF-style genericity penalty) */
|
|
331
|
+
fileCounts: Map<string, number>;
|
|
332
|
+
ms: number;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/** Files a definition search never needs: prose/data (definitions of code identifiers live in code or SQL). */
|
|
336
|
+
export const DEF_SCAN_EXCLUDES = ["!*.{md,mdx,txt,rst,json,jsonc,html,csv,svg,snap,xml}"];
|
|
337
|
+
/** Test files, excluded from the definition scan unless the question is about tests. */
|
|
338
|
+
export const TEST_EXCLUDES = [
|
|
339
|
+
"!**/test/**",
|
|
340
|
+
"!**/tests/**",
|
|
341
|
+
"!**/__tests__/**",
|
|
342
|
+
"!**/e2e/**",
|
|
343
|
+
"!**/fixtures/**",
|
|
344
|
+
"!*.test.*",
|
|
345
|
+
"!*.spec.*",
|
|
346
|
+
];
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* ONE rg pass (ASCII-mode regex, full lines) for definition-shaped lines of all names over code and SQL files
|
|
350
|
+
* (tests only when the question is about tests), classified per name in JS: declarations/SQL functions/methods
|
|
351
|
+
* first; object keys / fields / config keys (`name:` / `name =`) only for names without a stronger definition,
|
|
352
|
+
* and a name whose only definitions are more than maxKeyOnlyDefs keys is a generic field (sessionId, accountId)
|
|
353
|
+
* and gets none. fileCounts = files with any definition-shaped line for the name (generic fields and helpers
|
|
354
|
+
* redefined everywhere score high), used as a genericity penalty. About 0.4 CPU-s on a 6k-file repository; a
|
|
355
|
+
* second pass that counted every mention cost another ~0.7 CPU-s and full-line mention output was 80k lines.
|
|
356
|
+
* Names that do not fit one pattern under the cap are searched in a few passes, merged line by line.
|
|
357
|
+
* A failed definition search yields no hits (the leads are then dropped), as in scout.
|
|
358
|
+
*/
|
|
359
|
+
export async function locateDefinitions(
|
|
360
|
+
session: WorkspaceSession,
|
|
361
|
+
names: string[],
|
|
362
|
+
cfg: CodeSearchConfig,
|
|
363
|
+
excludeArgs: string[],
|
|
364
|
+
maxKeyOnlyDefs = 12,
|
|
365
|
+
allowTests = false,
|
|
366
|
+
): Promise<DefinitionSearch> {
|
|
367
|
+
const t0 = performance.now();
|
|
368
|
+
if (!names.length) return { hits: [], fileCounts: new Map(), ms: 0 };
|
|
369
|
+
const extra = [...DEF_SCAN_EXCLUDES, ...(allowTests ? [] : TEST_EXCLUDES)].flatMap((g) => [
|
|
370
|
+
"-g",
|
|
371
|
+
g,
|
|
372
|
+
]);
|
|
373
|
+
const defOuts = await mapLimit(definitionPatterns(names), RIPGREP_SPLIT_CONCURRENCY, (pattern) =>
|
|
374
|
+
session.ripgrep(
|
|
375
|
+
[
|
|
376
|
+
"--null",
|
|
377
|
+
"--line-number",
|
|
378
|
+
"--with-filename",
|
|
379
|
+
"--no-heading",
|
|
380
|
+
"--color",
|
|
381
|
+
"never",
|
|
382
|
+
"--no-require-git",
|
|
383
|
+
"--max-columns",
|
|
384
|
+
String(cfg.recall.maxLineColumns),
|
|
385
|
+
"--max-filesize",
|
|
386
|
+
String(cfg.recall.maxFileBytes),
|
|
387
|
+
...excludeArgs,
|
|
388
|
+
...extra,
|
|
389
|
+
"-e",
|
|
390
|
+
pattern,
|
|
391
|
+
"--",
|
|
392
|
+
".",
|
|
393
|
+
],
|
|
394
|
+
{ allowFailure: true },
|
|
395
|
+
),
|
|
396
|
+
);
|
|
397
|
+
const kinds = new Map(names.map((n) => [n, definitionKinds(n)]));
|
|
398
|
+
const nameSet = new Set(names);
|
|
399
|
+
const files = new Map<string, Set<string>>();
|
|
400
|
+
const strong: DefinitionHit[] = [];
|
|
401
|
+
const keys: DefinitionHit[] = [];
|
|
402
|
+
const seenRows = new Set<string>();
|
|
403
|
+
for (const row of defOuts.flatMap((out) => out.split("\n"))) {
|
|
404
|
+
const z = row.indexOf("\0");
|
|
405
|
+
if (z <= 0) continue;
|
|
406
|
+
const colon = row.indexOf(":", z + 1);
|
|
407
|
+
if (colon < 0) continue;
|
|
408
|
+
// a line matched by two of the split patterns is classified once, as with one pattern
|
|
409
|
+
const at = row.slice(0, colon);
|
|
410
|
+
if (seenRows.has(at)) continue;
|
|
411
|
+
seenRows.add(at);
|
|
412
|
+
const text = row.slice(colon + 1);
|
|
413
|
+
if (text.length > 400 || text.startsWith("[Omitted long line")) continue; // a definition line is short
|
|
414
|
+
const path = row.slice(0, z).replace(/^\.\//, "");
|
|
415
|
+
const line = Number(row.slice(z + 1, colon));
|
|
416
|
+
const present = new Set((text.match(/[A-Za-z_$][\w$]*/g) ?? []).filter((t) => nameSet.has(t)));
|
|
417
|
+
for (const name of present) {
|
|
418
|
+
for (const k of kinds.get(name)!) {
|
|
419
|
+
if (!k.re.test(text)) continue;
|
|
420
|
+
(k.kind === "key" ? keys : strong).push({
|
|
421
|
+
name,
|
|
422
|
+
path,
|
|
423
|
+
line,
|
|
424
|
+
kind: k.kind,
|
|
425
|
+
text: text.trim().slice(0, 200),
|
|
426
|
+
});
|
|
427
|
+
let fs = files.get(name);
|
|
428
|
+
if (!fs) files.set(name, (fs = new Set()));
|
|
429
|
+
fs.add(path);
|
|
430
|
+
break;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
const hasStrong = new Set(strong.map((h) => h.name));
|
|
435
|
+
const keyHits = keys.filter((h) => !hasStrong.has(h.name));
|
|
436
|
+
const keyCount = new Map<string, number>();
|
|
437
|
+
for (const h of keyHits) keyCount.set(h.name, (keyCount.get(h.name) ?? 0) + 1);
|
|
438
|
+
const hits = [...strong, ...keyHits.filter((h) => (keyCount.get(h.name) ?? 0) <= maxKeyOnlyDefs)];
|
|
439
|
+
hits.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : a.line - b.line));
|
|
440
|
+
const fileCounts = new Map(names.map((n) => [n, files.get(n)?.size ?? 0]));
|
|
441
|
+
return { hits, fileCounts, ms: Math.round(performance.now() - t0) };
|
|
442
|
+
}
|