@opengeni/jev 0.1.0-canary.36199476632001
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +190 -0
- package/README.md +140 -0
- package/dist/circuit-breaker.d.ts +58 -0
- package/dist/client.d.ts +173 -0
- package/dist/code-search/config.d.ts +142 -0
- package/dist/code-search/judge.d.ts +169 -0
- package/dist/code-search/leads.d.ts +96 -0
- package/dist/code-search/pack.d.ts +106 -0
- package/dist/code-search/recall.d.ts +155 -0
- package/dist/code-search/search.d.ts +93 -0
- package/dist/code-search/session.d.ts +33 -0
- package/dist/code-search/text.d.ts +35 -0
- package/dist/code-search/tool.d.ts +34 -0
- package/dist/code-search/windows.d.ts +85 -0
- package/dist/code-search/workspace.d.ts +51 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +3110 -0
- package/dist/index.js.map +1 -0
- package/package.json +39 -0
- package/src/circuit-breaker.ts +135 -0
- package/src/client.ts +577 -0
- package/src/code-search/config.ts +282 -0
- package/src/code-search/judge.ts +413 -0
- package/src/code-search/leads.ts +442 -0
- package/src/code-search/pack.ts +354 -0
- package/src/code-search/recall.ts +648 -0
- package/src/code-search/search.ts +773 -0
- package/src/code-search/session.ts +89 -0
- package/src/code-search/text.ts +159 -0
- package/src/code-search/tool.ts +209 -0
- package/src/code-search/windows.ts +617 -0
- package/src/code-search/workspace.ts +55 -0
- package/src/index.ts +71 -0
|
@@ -0,0 +1,617 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* windows.ts - turn hit lines into readable passages.
|
|
3
|
+
*
|
|
4
|
+
* For each hit: walk up (<= maxUp lines) to the nearest declaration line at lower-or-equal indentation
|
|
5
|
+
* whose block encloses the hit (function/class/const/export/interface/type/enum/describe/route handler,
|
|
6
|
+
* markdown heading, SQL statement, python def/class), then walk down to where that block closes
|
|
7
|
+
* (<= maxDown lines below the hit). No enclosing declaration -> fixed context around the hit, labelled
|
|
8
|
+
* with the nearest enclosing declaration further up. Overlapping windows merge; windows longer than
|
|
9
|
+
* maxWindowLines split around hit clusters; each file keeps its best windowsPerFile windows.
|
|
10
|
+
* All line numbers in this module's public API are 1-based. A line past the end of the file as read (it
|
|
11
|
+
* changed after ripgrep saw it, or could not be read) is skipped by the callers and treated as blank here.
|
|
12
|
+
*/
|
|
13
|
+
import type { CodeSearchConfig } from "./config";
|
|
14
|
+
import type { HitLine, KeywordInfo } from "./recall";
|
|
15
|
+
|
|
16
|
+
export type Lang = "brace" | "indent" | "markdown" | "sql" | "other";
|
|
17
|
+
|
|
18
|
+
export interface Window {
|
|
19
|
+
start: number;
|
|
20
|
+
end: number;
|
|
21
|
+
/** Hit lines inside the window (1-based). */
|
|
22
|
+
hits: number[];
|
|
23
|
+
/** Enclosing declaration when it is above the window start (or the window starts at it). */
|
|
24
|
+
label?: { line: number; text: string } | undefined;
|
|
25
|
+
kind: "hit" | "header" | "def";
|
|
26
|
+
score: number;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function langOf(path: string): Lang {
|
|
30
|
+
const ext = path.toLowerCase().split(".").pop() ?? "";
|
|
31
|
+
if (
|
|
32
|
+
[
|
|
33
|
+
"ts",
|
|
34
|
+
"tsx",
|
|
35
|
+
"js",
|
|
36
|
+
"jsx",
|
|
37
|
+
"mjs",
|
|
38
|
+
"cjs",
|
|
39
|
+
"mts",
|
|
40
|
+
"cts",
|
|
41
|
+
"rs",
|
|
42
|
+
"go",
|
|
43
|
+
"java",
|
|
44
|
+
"c",
|
|
45
|
+
"h",
|
|
46
|
+
"cc",
|
|
47
|
+
"cpp",
|
|
48
|
+
"hpp",
|
|
49
|
+
"cs",
|
|
50
|
+
"swift",
|
|
51
|
+
"kt",
|
|
52
|
+
"scala",
|
|
53
|
+
"css",
|
|
54
|
+
"scss",
|
|
55
|
+
"json",
|
|
56
|
+
"jsonc",
|
|
57
|
+
"proto",
|
|
58
|
+
"tf",
|
|
59
|
+
"hcl",
|
|
60
|
+
].includes(ext)
|
|
61
|
+
) {
|
|
62
|
+
return "brace";
|
|
63
|
+
}
|
|
64
|
+
if (["py", "yaml", "yml"].includes(ext)) return "indent";
|
|
65
|
+
if (["md", "mdx"].includes(ext)) return "markdown";
|
|
66
|
+
if (ext === "sql") return "sql";
|
|
67
|
+
return "other";
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export function indentOf(line: string): number {
|
|
71
|
+
let n = 0;
|
|
72
|
+
for (const ch of line) {
|
|
73
|
+
if (ch === " ") n += 1;
|
|
74
|
+
else if (ch === "\t") n += 4;
|
|
75
|
+
else break;
|
|
76
|
+
}
|
|
77
|
+
return n;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
const CONTROL = new Set([
|
|
81
|
+
"if",
|
|
82
|
+
"for",
|
|
83
|
+
"while",
|
|
84
|
+
"switch",
|
|
85
|
+
"catch",
|
|
86
|
+
"return",
|
|
87
|
+
"else",
|
|
88
|
+
"do",
|
|
89
|
+
"try",
|
|
90
|
+
"with",
|
|
91
|
+
"await",
|
|
92
|
+
"new",
|
|
93
|
+
"typeof",
|
|
94
|
+
"function",
|
|
95
|
+
"throw",
|
|
96
|
+
"yield",
|
|
97
|
+
"delete",
|
|
98
|
+
"void",
|
|
99
|
+
"in",
|
|
100
|
+
"of",
|
|
101
|
+
"case",
|
|
102
|
+
"super",
|
|
103
|
+
"this",
|
|
104
|
+
"import",
|
|
105
|
+
"export",
|
|
106
|
+
]);
|
|
107
|
+
|
|
108
|
+
const TS_DECL =
|
|
109
|
+
/^\s*(?:export\s+)?(?:default\s+)?(?:declare\s+)?(?:abstract\s+)?(?:async\s+)?(?:function\b|class\s|interface\s|type\s+[A-Za-z_$][\w$]*\s*(?:<[^>]*>)?\s*=|enum\s|const\s+[A-Za-z_$[{]|let\s+[A-Za-z_$[{]|var\s+[A-Za-z_$]|namespace\s|module\s)/;
|
|
110
|
+
const EXPORT_DEFAULT = /^\s*export\s+default\b/;
|
|
111
|
+
const RUST_DECL =
|
|
112
|
+
/^\s*(?:pub(?:\([^)]*\))?\s+)?(?:async\s+)?(?:unsafe\s+)?(?:const\s+)?(?:fn|struct|enum|trait|impl|mod|macro_rules!)[\s<]/;
|
|
113
|
+
const GO_DECL = /^\s*func\s/;
|
|
114
|
+
const TEST_DECL = /^\s*(?:describe|it|test)(?:\.\w+)?\s*\(/;
|
|
115
|
+
const METHOD =
|
|
116
|
+
/^\s*(?:(?:public|private|protected|static|readonly|async|override|get|set)\s+)*\*?\s*([A-Za-z_$][\w$]*)\s*(?:<[^>()]*>)?\s*\([^;]*$/;
|
|
117
|
+
const PROP_FN =
|
|
118
|
+
/^\s*(?:(?:public|private|protected|static|readonly)\s+)*[A-Za-z_$][\w$]*\s*[:=]\s*(?:async\s*)?(?:function\b|\([^)]*\)?[^;]*=>|[A-Za-z_$][\w$]*\s*=>)/;
|
|
119
|
+
const OBJ_KEY_OPEN = /^\s*["']?[A-Za-z_$][\w$-]*["']?\s*:\s*(?:[\w$.]+\()?[{[]\s*$/;
|
|
120
|
+
const CALLBACK_OPEN = /^\s*(?:await\s+)?[\w$]+(?:\.[\w$]+)+\s*\(.*(?:=>|function\b).*\{\s*$/;
|
|
121
|
+
const MD_HEADING = /^(#{1,6})\s/;
|
|
122
|
+
const SQL_STMT =
|
|
123
|
+
/^(?:CREATE|ALTER|DROP|INSERT|UPDATE|DELETE|WITH|SELECT|DO|COMMENT|GRANT|REVOKE|BEGIN|SET|LOCK|TRUNCATE)\b/i;
|
|
124
|
+
const PY_DECL = /^\s*(?:async\s+)?(?:def|class)\s/;
|
|
125
|
+
const YAML_KEY = /^\s*(?:- )?["']?[A-Za-z_$][\w$ .-]*["']?\s*:(?:\s|$)/;
|
|
126
|
+
|
|
127
|
+
export function isDeclLine(line: string, lang: Lang): boolean {
|
|
128
|
+
if (!line.trim()) return false;
|
|
129
|
+
switch (lang) {
|
|
130
|
+
case "markdown":
|
|
131
|
+
return MD_HEADING.test(line);
|
|
132
|
+
case "sql":
|
|
133
|
+
return indentOf(line) === 0 && SQL_STMT.test(line);
|
|
134
|
+
case "indent":
|
|
135
|
+
return PY_DECL.test(line) || YAML_KEY.test(line);
|
|
136
|
+
case "brace": {
|
|
137
|
+
if (
|
|
138
|
+
TS_DECL.test(line) ||
|
|
139
|
+
EXPORT_DEFAULT.test(line) ||
|
|
140
|
+
RUST_DECL.test(line) ||
|
|
141
|
+
GO_DECL.test(line) ||
|
|
142
|
+
TEST_DECL.test(line)
|
|
143
|
+
)
|
|
144
|
+
return true;
|
|
145
|
+
if (PROP_FN.test(line) || OBJ_KEY_OPEN.test(line) || CALLBACK_OPEN.test(line)) return true;
|
|
146
|
+
const m = METHOD.exec(line);
|
|
147
|
+
if (
|
|
148
|
+
m &&
|
|
149
|
+
!CONTROL.has(m[1]!) &&
|
|
150
|
+
/(\{|\(|,)\s*$/.test(line.trimEnd()) &&
|
|
151
|
+
!/^\s*[\w$]+\s*\(.*\)\s*;?\s*$/.test(line)
|
|
152
|
+
) {
|
|
153
|
+
return true;
|
|
154
|
+
}
|
|
155
|
+
return false;
|
|
156
|
+
}
|
|
157
|
+
default:
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/** Remove string literals and comments so bracket counting is roughly right. */
|
|
163
|
+
export function stripForBrackets(line: string): string {
|
|
164
|
+
const t = line.trim();
|
|
165
|
+
if (t.startsWith("*") || t.startsWith("/*") || t.startsWith("//")) return "";
|
|
166
|
+
return line
|
|
167
|
+
.replace(/"(?:[^"\\]|\\.)*"/g, '""')
|
|
168
|
+
.replace(/'(?:[^'\\]|\\.)*'/g, "''")
|
|
169
|
+
.replace(/`(?:[^`\\]|\\.)*`/g, "``")
|
|
170
|
+
.replace(/\/\*.*?\*\//g, "")
|
|
171
|
+
.replace(/\/\/.*$/, "");
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
const blockEndCache = new WeakMap<string[], Map<number, number | null>>();
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* 0-based index of the last line of the block starting at 0-based line d, or null if unknown
|
|
178
|
+
* (no structure, or longer than maxScan lines).
|
|
179
|
+
*/
|
|
180
|
+
export function blockEnd(lines: string[], d: number, lang: Lang, maxScan = 3000): number | null {
|
|
181
|
+
let cache = blockEndCache.get(lines);
|
|
182
|
+
if (!cache) {
|
|
183
|
+
cache = new Map();
|
|
184
|
+
blockEndCache.set(lines, cache);
|
|
185
|
+
}
|
|
186
|
+
const key = d * 8 + ["brace", "indent", "markdown", "sql", "other"].indexOf(lang);
|
|
187
|
+
if (cache.has(key)) return cache.get(key)!;
|
|
188
|
+
const r = computeBlockEnd(lines, d, lang, maxScan);
|
|
189
|
+
cache.set(key, r);
|
|
190
|
+
return r;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
function computeBlockEnd(lines: string[], d: number, lang: Lang, maxScan: number): number | null {
|
|
194
|
+
const n = lines.length;
|
|
195
|
+
if (d < 0 || d >= n) return null;
|
|
196
|
+
const last = Math.min(n - 1, d + maxScan);
|
|
197
|
+
if (lang === "brace") {
|
|
198
|
+
let depth = 0;
|
|
199
|
+
let maxDepth = 0;
|
|
200
|
+
for (let i = d; i <= last; i++) {
|
|
201
|
+
const s = stripForBrackets(lines[i]!);
|
|
202
|
+
for (const ch of s) {
|
|
203
|
+
if (ch === "{" || ch === "(" || ch === "[") {
|
|
204
|
+
depth++;
|
|
205
|
+
if (depth > maxDepth) maxDepth = depth;
|
|
206
|
+
} else if (ch === "}" || ch === ")" || ch === "]") depth--;
|
|
207
|
+
}
|
|
208
|
+
if (maxDepth > 0 && depth <= 0) return i;
|
|
209
|
+
if (maxDepth === 0 && /;\s*$/.test(s)) return i;
|
|
210
|
+
}
|
|
211
|
+
return null;
|
|
212
|
+
}
|
|
213
|
+
if (lang === "indent") {
|
|
214
|
+
const ind = indentOf(lines[d]!);
|
|
215
|
+
let lastNonEmpty = d;
|
|
216
|
+
for (let i = d + 1; i <= last; i++) {
|
|
217
|
+
const l = lines[i]!;
|
|
218
|
+
if (!l.trim()) continue;
|
|
219
|
+
if (indentOf(l) <= ind) return lastNonEmpty;
|
|
220
|
+
lastNonEmpty = i;
|
|
221
|
+
}
|
|
222
|
+
return last === n - 1 ? lastNonEmpty : null;
|
|
223
|
+
}
|
|
224
|
+
if (lang === "markdown") {
|
|
225
|
+
const m = MD_HEADING.exec(lines[d]!);
|
|
226
|
+
if (!m) return null;
|
|
227
|
+
const level = m[1]!.length;
|
|
228
|
+
let inFence = false;
|
|
229
|
+
for (let i = d + 1; i <= last; i++) {
|
|
230
|
+
const l = lines[i]!;
|
|
231
|
+
if (/^\s*(```|~~~)/.test(l)) inFence = !inFence;
|
|
232
|
+
if (inFence) continue;
|
|
233
|
+
const h = MD_HEADING.exec(l);
|
|
234
|
+
if (h && h[1]!.length <= level) return i - 1;
|
|
235
|
+
}
|
|
236
|
+
return last === n - 1 ? n - 1 : null;
|
|
237
|
+
}
|
|
238
|
+
if (lang === "sql") {
|
|
239
|
+
let inDollar = false;
|
|
240
|
+
for (let i = d; i <= last; i++) {
|
|
241
|
+
const l = lines[i]!.replace(/--.*$/, "");
|
|
242
|
+
const toggles = (l.match(/\$[A-Za-z_]*\$/g) ?? []).length;
|
|
243
|
+
if (toggles % 2 === 1) inDollar = !inDollar;
|
|
244
|
+
if (!inDollar && /;\s*$/.test(l)) return i;
|
|
245
|
+
}
|
|
246
|
+
return null;
|
|
247
|
+
}
|
|
248
|
+
return null;
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
function isCommentOrDecorator(line: string): boolean {
|
|
252
|
+
return /^(\/\/|\/\*\*?|\*|#\[|@[A-Za-z]|--)/.test(line.trim());
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/** Include up to maxLines of doc comments / decorators directly above a declaration (0-based). */
|
|
256
|
+
function leadingComments(lines: string[], d: number, lang: Lang, maxLines = 6): number {
|
|
257
|
+
if (lang === "markdown" || lang === "other") return d;
|
|
258
|
+
let s = d;
|
|
259
|
+
while (s > 0 && d - s < maxLines && isCommentOrDecorator(lines[s - 1] ?? "")) s--;
|
|
260
|
+
return s;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/** Nearest declaration above (0-based idx) with indentation below the given one; cheap, for labels only. */
|
|
264
|
+
export function nearestEnclosingLabel(
|
|
265
|
+
lines: string[],
|
|
266
|
+
idx: number,
|
|
267
|
+
lang: Lang,
|
|
268
|
+
maxScan = 2000,
|
|
269
|
+
): { line: number; text: string } | undefined {
|
|
270
|
+
const ind = indentOf(lines[idx] ?? "");
|
|
271
|
+
for (let i = idx; i >= Math.max(0, idx - maxScan); i--) {
|
|
272
|
+
const l = lines[i] ?? "";
|
|
273
|
+
if (!isDeclLine(l, lang)) continue;
|
|
274
|
+
if (
|
|
275
|
+
lang === "markdown" ||
|
|
276
|
+
lang === "sql" ||
|
|
277
|
+
i === idx ||
|
|
278
|
+
indentOf(l) < ind ||
|
|
279
|
+
(ind === 0 && indentOf(l) === 0)
|
|
280
|
+
) {
|
|
281
|
+
return { line: i + 1, text: l.trim().slice(0, 140) };
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
return undefined;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/** The window around one hit line (1-based in, 1-based out). */
|
|
288
|
+
export function enclosingWindow(
|
|
289
|
+
lines: string[],
|
|
290
|
+
hitLine: number,
|
|
291
|
+
lang: Lang,
|
|
292
|
+
cfg: CodeSearchConfig,
|
|
293
|
+
): Window {
|
|
294
|
+
const w = cfg.wave2;
|
|
295
|
+
const hi = hitLine - 1;
|
|
296
|
+
const n = lines.length;
|
|
297
|
+
const hitIndent = indentOf(lines[hi] ?? "");
|
|
298
|
+
if (lang !== "other") {
|
|
299
|
+
for (let i = hi; i >= Math.max(0, hi - w.maxUp); i--) {
|
|
300
|
+
const l = lines[i] ?? "";
|
|
301
|
+
if (!isDeclLine(l, lang)) continue;
|
|
302
|
+
if ((lang === "brace" || lang === "indent") && i !== hi && indentOf(l) > hitIndent) continue;
|
|
303
|
+
const e = blockEnd(lines, i, lang);
|
|
304
|
+
if (e === null || e < hi) continue;
|
|
305
|
+
// a short statement starting on the hit line (`const x = f(...);`) does not enclose anything
|
|
306
|
+
if (i === hi && e - hi < 2) continue;
|
|
307
|
+
const start = leadingComments(lines, i, lang);
|
|
308
|
+
const end = Math.min(e, hi + w.maxDown, n - 1);
|
|
309
|
+
return padWindow(
|
|
310
|
+
{
|
|
311
|
+
start: start + 1,
|
|
312
|
+
end: end + 1,
|
|
313
|
+
hits: [hitLine],
|
|
314
|
+
label: { line: i + 1, text: l.trim().slice(0, 140) },
|
|
315
|
+
kind: "hit",
|
|
316
|
+
score: 0,
|
|
317
|
+
},
|
|
318
|
+
n,
|
|
319
|
+
cfg,
|
|
320
|
+
);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
const start = Math.max(0, hi - w.fallbackBefore);
|
|
324
|
+
const end = Math.min(n - 1, hi + w.fallbackAfter);
|
|
325
|
+
const label = lang === "other" ? undefined : nearestEnclosingLabel(lines, hi, lang);
|
|
326
|
+
return { start: start + 1, end: end + 1, hits: [hitLine], label, kind: "hit", score: 0 };
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
/** Grow windows shorter than minWindowLines with surrounding context (1/3 above, 2/3 below). */
|
|
330
|
+
export function padWindow(win: Window, nLines: number, cfg: CodeSearchConfig): Window {
|
|
331
|
+
const min = cfg.wave2.minWindowLines;
|
|
332
|
+
const len = win.end - win.start + 1;
|
|
333
|
+
if (len >= min) return win;
|
|
334
|
+
const need = min - len;
|
|
335
|
+
const up = Math.floor(need / 3);
|
|
336
|
+
let start = Math.max(1, win.start - up);
|
|
337
|
+
let end = Math.min(nLines, win.end + (need - (win.start - start)));
|
|
338
|
+
if (end - start + 1 < min) start = Math.max(1, end - min + 1);
|
|
339
|
+
return { ...win, start, end };
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
/** Merge overlapping windows or windows separated by <= gap lines. Input any order; output by start. */
|
|
343
|
+
export function mergeWindows(ws: Window[], gap: number): Window[] {
|
|
344
|
+
const sorted = [...ws].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
345
|
+
const out: Window[] = [];
|
|
346
|
+
for (const w of sorted) {
|
|
347
|
+
const prev = out[out.length - 1];
|
|
348
|
+
if (prev && w.start <= prev.end + gap + 1) {
|
|
349
|
+
prev.end = Math.max(prev.end, w.end);
|
|
350
|
+
prev.hits = [...new Set([...prev.hits, ...w.hits])].sort((a, b) => a - b);
|
|
351
|
+
// keep the outermost label (earliest declaration)
|
|
352
|
+
if (w.label && (!prev.label || w.label.line < prev.label.line)) prev.label = w.label;
|
|
353
|
+
if (prev.kind !== w.kind && w.kind === "def") prev.kind = "def";
|
|
354
|
+
} else {
|
|
355
|
+
out.push({ ...w, hits: [...w.hits] });
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
return out;
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
/** Split a window longer than maxLines into chunks around its hit clusters (chunks without hits are dropped). */
|
|
362
|
+
export function splitWindow(w: Window, maxLines: number, before = 20): Window[] {
|
|
363
|
+
if (w.end - w.start + 1 <= maxLines) return [w];
|
|
364
|
+
const hits = [...w.hits].sort((a, b) => a - b);
|
|
365
|
+
if (!hits.length) return [{ ...w, end: w.start + maxLines - 1 }];
|
|
366
|
+
const out: Window[] = [];
|
|
367
|
+
let i = 0;
|
|
368
|
+
while (i < hits.length) {
|
|
369
|
+
let start = Math.max(w.start, hits[i]! - before);
|
|
370
|
+
// keep the declaration line if it is close
|
|
371
|
+
if (w.start >= hits[i]! - 2 * before) start = w.start;
|
|
372
|
+
const end = Math.min(w.end, start + maxLines - 1);
|
|
373
|
+
const chunkHits: number[] = [];
|
|
374
|
+
while (i < hits.length && hits[i]! <= end) chunkHits.push(hits[i++]!);
|
|
375
|
+
if (!chunkHits.length) {
|
|
376
|
+
// hit before start cannot happen; guard against infinite loops
|
|
377
|
+
i++;
|
|
378
|
+
continue;
|
|
379
|
+
}
|
|
380
|
+
out.push({
|
|
381
|
+
...w,
|
|
382
|
+
start,
|
|
383
|
+
end,
|
|
384
|
+
hits: chunkHits,
|
|
385
|
+
label: w.label && w.label.line < start ? w.label : w.label,
|
|
386
|
+
});
|
|
387
|
+
}
|
|
388
|
+
return out;
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/** Keywords (indices) whose pattern occurs anywhere in lines[start..end] (1-based inclusive). */
|
|
392
|
+
const kwRegexCache = new WeakMap<KeywordInfo, RegExp | null>();
|
|
393
|
+
export function keywordRegex(k: KeywordInfo): RegExp | null {
|
|
394
|
+
if (!kwRegexCache.has(k)) {
|
|
395
|
+
let re: RegExp | null = null;
|
|
396
|
+
try {
|
|
397
|
+
re = new RegExp(k.pattern, "i");
|
|
398
|
+
} catch {
|
|
399
|
+
re = null;
|
|
400
|
+
}
|
|
401
|
+
kwRegexCache.set(k, re);
|
|
402
|
+
}
|
|
403
|
+
return kwRegexCache.get(k)!;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
export function textHasKeyword(text: string, k: KeywordInfo): boolean {
|
|
407
|
+
const re = keywordRegex(k);
|
|
408
|
+
if (re) return re.test(text);
|
|
409
|
+
const lower = text.toLowerCase();
|
|
410
|
+
return k.variants.some((v) => lower.includes(v.toLowerCase()));
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
export function keywordsInRange(
|
|
414
|
+
lines: string[],
|
|
415
|
+
start: number,
|
|
416
|
+
end: number,
|
|
417
|
+
keywords: KeywordInfo[],
|
|
418
|
+
): number[] {
|
|
419
|
+
const text = lines.slice(start - 1, end).join("\n");
|
|
420
|
+
return keywords.filter((k) => k.idf > 0 && textHasKeyword(text, k)).map((k) => k.index);
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
export function scoreWindow(
|
|
424
|
+
lines: string[],
|
|
425
|
+
w: Window,
|
|
426
|
+
keywords: KeywordInfo[],
|
|
427
|
+
hitWeight = 0.1,
|
|
428
|
+
): number {
|
|
429
|
+
const kws = keywordsInRange(lines, w.start, w.end, keywords);
|
|
430
|
+
return kws.reduce((s, k) => s + keywords[k]!.idf, 0) + hitWeight * Math.log(1 + w.hits.length);
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
/**
|
|
434
|
+
* Windows for one file: window the strongest hit lines (skipping hits already covered), merge,
|
|
435
|
+
* split long ones, score, keep the best windowsPerFile, return in line order.
|
|
436
|
+
*/
|
|
437
|
+
export function buildFileWindows(
|
|
438
|
+
lines: string[],
|
|
439
|
+
hitLines: HitLine[],
|
|
440
|
+
keywords: KeywordInfo[],
|
|
441
|
+
lang: Lang,
|
|
442
|
+
cfg: CodeSearchConfig,
|
|
443
|
+
render?: RenderOpts,
|
|
444
|
+
): Window[] {
|
|
445
|
+
const w = cfg.wave2;
|
|
446
|
+
const ro: RenderOpts = render ?? { maxLineChars: w.maxLineChars };
|
|
447
|
+
// prose sections are long and their paragraphs independent: smaller sub-windows around each hit
|
|
448
|
+
const maxChars = lang === "markdown" ? w.maxProseWindowChars : w.maxWindowChars;
|
|
449
|
+
if (!lines.length) return [];
|
|
450
|
+
if (!hitLines.length) {
|
|
451
|
+
// path-only match: the top of the file (or its first section)
|
|
452
|
+
const e = lang === "markdown" ? Math.min(lines.length, 80) : Math.min(lines.length, 60);
|
|
453
|
+
return splitByChars(
|
|
454
|
+
lines,
|
|
455
|
+
{ start: 1, end: e, hits: [], kind: "header", score: 0 },
|
|
456
|
+
maxChars,
|
|
457
|
+
ro,
|
|
458
|
+
).slice(0, 1);
|
|
459
|
+
}
|
|
460
|
+
// hits past the end of the file as read (it changed after ripgrep saw it) are dropped
|
|
461
|
+
const inRange = hitLines.filter((h) => h.line >= 1 && h.line <= lines.length);
|
|
462
|
+
const weight = (h: HitLine) => h.kws.reduce((s, k) => s + (keywords[k]?.idf ?? 0), 0);
|
|
463
|
+
const ranked = [...inRange].sort((a, b) => weight(b) - weight(a) || a.line - b.line);
|
|
464
|
+
const budgetHits = w.seedHitsPerFile;
|
|
465
|
+
const raw: Window[] = [];
|
|
466
|
+
let used = 0;
|
|
467
|
+
for (const h of ranked) {
|
|
468
|
+
if (used >= budgetHits) break;
|
|
469
|
+
const inside = raw.find((r) => h.line >= r.start && h.line <= r.end);
|
|
470
|
+
if (inside) {
|
|
471
|
+
inside.hits.push(h.line);
|
|
472
|
+
continue;
|
|
473
|
+
}
|
|
474
|
+
raw.push(enclosingWindow(lines, h.line, lang, cfg));
|
|
475
|
+
used++;
|
|
476
|
+
}
|
|
477
|
+
// hits that fall inside kept windows but were not windowed themselves still count
|
|
478
|
+
for (const h of inRange) {
|
|
479
|
+
for (const r of raw)
|
|
480
|
+
if (h.line >= r.start && h.line <= r.end && !r.hits.includes(h.line)) r.hits.push(h.line);
|
|
481
|
+
}
|
|
482
|
+
const merged = mergeWindows(raw, w.mergeGap)
|
|
483
|
+
.flatMap((x) => splitWindow(x, w.maxWindowLines))
|
|
484
|
+
.flatMap((x) => splitByChars(lines, x, maxChars, ro));
|
|
485
|
+
for (const x of merged) x.score = scoreWindow(lines, x, keywords, cfg.recall.hitCountWeight);
|
|
486
|
+
return merged
|
|
487
|
+
.sort((a, b) => b.score - a.score || a.start - b.start)
|
|
488
|
+
.slice(0, w.windowsPerFile)
|
|
489
|
+
.sort((a, b) => a.start - b.start);
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/** Window for a definition found at 1-based defLine (lead following). */
|
|
493
|
+
export function definitionWindow(
|
|
494
|
+
lines: string[],
|
|
495
|
+
defLine: number,
|
|
496
|
+
lang: Lang,
|
|
497
|
+
cfg: CodeSearchConfig,
|
|
498
|
+
render?: RenderOpts,
|
|
499
|
+
): Window {
|
|
500
|
+
const w = cfg.wave2;
|
|
501
|
+
const d = defLine - 1;
|
|
502
|
+
const start = leadingComments(lines, d, lang);
|
|
503
|
+
const e = lang === "other" ? null : blockEnd(lines, d, lang);
|
|
504
|
+
let end =
|
|
505
|
+
e === null ? Math.min(lines.length - 1, d + w.fallbackAfter) : Math.min(e, d + w.maxDown);
|
|
506
|
+
end = Math.min(end, start + w.maxWindowLines - 1, lines.length - 1);
|
|
507
|
+
const win: Window = {
|
|
508
|
+
start: start + 1,
|
|
509
|
+
end: end + 1,
|
|
510
|
+
hits: [defLine],
|
|
511
|
+
label: { line: defLine, text: (lines[d] ?? "").trim().slice(0, 140) },
|
|
512
|
+
kind: "def",
|
|
513
|
+
score: 0,
|
|
514
|
+
};
|
|
515
|
+
// over the char cap: keep the part starting at the definition (the first sub-window)
|
|
516
|
+
const maxChars = lang === "markdown" ? w.maxProseWindowChars : w.maxWindowChars;
|
|
517
|
+
return splitByChars(
|
|
518
|
+
lines,
|
|
519
|
+
win,
|
|
520
|
+
maxChars,
|
|
521
|
+
render ?? { maxLineChars: w.maxLineChars },
|
|
522
|
+
defLine - win.start,
|
|
523
|
+
)[0]!;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
export interface RenderOpts {
|
|
527
|
+
/** Lines longer than this are cut (with an explicit marker). */
|
|
528
|
+
maxLineChars: number;
|
|
529
|
+
/** When a cut line has a match beyond the head, the cut keeps the region around the first match. */
|
|
530
|
+
needles?: RegExp[];
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
/**
|
|
534
|
+
* Cut one over-long line to about maxChars, marked explicitly. If a needle matches beyond the first ~60% of
|
|
535
|
+
* the budget, keep a short head plus the region around the match (prose docs such as AGENTS.md have
|
|
536
|
+
* single-line paragraphs of 1-6k chars whose key sentence is often far from the start).
|
|
537
|
+
*/
|
|
538
|
+
export function cutLine(t: string, maxChars: number, needles: RegExp[] = []): string {
|
|
539
|
+
if (t.length <= maxChars) return t;
|
|
540
|
+
let at = -1;
|
|
541
|
+
for (const re of needles) {
|
|
542
|
+
const m = re.exec(t);
|
|
543
|
+
if (m && (at < 0 || m.index < at)) at = m.index;
|
|
544
|
+
}
|
|
545
|
+
if (at < 0 || at < maxChars * 0.6)
|
|
546
|
+
return `${t.slice(0, maxChars)} ...[line cut, ${t.length} chars]`;
|
|
547
|
+
const head = Math.floor(maxChars * 0.25);
|
|
548
|
+
const before = Math.floor(maxChars * 0.3);
|
|
549
|
+
const s = Math.max(head, at - before);
|
|
550
|
+
const e = Math.min(t.length, s + (maxChars - head));
|
|
551
|
+
return `${t.slice(0, head)} ...[cut]... ${t.slice(s, e)}${e < t.length ? ` ...[line cut, ${t.length} chars]` : ""}`;
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
/** Render lines[start..end] (1-based inclusive) as `N| text`, cutting very long lines with an explicit marker. */
|
|
555
|
+
export function renderLines(
|
|
556
|
+
lines: string[],
|
|
557
|
+
start: number,
|
|
558
|
+
end: number,
|
|
559
|
+
opts: number | RenderOpts,
|
|
560
|
+
): string {
|
|
561
|
+
const o: RenderOpts = typeof opts === "number" ? { maxLineChars: opts } : opts;
|
|
562
|
+
const out: string[] = [];
|
|
563
|
+
for (let i = start; i <= end && i <= lines.length; i++) {
|
|
564
|
+
out.push(`${i}| ${cutLine(lines[i - 1]!.replace(/\t/g, " "), o.maxLineChars, o.needles)}`);
|
|
565
|
+
}
|
|
566
|
+
return out.join("\n");
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
/** Rendered length of one line (as renderLines would produce it, plus the newline). */
|
|
570
|
+
export function renderedLineLength(lines: string[], i: number, o: RenderOpts): number {
|
|
571
|
+
return renderLines(lines, i, i, o).length + 1;
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
/**
|
|
575
|
+
* Split a window whose rendered text exceeds maxChars into whole-line sub-windows around its hits:
|
|
576
|
+
* each sub-window starts a few lines above the first uncovered hit (up to a quarter of the budget) and
|
|
577
|
+
* extends down until the budget is used. Windows without hits (headers, definitions) keep their top.
|
|
578
|
+
*/
|
|
579
|
+
export function splitByChars(
|
|
580
|
+
lines: string[],
|
|
581
|
+
w: Window,
|
|
582
|
+
maxChars: number,
|
|
583
|
+
o: RenderOpts,
|
|
584
|
+
ctxUp = 3,
|
|
585
|
+
): Window[] {
|
|
586
|
+
const len = (i: number) => renderedLineLength(lines, i, o);
|
|
587
|
+
let total = 0;
|
|
588
|
+
for (let i = w.start; i <= w.end; i++) total += len(i);
|
|
589
|
+
if (total <= maxChars) return [w];
|
|
590
|
+
const hits = [...w.hits].filter((h) => h >= w.start && h <= w.end).sort((a, b) => a - b);
|
|
591
|
+
const anchors = hits.length ? hits : [w.start];
|
|
592
|
+
const out: Window[] = [];
|
|
593
|
+
let covered = w.start - 1;
|
|
594
|
+
for (const h of anchors) {
|
|
595
|
+
if (h <= covered) continue;
|
|
596
|
+
let s = h;
|
|
597
|
+
let used = len(h);
|
|
598
|
+
while (
|
|
599
|
+
s - 1 >= w.start &&
|
|
600
|
+
s - 1 > covered &&
|
|
601
|
+
h - (s - 1) <= ctxUp &&
|
|
602
|
+
used + len(s - 1) <= maxChars / 4
|
|
603
|
+
)
|
|
604
|
+
used += len(--s);
|
|
605
|
+
let e = h;
|
|
606
|
+
while (e + 1 <= w.end && used + len(e + 1) <= maxChars) used += len(++e);
|
|
607
|
+
out.push({ ...w, start: s, end: e, hits: hits.filter((x) => x >= s && x <= e) });
|
|
608
|
+
covered = e;
|
|
609
|
+
}
|
|
610
|
+
return out;
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
export function splitLines(text: string): string[] {
|
|
614
|
+
const lines = text.split(/\r?\n/);
|
|
615
|
+
if (lines.length && lines[lines.length - 1] === "") lines.pop();
|
|
616
|
+
return lines;
|
|
617
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The only way the code_search engine touches files and processes. The sandbox side implements it.
|
|
3
|
+
*
|
|
4
|
+
* The engine passes ripgrep only these flags, so an adapter may enforce an allowlist:
|
|
5
|
+
* --files, --null, --line-number, --with-filename, --no-heading, --color never, -i, -w, --no-require-git,
|
|
6
|
+
* --hidden, -m N, --max-columns N, --max-filesize N, -g GLOB, -e PATTERN, then `--` followed by
|
|
7
|
+
* workspace-relative paths ("." or paths without a leading "-", no absolute paths, no "..").
|
|
8
|
+
* A PATTERN is never longer than CODE_SEARCH_MAX_PATTERN_CHARS: the engine splits a longer alternation
|
|
9
|
+
* into several ripgrep calls and merges their output.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Longest `-e` pattern the engine passes to ripgrep, so an adapter may cap pattern length (the sandbox
|
|
14
|
+
* adapter rejects patterns over 16,384 characters).
|
|
15
|
+
*/
|
|
16
|
+
export const CODE_SEARCH_MAX_PATTERN_CHARS = 16_000;
|
|
17
|
+
|
|
18
|
+
export type CodeSearchRipgrepResult = {
|
|
19
|
+
stdout: string;
|
|
20
|
+
exitCode: number | null;
|
|
21
|
+
truncated: boolean;
|
|
22
|
+
timedOut: boolean;
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
export interface CodeSearchWorkspace {
|
|
26
|
+
/** Run ripgrep in the workspace root. `args` excludes the binary. stdout is exactly what rg prints (up to the adapter's byte cap; `truncated` reports a cut). exitCode follows rg: 0 matches, 1 none, 2 error. Throws CodeSearchWorkspaceError when ripgrep is missing or the workspace is unreachable. */
|
|
27
|
+
ripgrep(
|
|
28
|
+
args: readonly string[],
|
|
29
|
+
options: { signal?: AbortSignal; timeoutMs: number },
|
|
30
|
+
): Promise<CodeSearchRipgrepResult>;
|
|
31
|
+
/** Read a workspace-relative file as UTF-8. Null when missing. */
|
|
32
|
+
readText(
|
|
33
|
+
path: string,
|
|
34
|
+
options: { signal?: AbortSignal; maxBytes: number },
|
|
35
|
+
): Promise<{ text: string; truncated: boolean; binary: boolean } | null>;
|
|
36
|
+
/** Classify workspace-relative paths. */
|
|
37
|
+
pathKinds(
|
|
38
|
+
paths: readonly string[],
|
|
39
|
+
options: { signal?: AbortSignal },
|
|
40
|
+
): Promise<Record<string, "file" | "directory" | "missing">>;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export class CodeSearchWorkspaceError extends Error {}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Thrown by an adapter when the ripgrep binary is not installed, so the tool can tell the model to use
|
|
47
|
+
* other search commands. A plain CodeSearchWorkspaceError whose message says ripgrep is missing is
|
|
48
|
+
* recognised too.
|
|
49
|
+
*/
|
|
50
|
+
export class CodeSearchRipgrepMissingError extends CodeSearchWorkspaceError {
|
|
51
|
+
constructor(message = "ripgrep (rg) is not installed in this workspace") {
|
|
52
|
+
super(message);
|
|
53
|
+
this.name = "CodeSearchRipgrepMissingError";
|
|
54
|
+
}
|
|
55
|
+
}
|