@opengeni/jev 0.1.0-canary.36199476632001

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,89 @@
1
+ /**
2
+ * session.ts - one search's view of the workspace: counts calls, records partial ripgrep output and
3
+ * makes every call honour the search's AbortSignal.
4
+ */
5
+ import { CodeSearchWorkspaceError, type CodeSearchWorkspace } from "./workspace";
6
+
7
+ /** Parallel reads per search. */
8
+ export const READ_CONCURRENCY = 8;
9
+ /** Parallel ripgrep calls for one pattern split under CODE_SEARCH_MAX_PATTERN_CHARS. */
10
+ export const RIPGREP_SPLIT_CONCURRENCY = 4;
11
+
12
+ export class WorkspaceSession {
13
+ calls = 0;
14
+ /** A ripgrep call returned output cut at the adapter's byte cap. */
15
+ truncated = false;
16
+ /** A ripgrep call hit its time limit (its output is partial). */
17
+ timedOut = false;
18
+
19
+ constructor(
20
+ private readonly workspace: CodeSearchWorkspace,
21
+ readonly signal: AbortSignal,
22
+ private readonly ripgrepTimeoutMs: number,
23
+ ) {}
24
+
25
+ get partial(): boolean {
26
+ return this.truncated || this.timedOut;
27
+ }
28
+
29
+ /**
30
+ * ripgrep stdout. Exit code 2 (error) with some output keeps the output (for example unreadable files);
31
+ * with no output it throws unless `allowFailure`, which then yields "".
32
+ */
33
+ async ripgrep(
34
+ args: readonly string[],
35
+ options: { allowFailure?: boolean } = {},
36
+ ): Promise<string> {
37
+ this.signal.throwIfAborted();
38
+ this.calls++;
39
+ const r = await this.workspace.ripgrep(args, {
40
+ signal: this.signal,
41
+ timeoutMs: this.ripgrepTimeoutMs,
42
+ });
43
+ this.signal.throwIfAborted();
44
+ if (r.truncated) this.truncated = true;
45
+ if (r.timedOut) this.timedOut = true;
46
+ if (r.exitCode === 2 && !r.stdout && !r.truncated && !r.timedOut && !options.allowFailure) {
47
+ throw new CodeSearchWorkspaceError("ripgrep failed (exit code 2) without output");
48
+ }
49
+ return r.stdout;
50
+ }
51
+
52
+ /** File text, or null when missing or binary (NUL bytes). */
53
+ async readText(path: string, maxBytes: number): Promise<string | null> {
54
+ this.signal.throwIfAborted();
55
+ this.calls++;
56
+ const r = await this.workspace.readText(path, { signal: this.signal, maxBytes });
57
+ this.signal.throwIfAborted();
58
+ if (!r || r.binary || r.text.includes("\u0000")) return null;
59
+ return r.text;
60
+ }
61
+
62
+ async pathKinds(
63
+ paths: readonly string[],
64
+ ): Promise<Record<string, "file" | "directory" | "missing">> {
65
+ this.signal.throwIfAborted();
66
+ this.calls++;
67
+ const r = await this.workspace.pathKinds(paths, { signal: this.signal });
68
+ this.signal.throwIfAborted();
69
+ return r;
70
+ }
71
+ }
72
+
73
+ /** Map with at most `limit` calls in flight; results keep the input order. */
74
+ export async function mapLimit<T, R>(
75
+ items: readonly T[],
76
+ limit: number,
77
+ fn: (item: T) => Promise<R>,
78
+ ): Promise<R[]> {
79
+ const out = new Array<R>(items.length);
80
+ let next = 0;
81
+ const worker = async () => {
82
+ while (next < items.length) {
83
+ const i = next++;
84
+ out[i] = await fn(items[i]!);
85
+ }
86
+ };
87
+ await Promise.all(Array.from({ length: Math.min(limit, items.length) }, worker));
88
+ return out;
89
+ }
@@ -0,0 +1,159 @@
1
+ /**
2
+ * text.ts - keyword variants, word splitting, stopwords and small text helpers.
3
+ */
4
+
5
+ /** Split an identifier or phrase into lowercase words: fooBarBAZ_qux-v2 -> [foo, bar, baz, qux, v2]. */
6
+ export function splitWords(s: string): string[] {
7
+ return s
8
+ .replace(/([a-z0-9])([A-Z])/g, "$1 $2")
9
+ .replace(/([A-Z]+)([A-Z][a-z])/g, "$1 $2")
10
+ .split(/[^A-Za-z0-9]+/)
11
+ .filter(Boolean)
12
+ .map((w) => w.toLowerCase());
13
+ }
14
+
15
+ const cap = (w: string) => (w ? w[0]!.toUpperCase() + w.slice(1) : w);
16
+
17
+ /**
18
+ * Identifier variants of a keyword. The search is case-insensitive, so only the
19
+ * separator forms matter: concatenated (camel/Pascal/flat), snake/SCREAMING, kebab, spaced phrase.
20
+ * Returned in display casing (camel, snake, kebab, spaced) for the trace; deduped case-insensitively.
21
+ */
22
+ export function keywordVariants(kw: string): string[] {
23
+ const raw = kw.trim();
24
+ if (!raw) return [];
25
+ const words = splitWords(raw);
26
+ const out: string[] = [raw];
27
+ if (words.length >= 2) {
28
+ out.push(words[0] + words.slice(1).map(cap).join("")); // camelCase (== PascalCase / flat under -i)
29
+ out.push(words.join("_")); // snake_case (== SCREAMING_SNAKE under -i)
30
+ out.push(words.join("-")); // kebab-case
31
+ out.push(words.join(" ")); // spaced phrase (prose in docs)
32
+ }
33
+ const seen = new Set<string>();
34
+ return out.filter((v) => {
35
+ const k = v.toLowerCase();
36
+ if (seen.has(k) || k.length < 2) return false;
37
+ seen.add(k);
38
+ return true;
39
+ });
40
+ }
41
+
42
+ /** Contiguous 2-word fragments of a compound with >= 3 words (fallback when the full name has no hits). */
43
+ export function compoundFragments(kw: string): string[] {
44
+ const words = splitWords(kw);
45
+ if (words.length < 3) return [];
46
+ const out: string[] = [];
47
+ for (let i = 0; i + 1 < words.length; i++) {
48
+ const a = words[i]!;
49
+ const b = words[i + 1]!;
50
+ if (a.length + b.length < 8) continue; // skip tiny fragments like "is" + "on"
51
+ out.push(a + cap(b));
52
+ }
53
+ return [...new Set(out)];
54
+ }
55
+
56
+ /** A single plain word (no case change, no separator) short enough to need a word boundary. */
57
+ export function isShortPlainWord(kw: string, maxLen: number): boolean {
58
+ return /^[A-Za-z]+$/.test(kw) && kw.length <= maxLen && splitWords(kw).length === 1;
59
+ }
60
+
61
+ /** Escape for Rust regex (ripgrep). */
62
+ export function escapeRegex(s: string): string {
63
+ return s.replace(/[\\.+*?()|[\]{}^$#&\-~]/g, (m) => `\\${m}`);
64
+ }
65
+
66
+ export const STOPWORDS = new Set(
67
+ (
68
+ "a an and are as at be been but by can could did do does doing done for from had has have how i if in into is it its " +
69
+ "just me my no not of on or our should so some such than that the their them then there these they this those to " +
70
+ "too was we were what when where which while who why will with would you your also any each else ever every much " +
71
+ "more most other same very about above after again against all am before being below between both during few " +
72
+ "further here him his her hers himself itself let may might must nor now off once only own over under until up " +
73
+ "use used using want way get got make makes happen happens happening still actually really even whether instead " +
74
+ "rather something someone thing things one two three kind already mean means"
75
+ ).split(" "),
76
+ );
77
+
78
+ /** Crude stemmer: good enough for overlap scoring (not for display). */
79
+ export function stem(w: string): string {
80
+ let s = w.toLowerCase();
81
+ if (s.length > 5 && s.endsWith("ies")) s = s.slice(0, -3) + "y";
82
+ else if (s.length > 5 && (s.endsWith("ing") || s.endsWith("ers"))) s = s.slice(0, -3);
83
+ else if (s.length > 4 && (s.endsWith("ed") || s.endsWith("es") || s.endsWith("er")))
84
+ s = s.slice(0, -2);
85
+ else if (s.length > 3 && s.endsWith("s") && !s.endsWith("ss")) s = s.slice(0, -1);
86
+ return s;
87
+ }
88
+
89
+ /** Content terms of a text (identifiers split into words, stopwords dropped, stemmed, deduped). */
90
+ export function contentTerms(text: string, minLen = 3): string[] {
91
+ const out = new Set<string>();
92
+ for (const tok of text.match(/[A-Za-z][A-Za-z0-9_]*/g) ?? []) {
93
+ for (const w of splitWords(tok)) {
94
+ if (w.length < minLen || STOPWORDS.has(w)) continue;
95
+ out.add(stem(w));
96
+ }
97
+ }
98
+ return [...out];
99
+ }
100
+
101
+ /** Fraction of `terms` present in `textTerms`. */
102
+ export function overlap(terms: string[], textTerms: Set<string>): number {
103
+ if (!terms.length) return 0;
104
+ let n = 0;
105
+ for (const t of terms) if (textTerms.has(t)) n++;
106
+ return n / terms.length;
107
+ }
108
+
109
+ export function isTestPath(p: string): boolean {
110
+ return (
111
+ /(^|\/)(test|tests|__tests__|spec|e2e|fixtures?)\//.test(p) || /\.(test|spec)\.[a-z]+$/.test(p)
112
+ );
113
+ }
114
+
115
+ /** Release notes: CHANGELOG*, HISTORY*, RELEASE_NOTES*, .changeset/ entries (describe past changes, not current code). */
116
+ export function isChangelogPath(p: string): boolean {
117
+ return (
118
+ /(^|\/)(CHANGELOG|HISTORY|RELEASE[-_]?NOTES)[^/]*$/i.test(p) || /(^|\/)\.changeset\//.test(p)
119
+ );
120
+ }
121
+
122
+ /** Is the question about history / releases (then release notes are not down-weighted)? */
123
+ export function questionMentionsHistory(q: string): boolean {
124
+ return /\b(changelog|release notes?|released|history|historical|when (?:was|did)|which version|since version|changed in)\b/i.test(
125
+ q,
126
+ );
127
+ }
128
+
129
+ export function isDocPath(p: string): boolean {
130
+ return /\.(md|mdx|txt|rst)$/i.test(p);
131
+ }
132
+
133
+ export function questionMentionsTests(q: string): boolean {
134
+ return /\b(test|tests|testing|spec|specs|unit test|e2e|fixture)\b/i.test(q);
135
+ }
136
+
137
+ /** Trim a line to maxChars around the first occurrence of any needle (case-insensitive). */
138
+ export function trimAround(line: string, needles: string[], maxChars: number): string {
139
+ const t = line.replace(/\t/g, " ").trim();
140
+ if (t.length <= maxChars) return t;
141
+ const lower = t.toLowerCase();
142
+ let at = -1;
143
+ for (const n of needles) {
144
+ const i = lower.indexOf(n.toLowerCase());
145
+ if (i >= 0 && (at < 0 || i < at)) at = i;
146
+ }
147
+ if (at < 0) return t.slice(0, maxChars - 3) + "...";
148
+ const start = Math.max(0, Math.min(at - Math.floor(maxChars / 3), t.length - maxChars));
149
+ const s = t.slice(start, start + maxChars - 6);
150
+ return (start > 0 ? "..." : "") + s + (start + maxChars - 6 < t.length ? "..." : "");
151
+ }
152
+
153
+ export function estTokens(chars: number, charsPerToken: number): number {
154
+ return Math.ceil(chars / charsPerToken);
155
+ }
156
+
157
+ export function fmtK(n: number): string {
158
+ return n >= 1000 ? `${(n / 1000).toFixed(1)}k` : String(n);
159
+ }
@@ -0,0 +1,209 @@
1
+ /**
2
+ * tool.ts - the model-facing surface of code_search: name, description, input schema, argument
3
+ * parsing and short error texts. The worker wires these to runCodeSearch.
4
+ */
5
+ import { JevRequestError, JevUnavailableError } from "../client";
6
+ import { CodeSearchRipgrepMissingError, CodeSearchWorkspaceError } from "./workspace";
7
+
8
+ export const CODE_SEARCH_TOOL_NAME = "code_search";
9
+
10
+ export const CODE_SEARCH_TOOL_DESCRIPTION =
11
+ "Find where something is implemented, configured or decided in the code under the working directory, in one call " +
12
+ "instead of many separate searches and file reads. Give one precise question and 6-15 keywords: likely identifiers, " +
13
+ "file-name fragments, config keys, error strings and synonyms. Optional subQuestions split distinct parts (for " +
14
+ "'is X required?', add one for what could skip or override X); optional paths limit the search. It ranks files and " +
15
+ "passages with a fast relevance model, follows definitions one level, and returns the best passages verbatim with " +
16
+ "file paths and line numbers, plus an evidence rating for those passages. The rating cannot see what the search " +
17
+ "missed: use the passages instead of re-reading them, then check what they do not cover (other entry points, " +
18
+ "defaults, flags, exceptions) before concluding.";
19
+
20
+ export const CODE_SEARCH_LIMITS = {
21
+ questionMinChars: 3,
22
+ questionMaxChars: 2000,
23
+ keywordsMax: 20,
24
+ keywordMaxChars: 120,
25
+ subQuestionsMax: 3,
26
+ subQuestionMaxChars: 1000,
27
+ pathsMax: 8,
28
+ pathMaxChars: 1000,
29
+ } as const;
30
+
31
+ const L = CODE_SEARCH_LIMITS;
32
+
33
+ /** Plain JSON (mutable arrays), so it fits any JSON-schema slot of a tool definition. */
34
+ export type CodeSearchJsonValue =
35
+ | string
36
+ | number
37
+ | boolean
38
+ | null
39
+ | CodeSearchJsonValue[]
40
+ | { [key: string]: CodeSearchJsonValue };
41
+
42
+ /** JSON schema of the tool input; parseCodeSearchArguments enforces the same rules. */
43
+ export const codeSearchInputSchema: { [key: string]: CodeSearchJsonValue } = {
44
+ type: "object",
45
+ properties: {
46
+ question: {
47
+ type: "string",
48
+ minLength: L.questionMinChars,
49
+ maxLength: L.questionMaxChars,
50
+ description:
51
+ "One precise question about the code, for example where a behavior is implemented or how a value is computed.",
52
+ },
53
+ keywords: {
54
+ type: "array",
55
+ minItems: 1,
56
+ maxItems: L.keywordsMax,
57
+ items: { type: "string", minLength: 1, maxLength: L.keywordMaxChars },
58
+ description:
59
+ "6-15 search keywords: likely identifiers (camelCase, snake_case, UPPER_CASE), file-name fragments, config or env keys, error strings, synonyms. Case and camel/snake/kebab variants are searched automatically.",
60
+ },
61
+ subQuestions: {
62
+ type: "array",
63
+ maxItems: L.subQuestionsMax,
64
+ items: { type: "string", minLength: 1, maxLength: L.subQuestionMaxChars },
65
+ description: "Optional: the distinct parts of a multi-part question, one per entry.",
66
+ },
67
+ paths: {
68
+ type: "array",
69
+ maxItems: L.pathsMax,
70
+ items: { type: "string", minLength: 1, maxLength: L.pathMaxChars },
71
+ description:
72
+ "Optional: workspace-relative directories or files to limit the search to. Default: the whole workspace.",
73
+ },
74
+ },
75
+ required: ["question", "keywords"],
76
+ additionalProperties: false,
77
+ };
78
+
79
+ export interface CodeSearchArguments {
80
+ question: string;
81
+ keywords: string[];
82
+ subQuestions: string[];
83
+ paths: string[];
84
+ }
85
+
86
+ /** Invalid tool arguments; the message is written for the model. */
87
+ export class CodeSearchArgumentError extends Error {
88
+ constructor(message: string) {
89
+ super(message);
90
+ this.name = "CodeSearchArgumentError";
91
+ }
92
+ }
93
+
94
+ const KNOWN_ARGUMENTS = new Set(["question", "keywords", "subQuestions", "paths"]);
95
+
96
+ /** Validate and normalize tool arguments (trim, drop empty entries, dedupe, workspace-relative paths). */
97
+ export function parseCodeSearchArguments(args: Record<string, unknown>): CodeSearchArguments {
98
+ const unknown = Object.keys(args).filter((k) => !KNOWN_ARGUMENTS.has(k));
99
+ if (unknown.length) {
100
+ throw new CodeSearchArgumentError(
101
+ `unknown argument ${unknown.map((k) => `"${k}"`).join(", ")}; allowed: question, keywords, subQuestions, paths`,
102
+ );
103
+ }
104
+
105
+ if (typeof args.question !== "string")
106
+ throw new CodeSearchArgumentError("question is required and must be a string");
107
+ const question = args.question.trim();
108
+ if (question.length < L.questionMinChars)
109
+ throw new CodeSearchArgumentError(`question must be at least ${L.questionMinChars} characters`);
110
+ if (question.length > L.questionMaxChars)
111
+ throw new CodeSearchArgumentError(`question must be at most ${L.questionMaxChars} characters`);
112
+
113
+ const keywords = stringList(args.keywords, "keywords", true);
114
+ if (!keywords.length)
115
+ throw new CodeSearchArgumentError("keywords must contain at least one non-empty keyword");
116
+ if (keywords.length > L.keywordsMax)
117
+ throw new CodeSearchArgumentError(
118
+ `keywords accepts at most ${L.keywordsMax} entries; keep the ${L.keywordsMax} most specific`,
119
+ );
120
+ const longKeyword = keywords.find((k) => k.length > L.keywordMaxChars);
121
+ if (longKeyword)
122
+ throw new CodeSearchArgumentError(
123
+ `each keyword must be at most ${L.keywordMaxChars} characters; use short identifiers or phrases`,
124
+ );
125
+
126
+ const subQuestions = stringList(args.subQuestions, "subQuestions", false);
127
+ if (subQuestions.length > L.subQuestionsMax)
128
+ throw new CodeSearchArgumentError(`subQuestions accepts at most ${L.subQuestionsMax} entries`);
129
+ if (subQuestions.some((s) => s.length > L.subQuestionMaxChars)) {
130
+ throw new CodeSearchArgumentError(
131
+ `each sub-question must be at most ${L.subQuestionMaxChars} characters`,
132
+ );
133
+ }
134
+
135
+ const paths = [...new Set(stringList(args.paths, "paths", false).map(normalizePath))];
136
+ if (paths.length > L.pathsMax)
137
+ throw new CodeSearchArgumentError(`paths accepts at most ${L.pathsMax} entries`);
138
+
139
+ return { question, keywords, subQuestions, paths };
140
+ }
141
+
142
+ function stringList(value: unknown, name: string, required: boolean): string[] {
143
+ if (value === undefined || value === null) {
144
+ if (required)
145
+ throw new CodeSearchArgumentError(`${name} is required and must be an array of strings`);
146
+ return [];
147
+ }
148
+ if (!Array.isArray(value) || value.some((v) => typeof v !== "string")) {
149
+ throw new CodeSearchArgumentError(`${name} must be an array of strings`);
150
+ }
151
+ return [...new Set((value as string[]).map((v) => v.trim()).filter(Boolean))];
152
+ }
153
+
154
+ function normalizePath(raw: string): string {
155
+ if (raw.length > L.pathMaxChars)
156
+ throw new CodeSearchArgumentError(`each path must be at most ${L.pathMaxChars} characters`);
157
+ if (raw.startsWith("/") || raw.startsWith("~") || /^[A-Za-z]:[\\/]/.test(raw)) {
158
+ throw new CodeSearchArgumentError(
159
+ `path "${raw}" is absolute; give paths relative to the working directory`,
160
+ );
161
+ }
162
+ let p = raw;
163
+ while (p.startsWith("./")) p = p.slice(2);
164
+ p = p.replace(/\/+$/, "");
165
+ if (!p) p = ".";
166
+ if (p.split("/").some((seg) => seg === "..")) {
167
+ throw new CodeSearchArgumentError(
168
+ `path "${raw}" leaves the working directory; ".." is not allowed`,
169
+ );
170
+ }
171
+ if (p.startsWith("-")) throw new CodeSearchArgumentError(`path "${raw}" must not start with "-"`);
172
+ return p;
173
+ }
174
+
175
+ const FALLBACK = "Search with exec_command (rg, sed) instead.";
176
+
177
+ /** Short model-facing text for a failed code_search call. */
178
+ export function renderCodeSearchError(error: unknown): string {
179
+ const detail = (e: unknown) =>
180
+ (e instanceof Error ? e.message : String(e)).replace(/\s+/g, " ").trim().slice(0, 240);
181
+ if (error instanceof CodeSearchArgumentError)
182
+ return `code_search: invalid arguments: ${error.message}.`;
183
+ if (error instanceof JevUnavailableError)
184
+ return `code_search is unavailable right now (${detail(error)}). ${FALLBACK}`;
185
+ if (error instanceof JevRequestError)
186
+ return `code_search failed: the relevance model rejected the request (${detail(error)}). ${FALLBACK}`;
187
+ if (isRipgrepMissing(error)) {
188
+ return "code_search cannot run here: ripgrep (rg) is not installed in this workspace. Search with exec_command (grep, find, sed) instead.";
189
+ }
190
+ if (error instanceof CodeSearchWorkspaceError)
191
+ return `code_search could not search this workspace (${detail(error)}). ${FALLBACK}`;
192
+ if (isAbort(error)) return "code_search was cancelled.";
193
+ return `code_search failed unexpectedly (${detail(error)}). ${FALLBACK}`;
194
+ }
195
+
196
+ function isRipgrepMissing(error: unknown): boolean {
197
+ if (error instanceof CodeSearchRipgrepMissingError) return true;
198
+ if (!(error instanceof CodeSearchWorkspaceError)) return false;
199
+ const m = error.message;
200
+ return /ripgrep|\brg\b/i.test(m) && /not (installed|found)|missing|ENOENT|no such file/i.test(m);
201
+ }
202
+
203
+ function isAbort(error: unknown): boolean {
204
+ return (
205
+ typeof error === "object" &&
206
+ error !== null &&
207
+ (error as { name?: unknown }).name === "AbortError"
208
+ );
209
+ }