@selesai/code 0.13.33 → 0.13.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/dist/extensions/capability-gateway/catalog.ts +4 -1
- package/dist/extensions/capability-gateway/index.test.ts +93 -4
- package/dist/extensions/capability-gateway/index.ts +295 -59
- package/dist/extensions/capability-gateway/integration.test.ts +35 -0
- package/dist/extensions/capability-gateway/routing.test.ts +154 -1
- package/dist/extensions/capability-gateway/routing.ts +227 -45
- package/dist/extensions/jev/decisions.test.ts +37 -0
- package/dist/extensions/jev/decisions.ts +161 -43
- package/dist/extensions/jev-ask-tool.test.ts +501 -0
- package/dist/extensions/jev-ask-tool.ts +952 -0
- package/dist/extensions/package.json +1 -0
- package/dist/extensions/pi-hermes-memory/README.md +11 -36
- package/dist/extensions/pi-hermes-memory/src/config.ts +36 -8
- package/dist/extensions/pi-hermes-memory/src/constants.ts +5 -5
- package/dist/extensions/pi-hermes-memory/src/handlers/auto-consolidate.ts +60 -46
- package/dist/extensions/pi-hermes-memory/tests/config.test.ts +26 -2
- package/dist/extensions/pi-hermes-memory/tests/handlers/auto-consolidate.test.ts +7 -1
- package/dist/extensions/pi-intercom/index.ts +5 -1
- package/dist/extensions/pi-subagents/src/extension/public-execution.ts +6 -4
- package/dist/extensions/pi-subagents/src/extension/schemas.ts +1 -1
- package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +26 -1
- package/dist/extensions/pi-subagents/src/runs/shared/jev-subagent-routing.ts +255 -0
- package/dist/extensions/pi-subagents/test/unit/jev-subagent-routing.test.ts +117 -0
- package/dist/extensions/pi-subagents/test/unit/public-execution.test.ts +3 -1
- package/dist/extensions/rtk.test.ts +21 -13
- package/dist/extensions/tps.test.ts +32 -1
- package/dist/extensions/tps.ts +3 -1
- package/docs/settings.md +64 -7
- package/package.json +3 -3
|
@@ -0,0 +1,501 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The agent-callable `ask_jev` tool.
|
|
3
|
+
*
|
|
4
|
+
* The tool is driven the way the model drives it: one call carrying its own
|
|
5
|
+
* prose, the paths it wants read, one command, and a question block. Tests
|
|
6
|
+
* assert what the agent observes — the state Jev was sent, the answers that come
|
|
7
|
+
* back, and what never leaves the process — never private helper order.
|
|
8
|
+
*/
|
|
9
|
+
import { existsSync, mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
|
|
10
|
+
import { tmpdir } from "node:os";
|
|
11
|
+
import { join } from "node:path";
|
|
12
|
+
import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from "vitest";
|
|
13
|
+
import type { AssistantMessage } from "@earendil-works/pi-ai";
|
|
14
|
+
|
|
15
|
+
const state = vi.hoisted(() => ({
|
|
16
|
+
settingsPath: "",
|
|
17
|
+
commands: new Map<string, { output: string; exitCode: number }>(),
|
|
18
|
+
}));
|
|
19
|
+
|
|
20
|
+
vi.mock("@selesai/code", () => ({
|
|
21
|
+
getSettingsPath: () => state.settingsPath,
|
|
22
|
+
// jev_find runs the real ripgrep on PATH against the temp repo.
|
|
23
|
+
ensureTool: async () => "rg",
|
|
24
|
+
// The same local shell backend the bash tool uses, scripted per command.
|
|
25
|
+
createLocalBashOperations: () => ({
|
|
26
|
+
exec: async (command: string, _cwd: string, options: { onData: (data: Buffer) => void }) => {
|
|
27
|
+
const scripted = state.commands.get(command);
|
|
28
|
+
if (!scripted) return { exitCode: 127 };
|
|
29
|
+
options.onData(Buffer.from(scripted.output, "utf-8"));
|
|
30
|
+
return { exitCode: scripted.exitCode };
|
|
31
|
+
},
|
|
32
|
+
}),
|
|
33
|
+
}));
|
|
34
|
+
|
|
35
|
+
import jevAskToolExtension, { askPath, FIND_BATCH, fitFindPayload, MAX_ASK_QUESTIONS } from "./jev-ask-tool.ts";
|
|
36
|
+
import { JEV_ROUTING_EVENT, serializeJevRequest } from "./jev/decisions.ts";
|
|
37
|
+
import { jevResponse, providerTemplate } from "./jev/test-support.ts";
|
|
38
|
+
|
|
39
|
+
const FILE_SENTINEL = "FILE_BODY_MUST_RETURN_TO_JE ONLY";
|
|
40
|
+
const COMMAND_SENTINEL = "FAIL rounding test at round.ts:12";
|
|
41
|
+
|
|
42
|
+
let root: string;
|
|
43
|
+
let repo: string;
|
|
44
|
+
|
|
45
|
+
beforeAll(() => {
|
|
46
|
+
root = mkdtempSync(join(tmpdir(), "jev-ask-tool-"));
|
|
47
|
+
repo = join(root, "repo");
|
|
48
|
+
mkdirSync(join(repo, "src"), { recursive: true });
|
|
49
|
+
writeFileSync(join(repo, "src", "round.ts"), `export const round = () => ${FILE_SENTINEL};\n`, "utf-8");
|
|
50
|
+
state.settingsPath = join(root, "agent", "settings.json");
|
|
51
|
+
mkdirSync(join(root, "agent"), { recursive: true });
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
afterAll(() => rmSync(root, { recursive: true, force: true }));
|
|
55
|
+
|
|
56
|
+
beforeEach(() => {
|
|
57
|
+
state.commands.clear();
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
/** The answers envelope Jev returns, with either verdict field per question. */
|
|
61
|
+
function answerText(answers: Record<string, unknown>): string {
|
|
62
|
+
return JSON.stringify({ answers });
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
interface ToolDefinition {
|
|
66
|
+
name: string;
|
|
67
|
+
description: string;
|
|
68
|
+
promptSnippet?: string;
|
|
69
|
+
promptGuidelines?: string[];
|
|
70
|
+
execute: (
|
|
71
|
+
id: string,
|
|
72
|
+
params: Record<string, unknown>,
|
|
73
|
+
signal: AbortSignal | undefined,
|
|
74
|
+
onUpdate: undefined,
|
|
75
|
+
ctx: unknown,
|
|
76
|
+
) => Promise<{ content: Array<{ type: string; text: string }>; details: Record<string, unknown> }>;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
interface AskHarness {
|
|
80
|
+
tool: ToolDefinition;
|
|
81
|
+
complete: ReturnType<typeof vi.fn>;
|
|
82
|
+
telemetry: Record<string, unknown>[];
|
|
83
|
+
/** Warnings shown to the user. */
|
|
84
|
+
notices: string[];
|
|
85
|
+
/** The live active-tool loadout the extension may trim or restore. */
|
|
86
|
+
active: string[];
|
|
87
|
+
/** Fire the extension's `before_agent_start` hook, as the session does before every run. */
|
|
88
|
+
beforeRun(): Promise<void>;
|
|
89
|
+
/** Swap credential availability mid-session, as `/tokenin add` would. */
|
|
90
|
+
setCredential(available: boolean): void;
|
|
91
|
+
ask(params: Record<string, unknown>, signal?: AbortSignal): Promise<{ text: string; details: Record<string, unknown> }>;
|
|
92
|
+
/** The decision request Jev was sent, parsed. */
|
|
93
|
+
sent(): Record<string, unknown> | undefined;
|
|
94
|
+
/** Every tool the extension registered, by name. */
|
|
95
|
+
tools: Record<string, ToolDefinition>;
|
|
96
|
+
ctx: unknown;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function harness(
|
|
100
|
+
options: { enabled?: boolean; credential?: boolean; template?: boolean; payloadBytes?: number } = {},
|
|
101
|
+
): AskHarness {
|
|
102
|
+
writeFileSync(
|
|
103
|
+
state.settingsPath,
|
|
104
|
+
JSON.stringify({
|
|
105
|
+
jevAdvisory: {
|
|
106
|
+
routes: {
|
|
107
|
+
ask: {
|
|
108
|
+
enabled: options.enabled ?? true,
|
|
109
|
+
...(options.payloadBytes === undefined ? {} : { payloadBytes: options.payloadBytes }),
|
|
110
|
+
},
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
}),
|
|
114
|
+
"utf-8",
|
|
115
|
+
);
|
|
116
|
+
const telemetry: Record<string, unknown>[] = [];
|
|
117
|
+
const notices: string[] = [];
|
|
118
|
+
const active = ["read", "bash", "ask_jev"];
|
|
119
|
+
const hooks: Record<string, (event: unknown, ctx: unknown) => unknown> = {};
|
|
120
|
+
let credential = options.credential !== false;
|
|
121
|
+
const complete = vi.fn(async () => jevResponse(answerText({})));
|
|
122
|
+
const tools: Record<string, ToolDefinition> = {};
|
|
123
|
+
jevAskToolExtension({
|
|
124
|
+
registerTool: (definition: ToolDefinition) => {
|
|
125
|
+
tools[definition.name] = definition;
|
|
126
|
+
},
|
|
127
|
+
on: (event: string, handler: (event: unknown, ctx: unknown) => unknown) => {
|
|
128
|
+
hooks[event] = handler;
|
|
129
|
+
},
|
|
130
|
+
getActiveTools: () => [...active],
|
|
131
|
+
setActiveTools: (names: string[]) => {
|
|
132
|
+
active.splice(0, active.length, ...names);
|
|
133
|
+
},
|
|
134
|
+
events: {
|
|
135
|
+
emit: (channel: string, data: unknown) => {
|
|
136
|
+
if (channel === JEV_ROUTING_EVENT) telemetry.push(data as Record<string, unknown>);
|
|
137
|
+
},
|
|
138
|
+
},
|
|
139
|
+
} as never);
|
|
140
|
+
const ctx = {
|
|
141
|
+
cwd: repo,
|
|
142
|
+
hasUI: true,
|
|
143
|
+
ui: { notify: (message: string) => notices.push(message) },
|
|
144
|
+
modelRegistry: {
|
|
145
|
+
getAll: () => (options.template === false ? [] : [providerTemplate()]),
|
|
146
|
+
getApiKeyAndHeaders: async () =>
|
|
147
|
+
credential ? { ok: true, apiKey: "key", headers: {} } : { ok: false, error: "no key" },
|
|
148
|
+
complete,
|
|
149
|
+
},
|
|
150
|
+
};
|
|
151
|
+
const tool = tools.ask_jev as ToolDefinition;
|
|
152
|
+
return {
|
|
153
|
+
tool,
|
|
154
|
+
tools,
|
|
155
|
+
ctx,
|
|
156
|
+
complete,
|
|
157
|
+
telemetry,
|
|
158
|
+
notices,
|
|
159
|
+
active,
|
|
160
|
+
async beforeRun() {
|
|
161
|
+
await hooks.before_agent_start?.({ prompt: "x" }, ctx);
|
|
162
|
+
},
|
|
163
|
+
setCredential(available) {
|
|
164
|
+
credential = available;
|
|
165
|
+
},
|
|
166
|
+
async ask(params, signal) {
|
|
167
|
+
const result = await tool.execute("call-1", params, signal, undefined, ctx);
|
|
168
|
+
const content = result.content[0];
|
|
169
|
+
return { text: content?.type === "text" ? content.text : "", details: result.details };
|
|
170
|
+
},
|
|
171
|
+
sent() {
|
|
172
|
+
const call = complete.mock.calls.at(-1) as unknown[] | undefined;
|
|
173
|
+
const content = (call?.[1] as { messages?: Array<{ content?: string }> } | undefined)?.messages?.[0]?.content;
|
|
174
|
+
return typeof content === "string" ? (JSON.parse(content) as Record<string, unknown>) : undefined;
|
|
175
|
+
},
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
function sentBytes(complete: ReturnType<typeof vi.fn>): number {
|
|
180
|
+
const call = complete.mock.calls.at(-1) as unknown[] | undefined;
|
|
181
|
+
const content = (call?.[1] as { messages?: Array<{ content?: string }> } | undefined)?.messages?.[0]?.content;
|
|
182
|
+
return Buffer.byteLength(content ?? "", "utf-8");
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const CHOICE = {
|
|
186
|
+
type: "choice",
|
|
187
|
+
instructions: "What kind of failure is this?",
|
|
188
|
+
criteria: { bug_in_code: "The code is wrong.", flaky_test: "The test is nondeterministic.", environment: "The setup is wrong." },
|
|
189
|
+
};
|
|
190
|
+
const NOUL = { type: "noul", instructions: "Do the tests pass?", criteria: { true: "They pass.", false: "They do not." } };
|
|
191
|
+
const SCORE = { type: "score", instructions: "How risky is this diff?", criteria: ["safe", "needs review", "dangerous"] };
|
|
192
|
+
|
|
193
|
+
const TYPICAL = {
|
|
194
|
+
state: "The tests are red and I have not read anything yet.",
|
|
195
|
+
paths: ["src/round.ts"],
|
|
196
|
+
command: "npm test",
|
|
197
|
+
questions: { failure_kind: CHOICE, tests_pass: NOUL },
|
|
198
|
+
};
|
|
199
|
+
|
|
200
|
+
describe("registration", () => {
|
|
201
|
+
it("registers one ask_jev tool carrying its question schema and the prompt nudge", () => {
|
|
202
|
+
const { tool } = harness();
|
|
203
|
+
expect(tool.name).toBe("ask_jev");
|
|
204
|
+
for (const schemaPart of ["choice", "score", "noul", "criteria"]) {
|
|
205
|
+
expect(tool.description).toContain(schemaPart);
|
|
206
|
+
}
|
|
207
|
+
expect(tool.description).toContain("answers only");
|
|
208
|
+
expect(tool.promptSnippet).toBeTruthy();
|
|
209
|
+
expect(tool.promptGuidelines?.[0]).toContain("ask_jev");
|
|
210
|
+
});
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
describe("one state, one call", () => {
|
|
214
|
+
it("assembles prose, code, and command output into one state and returns answers only", async () => {
|
|
215
|
+
state.commands.set("npm test", { output: COMMAND_SENTINEL, exitCode: 1 });
|
|
216
|
+
const session = harness();
|
|
217
|
+
session.complete.mockResolvedValue(
|
|
218
|
+
jevResponse(answerText({ failure_kind: { choice: "bug_in_code", confidence: 0.99 }, tests_pass: { noul: 0.04 } })),
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
const result = await session.ask(TYPICAL);
|
|
222
|
+
const sent = session.sent() as { state: Record<string, unknown> };
|
|
223
|
+
|
|
224
|
+
expect(session.complete).toHaveBeenCalledTimes(1);
|
|
225
|
+
expect(sent.state.request).toBe(TYPICAL.state);
|
|
226
|
+
expect(sent.state.command).toMatchObject({ command: "npm test", exitCode: 1 });
|
|
227
|
+
expect((sent.state.files as Record<string, string>)["src/round.ts"]).toContain(FILE_SENTINEL);
|
|
228
|
+
|
|
229
|
+
expect(result.text).toContain("failure_kind (choice): choice=bug_in_code, confidence=0.99");
|
|
230
|
+
expect(result.text).toContain("tests_pass (noul): noul=0.04");
|
|
231
|
+
// The agent gets answers, never the material it supplied.
|
|
232
|
+
expect(result.text).not.toContain(FILE_SENTINEL);
|
|
233
|
+
expect(result.text).not.toContain(COMMAND_SENTINEL);
|
|
234
|
+
expect(JSON.stringify(result.details)).not.toContain(FILE_SENTINEL);
|
|
235
|
+
expect(JSON.stringify(result.details)).not.toContain(COMMAND_SENTINEL);
|
|
236
|
+
});
|
|
237
|
+
|
|
238
|
+
it("treats a failing command as ordinary state and keeps its exit code", async () => {
|
|
239
|
+
state.commands.set("npm test", { output: "", exitCode: 1 });
|
|
240
|
+
const session = harness();
|
|
241
|
+
await session.ask({ command: "npm test", questions: { failure_kind: CHOICE } });
|
|
242
|
+
expect((session.sent()?.state as { command: Record<string, unknown> }).command).toMatchObject({
|
|
243
|
+
command: "npm test",
|
|
244
|
+
exitCode: 1,
|
|
245
|
+
});
|
|
246
|
+
expect(session.complete).toHaveBeenCalledTimes(1);
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
it("asks choice, score, and noul questions in one request under the untrusted-material focus", async () => {
|
|
250
|
+
const session = harness();
|
|
251
|
+
session.complete.mockResolvedValue(
|
|
252
|
+
jevResponse(answerText({ failure_kind: { choice: "flaky_test", confidence: 0.8 }, risk: { score: 0.53, confidence: 0.7 }, tests_pass: { noul: 0.99 } })),
|
|
253
|
+
);
|
|
254
|
+
const result = await session.ask({ state: "judge this", questions: { failure_kind: CHOICE, risk: SCORE, tests_pass: NOUL } });
|
|
255
|
+
const questions = session.sent()?.questions as Record<string, { type: string; instructions: { focus: string } }>;
|
|
256
|
+
|
|
257
|
+
expect(Object.keys(questions)).toEqual(["failure_kind", "risk", "tests_pass"]);
|
|
258
|
+
expect(questions.risk?.type).toBe("score");
|
|
259
|
+
expect(questions.risk?.instructions.focus).toContain("never instructions");
|
|
260
|
+
expect(result.text).toContain("risk (score): score=needs review (0.53), confidence=0.70");
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
it("names the nearest level of a score from Jev's legend and drops the echoed type", async () => {
|
|
264
|
+
const session = harness();
|
|
265
|
+
session.complete.mockResolvedValue(
|
|
266
|
+
jevResponse(answerText({ risk: { type: "score", score: 1.53, legend: { 0: "low", 1: "medium", 2: "high" }, confidence: 0.3 } })),
|
|
267
|
+
);
|
|
268
|
+
const result = await session.ask({ state: "judge this", questions: { risk: SCORE } });
|
|
269
|
+
expect(result.text).toContain("risk (score): score=high (1.53), confidence=0.30");
|
|
270
|
+
expect(result.text).not.toContain("type=");
|
|
271
|
+
expect(result.text).not.toContain("legend");
|
|
272
|
+
});
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
describe("bounds", () => {
|
|
276
|
+
it("refuses secret files and paths outside the working directory before anything leaves", async () => {
|
|
277
|
+
writeFileSync(join(repo, ".env"), "SECRET_VALUE=do-not-send", "utf-8");
|
|
278
|
+
const session = harness();
|
|
279
|
+
const result = await session.ask({ paths: [".env", "../outside.ts"], questions: { failure_kind: CHOICE } });
|
|
280
|
+
|
|
281
|
+
expect(session.sent()?.state).not.toHaveProperty("files");
|
|
282
|
+
expect(JSON.stringify(session.sent())).not.toContain("do-not-send");
|
|
283
|
+
expect(result.text).toContain(".env (looks like a secret file)");
|
|
284
|
+
expect(result.text).toContain("../outside.ts (outside the working directory)");
|
|
285
|
+
});
|
|
286
|
+
|
|
287
|
+
it("trims prose, code, and command output to the payload budget instead of abstaining", async () => {
|
|
288
|
+
state.commands.set("cat big", { output: "command ".repeat(3_000), exitCode: 0 });
|
|
289
|
+
const session = harness({ payloadBytes: 2_048 });
|
|
290
|
+
const result = await session.ask({
|
|
291
|
+
state: "prose ".repeat(3_000),
|
|
292
|
+
paths: ["src/round.ts"],
|
|
293
|
+
command: "cat big",
|
|
294
|
+
questions: { failure_kind: CHOICE },
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
expect(session.complete).toHaveBeenCalledTimes(1);
|
|
298
|
+
expect(sentBytes(session.complete)).toBeLessThanOrEqual(2_048);
|
|
299
|
+
expect(result.text).toContain("Truncated to fit the request budget");
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
it("rejects an unusable question block before any call", async () => {
|
|
303
|
+
const session = harness();
|
|
304
|
+
const unknownType = await session.ask({ questions: { q: { type: "magic", instructions: "?" } } });
|
|
305
|
+
expect(unknownType.text).toContain("magic");
|
|
306
|
+
|
|
307
|
+
const missingCriteria = await session.ask({ questions: { q: { type: "choice", instructions: "pick one" } } });
|
|
308
|
+
expect(missingCriteria.text).toContain("criteria");
|
|
309
|
+
|
|
310
|
+
const tooMany = await session.ask({
|
|
311
|
+
questions: Object.fromEntries(Array.from({ length: MAX_ASK_QUESTIONS + 1 }, (_, i) => [`q${i}`, CHOICE])),
|
|
312
|
+
});
|
|
313
|
+
expect(tooMany.text).toContain(`At most ${MAX_ASK_QUESTIONS}`);
|
|
314
|
+
|
|
315
|
+
expect(session.complete).not.toHaveBeenCalled();
|
|
316
|
+
});
|
|
317
|
+
});
|
|
318
|
+
|
|
319
|
+
describe("unavailability", () => {
|
|
320
|
+
it("stays silent while the route is disabled", async () => {
|
|
321
|
+
const session = harness({ enabled: false });
|
|
322
|
+
const result = await session.ask(TYPICAL);
|
|
323
|
+
expect(result.text).toContain("disabled");
|
|
324
|
+
expect(session.complete).not.toHaveBeenCalled();
|
|
325
|
+
});
|
|
326
|
+
|
|
327
|
+
it("reports an abstention rather than failing when Jev is unavailable", async () => {
|
|
328
|
+
const noCredential = harness({ credential: false });
|
|
329
|
+
const credentialless = await noCredential.ask({ state: "x", questions: { failure_kind: CHOICE } });
|
|
330
|
+
expect(noCredential.complete).not.toHaveBeenCalled();
|
|
331
|
+
expect(credentialless.text).toContain("no-credential");
|
|
332
|
+
expect(credentialless.details.failure).toBe("no-credential");
|
|
333
|
+
expect(credentialless.text).toContain("failure_kind: no answer (no-credential)");
|
|
334
|
+
|
|
335
|
+
const noTemplate = harness({ template: false });
|
|
336
|
+
expect((await noTemplate.ask({ state: "x", questions: { failure_kind: CHOICE } })).text).toContain("no-template");
|
|
337
|
+
});
|
|
338
|
+
});
|
|
339
|
+
|
|
340
|
+
describe("telemetry", () => {
|
|
341
|
+
it("names the unanswered questions and reports decisions without any content", async () => {
|
|
342
|
+
const session = harness();
|
|
343
|
+
session.complete.mockResolvedValue(jevResponse(answerText({ failure_kind: { choice: "flaky_test", confidence: 0.8 } })));
|
|
344
|
+
const result = await session.ask({ state: "SENTINEL_STATE", questions: { failure_kind: CHOICE, tests_pass: NOUL } });
|
|
345
|
+
|
|
346
|
+
expect(result.text).toContain("tests_pass: no answer (missing)");
|
|
347
|
+
expect(session.telemetry.at(-1)).toMatchObject({
|
|
348
|
+
event: "decision",
|
|
349
|
+
route: "ask",
|
|
350
|
+
outcome: "jev",
|
|
351
|
+
candidates: 2,
|
|
352
|
+
confidence: "medium",
|
|
353
|
+
});
|
|
354
|
+
const serialized = JSON.stringify(session.telemetry);
|
|
355
|
+
expect(serialized).not.toContain("SENTINEL_STATE");
|
|
356
|
+
expect(serialized).not.toContain("flaky_test");
|
|
357
|
+
});
|
|
358
|
+
});
|
|
359
|
+
|
|
360
|
+
describe("without a reachable Jev", () => {
|
|
361
|
+
it("neither reads the files nor runs the command, and warns once per session", async () => {
|
|
362
|
+
const session = harness({ credential: false });
|
|
363
|
+
const marker = join(repo, "command-ran.txt");
|
|
364
|
+
const result = await session.ask({ state: "x", paths: ["src/round.ts"], command: `touch ${marker}`, questions: { failure_kind: CHOICE } });
|
|
365
|
+
|
|
366
|
+
expect(result.text).toContain("Nothing was read, run, or sent");
|
|
367
|
+
expect(existsSync(marker)).toBe(false);
|
|
368
|
+
expect(session.complete).not.toHaveBeenCalled();
|
|
369
|
+
|
|
370
|
+
await session.ask({ state: "x", questions: { failure_kind: CHOICE } });
|
|
371
|
+
expect(session.notices).toHaveLength(1);
|
|
372
|
+
expect(session.notices[0]).toContain("/tokenin add");
|
|
373
|
+
});
|
|
374
|
+
|
|
375
|
+
it("drops ask_jev from the loadout while Jev is out of reach and restores it once it is back", async () => {
|
|
376
|
+
const session = harness({ credential: false });
|
|
377
|
+
await session.beforeRun();
|
|
378
|
+
expect(session.active).not.toContain("ask_jev");
|
|
379
|
+
// Hiding is silent: a user without a subscription is not nagged every session.
|
|
380
|
+
expect(session.notices).toEqual([]);
|
|
381
|
+
|
|
382
|
+
session.setCredential(true);
|
|
383
|
+
await session.beforeRun();
|
|
384
|
+
expect(session.active).toContain("ask_jev");
|
|
385
|
+
});
|
|
386
|
+
|
|
387
|
+
it("hides the tool when the route is turned off, and never re-adds a tool it did not remove", async () => {
|
|
388
|
+
const disabled = harness({ enabled: false });
|
|
389
|
+
await disabled.beforeRun();
|
|
390
|
+
expect(disabled.active).not.toContain("ask_jev");
|
|
391
|
+
|
|
392
|
+
const trimmed = harness();
|
|
393
|
+
trimmed.active.splice(trimmed.active.indexOf("ask_jev"), 1);
|
|
394
|
+
await trimmed.beforeRun();
|
|
395
|
+
expect(trimmed.active).not.toContain("ask_jev");
|
|
396
|
+
});
|
|
397
|
+
});
|
|
398
|
+
|
|
399
|
+
describe("missing material", () => {
|
|
400
|
+
it("leads with a warning when none of the passed files reached Jev, and says why each was skipped", async () => {
|
|
401
|
+
const session = harness();
|
|
402
|
+
session.complete.mockResolvedValue(jevResponse(answerText({ failure_kind: { choice: "bug_in_code", confidence: 0.7 } })));
|
|
403
|
+
const result = await session.ask({ state: "x", paths: ["src/missing.ts", "src"], questions: { failure_kind: CHOICE } });
|
|
404
|
+
|
|
405
|
+
const lines = result.text.split("\n");
|
|
406
|
+
expect(lines[1]).toContain("WARNING: Jev answered without any of the 2 files you passed");
|
|
407
|
+
expect(lines[1]).toContain("src/missing.ts (not found)");
|
|
408
|
+
expect(lines[1]).toContain("src (not a file)");
|
|
409
|
+
expect(lines[1]).toContain(repo);
|
|
410
|
+
});
|
|
411
|
+
|
|
412
|
+
it("names partly skipped files before the answers, without the all-missing warning", async () => {
|
|
413
|
+
const session = harness();
|
|
414
|
+
session.complete.mockResolvedValue(jevResponse(answerText({ failure_kind: { choice: "bug_in_code", confidence: 0.7 } })));
|
|
415
|
+
const result = await session.ask({ state: "x", paths: ["src/round.ts", "src/missing.ts"], questions: { failure_kind: CHOICE } });
|
|
416
|
+
|
|
417
|
+
expect(result.text.split("\n")[1]).toBe("Not sent to Jev: src/missing.ts (not found).");
|
|
418
|
+
expect(result.text).not.toContain("WARNING");
|
|
419
|
+
});
|
|
420
|
+
});
|
|
421
|
+
|
|
422
|
+
describe("jev_find", () => {
|
|
423
|
+
const find = async (session: AskHarness, params: Record<string, unknown>) => {
|
|
424
|
+
const result = await session.tools.jev_find.execute("call-f", params, undefined, undefined, session.ctx);
|
|
425
|
+
return { text: result.content[0]?.text ?? "", details: result.details };
|
|
426
|
+
};
|
|
427
|
+
|
|
428
|
+
beforeAll(() => {
|
|
429
|
+
writeFileSync(join(repo, "src", "other.ts"), "// mentions round only in passing\nexport const x = 1;\n", "utf-8");
|
|
430
|
+
writeFileSync(join(repo, ".env"), "round=SECRET_ENV_VALUE\n", "utf-8");
|
|
431
|
+
});
|
|
432
|
+
|
|
433
|
+
it("ranks ripgrep's candidates by Jev relevance and returns pointers, never contents", async () => {
|
|
434
|
+
const session = harness();
|
|
435
|
+
// Jev: the file whose question names round.ts is relevant, the rest are not.
|
|
436
|
+
session.complete.mockImplementation(async (_model: unknown, request: { messages: Array<{ content: string }> }) => {
|
|
437
|
+
const sent = JSON.parse(request.messages[0].content) as { questions: Record<string, { instructions: { question: string } }> };
|
|
438
|
+
const answers = Object.fromEntries(
|
|
439
|
+
Object.entries(sent.questions).map(([name, q]) => [name, { noul: q.instructions.question.includes("round.ts") ? 0.93 : 0.08 }]),
|
|
440
|
+
);
|
|
441
|
+
return jevResponse(answerText(answers));
|
|
442
|
+
});
|
|
443
|
+
const result = await find(session, { question: "where is rounding implemented", pattern: "round", ignoreCase: true });
|
|
444
|
+
|
|
445
|
+
expect(result.text).toMatch(/judged 2 of 2 candidate/);
|
|
446
|
+
expect(result.text).toContain("- src/round.ts:1 (0.93)");
|
|
447
|
+
expect(result.text).not.toContain("other.ts");
|
|
448
|
+
expect(result.text).not.toContain(FILE_SENTINEL);
|
|
449
|
+
const sent = session.sent() as { state: { files: Record<string, string> } };
|
|
450
|
+
expect(Object.keys(sent.state.files).sort()).toEqual(["src/other.ts", "src/round.ts"]);
|
|
451
|
+
expect(sent.state.files["src/round.ts"]).toContain(FILE_SENTINEL);
|
|
452
|
+
expect(JSON.stringify(sent)).not.toContain("SECRET_ENV_VALUE");
|
|
453
|
+
});
|
|
454
|
+
|
|
455
|
+
it("lists a directory by glob and still returns candidates when Jev is unreachable", async () => {
|
|
456
|
+
const session = harness({ credential: false });
|
|
457
|
+
const result = await find(session, { question: "rounding", path: "src", glob: "*.ts" });
|
|
458
|
+
|
|
459
|
+
expect(session.complete).not.toHaveBeenCalled();
|
|
460
|
+
expect(result.text).toMatch(/Jev did not judge \(no-credential\); 2 candidate/);
|
|
461
|
+
// The question word in the path orders the unjudged listing.
|
|
462
|
+
expect(result.text.split("\n")[1]).toBe("- src/round.ts");
|
|
463
|
+
});
|
|
464
|
+
|
|
465
|
+
it("shrinks escape-heavy snippets until a full batch fits the request budget", () => {
|
|
466
|
+
// Quotes and newlines double under JSON escaping, which is what overflowed live.
|
|
467
|
+
const snippet = '"\n\t\\'.repeat(1_000);
|
|
468
|
+
const batch = Array.from({ length: FIND_BATCH }, (_, i) => ({ path: `src/f${i}.ts`, lines: [], snippet }));
|
|
469
|
+
const payload = fitFindPayload("q", batch, 32 * 1024);
|
|
470
|
+
expect(serializeJevRequest(payload, 32 * 1024)).toBeDefined();
|
|
471
|
+
expect(Object.keys(payload.questions as object)).toHaveLength(FIND_BATCH);
|
|
472
|
+
});
|
|
473
|
+
|
|
474
|
+
it("refuses a search root outside the working directory", async () => {
|
|
475
|
+
const result = await find(harness(), { question: "x", path: ".." });
|
|
476
|
+
expect(result.text).toContain("outside the working directory");
|
|
477
|
+
});
|
|
478
|
+
});
|
|
479
|
+
|
|
480
|
+
describe("askPath symlinks", () => {
|
|
481
|
+
it("refuses links that escape the working directory or point at a secret", () => {
|
|
482
|
+
const root = mkdtempSync(join(tmpdir(), "ask-path-"));
|
|
483
|
+
const cwd = join(root, "repo");
|
|
484
|
+
mkdirSync(cwd);
|
|
485
|
+
writeFileSync(join(root, "outside.txt"), "x");
|
|
486
|
+
writeFileSync(join(cwd, ".env"), "SECRET=1");
|
|
487
|
+
writeFileSync(join(cwd, "ok.ts"), "x");
|
|
488
|
+
symlinkSync(join(root, "outside.txt"), join(cwd, "escape.txt"));
|
|
489
|
+
symlinkSync(root, join(cwd, "up"));
|
|
490
|
+
symlinkSync(join(cwd, ".env"), join(cwd, "notes.txt"));
|
|
491
|
+
try {
|
|
492
|
+
expect(askPath("escape.txt", cwd)).toEqual({ refused: "outside the working directory" });
|
|
493
|
+
expect(askPath("up/outside.txt", cwd)).toEqual({ refused: "outside the working directory" });
|
|
494
|
+
expect(askPath("notes.txt", cwd)).toEqual({ refused: "looks like a secret file" });
|
|
495
|
+
expect(askPath("ok.ts", cwd)).toEqual({ full: join(cwd, "ok.ts") });
|
|
496
|
+
expect(askPath("missing.ts", cwd)).toEqual({ full: join(cwd, "missing.ts") });
|
|
497
|
+
} finally {
|
|
498
|
+
rmSync(root, { recursive: true, force: true });
|
|
499
|
+
}
|
|
500
|
+
});
|
|
501
|
+
});
|