@selesai/code 0.13.33 → 0.13.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,587 @@
1
+ /**
2
+ * jev-ask-tool — the agent-callable `ask_jev` tool.
3
+ *
4
+ * The host does not decide when Jev runs here. The agent assembles one state
5
+ * from its own prose (`state`), code read from `paths`, and the output of one
6
+ * `command`, hands over a block of typed questions, and gets answers back. The
7
+ * file contents and command output it supplied are never returned, and only the
8
+ * answers reach the transcript.
9
+ *
10
+ * On by default; `jevAdvisory.routes.ask.enabled: false` turns it off. Files must resolve
11
+ * inside the working directory and obvious secret files are refused by name; the
12
+ * command runs through the same local shell operations as the bash tool, on the
13
+ * tool's own abort signal. Every part is clipped to the route's payload budget
14
+ * before the call, so an oversized request is trimmed rather than abstained.
15
+ */
16
+ import { closeSync, openSync, readSync, realpathSync, statSync } from "node:fs";
17
+ import { basename, isAbsolute, relative, resolve } from "node:path";
18
+ import { StringEnum } from "@earendil-works/pi-ai";
19
+ import {
20
+ createLocalBashOperations,
21
+ type ExtensionAPI,
22
+ type ExtensionContext,
23
+ getSettingsPath,
24
+ } from "@selesai/code";
25
+ import { Type } from "typebox";
26
+ import {
27
+ askJevAnswers,
28
+ confidenceBucket,
29
+ emitJevTelemetry,
30
+ jevConnection,
31
+ jevUnavailable,
32
+ readJevAdvisoryConfig,
33
+ serializeJevRequest,
34
+ UNTRUSTED_MATERIAL_FOCUS,
35
+ warnJevUnavailableOnce,
36
+ type JevAbstainReason,
37
+ } from "./jev/decisions.ts";
38
+
39
+ /** The tool's registered name. */
40
+ export const ASK_JEV_TOOL = "ask_jev";
41
+
42
+ /** Question types the tool accepts, one per Jev primitive. */
43
+ export const ASK_JEV_QUESTION_TYPES = ["choice", "score", "noul"] as const;
44
+ export type AskJevQuestionType = (typeof ASK_JEV_QUESTION_TYPES)[number];
45
+
46
+ /** Ceilings applied before anything leaves the process. */
47
+ export const MAX_ASK_QUESTIONS = 8;
48
+ export const MAX_ASK_PATHS = 8;
49
+ export const MAX_ASK_STATE_CHARS = 4_000;
50
+ export const MAX_ASK_FILE_BYTES = 16 * 1024;
51
+ export const MAX_ASK_COMMAND_BYTES = 16 * 1024;
52
+ export const ASK_COMMAND_TIMEOUT_SECONDS = 60;
53
+ /** Fixed JSON scaffolding (`{"state":{"files":{},"command":{}}}` and friends). */
54
+ const ASK_ENVELOPE_BYTES = 512;
55
+
56
+ /** Marks code Jev is shown as incomplete, so it judges a clipped file as clipped. */
57
+ const TRUNCATION_MARKER = "\n… [truncated]";
58
+
59
+ /** File names never sent to Jev, whatever the agent asks for. */
60
+ const SECRET_BASENAME =
61
+ /^(?:\.env(?:\..+)?|\.npmrc|\.netrc|auth\.json|credentials(?:\..+)?|id_(?:rsa|dsa|ecdsa|ed25519)(?:\.pub)?|.+\.(?:pem|key|p12|pfx|keystore))$/i;
62
+
63
+ function isRecord(value: unknown): value is Record<string, unknown> {
64
+ return typeof value === "object" && value !== null && !Array.isArray(value);
65
+ }
66
+
67
+ // ---------------------------------------------------------------------------
68
+ // Questions
69
+ // ---------------------------------------------------------------------------
70
+
71
+ export interface AskJevQuestion {
72
+ name: string;
73
+ type: AskJevQuestionType;
74
+ instructions: string;
75
+ criteria?: unknown;
76
+ focus?: string;
77
+ }
78
+
79
+ export interface ParsedAskQuestions {
80
+ questions: AskJevQuestion[];
81
+ /** Set when the block cannot be asked; `questions` is empty then. */
82
+ error?: string;
83
+ }
84
+
85
+ /** Validate the agent's question block, or explain why it cannot be asked. */
86
+ export function parseAskQuestions(raw: Record<string, unknown> | undefined): ParsedAskQuestions {
87
+ const entries = Object.entries(raw ?? {});
88
+ if (entries.length === 0) return { questions: [], error: "At least one question is required." };
89
+ if (entries.length > MAX_ASK_QUESTIONS) {
90
+ return { questions: [], error: `At most ${MAX_ASK_QUESTIONS} questions per call; got ${entries.length}.` };
91
+ }
92
+ const questions: AskJevQuestion[] = [];
93
+ for (const [name, value] of entries) {
94
+ const spec = isRecord(value) ? value : {};
95
+ const rawType = spec.type;
96
+ if (typeof rawType !== "string" || !ASK_JEV_QUESTION_TYPES.includes(rawType as AskJevQuestionType)) {
97
+ return {
98
+ questions: [],
99
+ error: `Question "${name}" has type ${JSON.stringify(rawType)}; use one of ${ASK_JEV_QUESTION_TYPES.join(", ")}.`,
100
+ };
101
+ }
102
+ const instructions = spec.instructions;
103
+ if (typeof instructions !== "string" || instructions.trim() === "") {
104
+ return { questions: [], error: `Question "${name}" needs non-empty instructions.` };
105
+ }
106
+ const criteria = spec.criteria;
107
+ if (rawType === "choice" && (!isRecord(criteria) || Object.keys(criteria).length < 2)) {
108
+ return {
109
+ questions: [],
110
+ error: `Choice question "${name}" needs criteria as an object of at least two options, e.g. {"option": "what the option means"}.`,
111
+ };
112
+ }
113
+ if (rawType === "score" && (!Array.isArray(criteria) || criteria.length < 2)) {
114
+ return {
115
+ questions: [],
116
+ error: `Score question "${name}" needs criteria as an ordered list of at least two level labels.`,
117
+ };
118
+ }
119
+ if (rawType === "noul" && criteria !== undefined && !isRecord(criteria)) {
120
+ return {
121
+ questions: [],
122
+ error: `Noul question "${name}" needs criteria as {"true": "...", "false": "..."}.`,
123
+ };
124
+ }
125
+ questions.push({
126
+ name,
127
+ type: rawType as AskJevQuestionType,
128
+ instructions: instructions.trim(),
129
+ ...(criteria === undefined ? {} : { criteria }),
130
+ ...(typeof spec.focus === "string" && spec.focus.trim() !== "" ? { focus: spec.focus.trim() } : {}),
131
+ });
132
+ }
133
+ return { questions };
134
+ }
135
+
136
+ /**
137
+ * The decisions request: one assembled state and one question per name. Every
138
+ * criterion and every state part stays untrusted material to judge, never an
139
+ * instruction, exactly as [`buildJevPayload`] frames it.
140
+ */
141
+ export function buildAskPayload(
142
+ questions: readonly AskJevQuestion[],
143
+ state: Record<string, unknown>,
144
+ ): Record<string, unknown> {
145
+ return {
146
+ state,
147
+ questions: Object.fromEntries(
148
+ questions.map((question) => [
149
+ question.name,
150
+ {
151
+ type: question.type,
152
+ instructions: {
153
+ question: question.instructions,
154
+ focus: question.focus ? `${question.focus} ${UNTRUSTED_MATERIAL_FOCUS}` : UNTRUSTED_MATERIAL_FOCUS,
155
+ },
156
+ ...(question.criteria === undefined ? {} : { criteria: question.criteria }),
157
+ },
158
+ ]),
159
+ ),
160
+ };
161
+ }
162
+
163
+ // ---------------------------------------------------------------------------
164
+ // Bounded state
165
+ // ---------------------------------------------------------------------------
166
+
167
+ /** Per-part byte caps that together stay under the route's request budget. */
168
+ export function askPartBudget(
169
+ payloadBytes: number,
170
+ fileCount: number,
171
+ questionBytes: number,
172
+ ): { stateChars: number; commandBytes: number; fileBytes: number } {
173
+ const available = Math.max(0, payloadBytes - questionBytes - ASK_ENVELOPE_BYTES);
174
+ const stateChars = Math.min(MAX_ASK_STATE_CHARS, Math.floor(available * 0.2));
175
+ const commandBytes = Math.min(MAX_ASK_COMMAND_BYTES, Math.floor(available * 0.4));
176
+ const forFiles = Math.max(0, available - stateChars - commandBytes);
177
+ return {
178
+ stateChars,
179
+ commandBytes,
180
+ fileBytes: fileCount === 0 ? 0 : Math.min(MAX_ASK_FILE_BYTES, Math.floor(forFiles / fileCount)),
181
+ };
182
+ }
183
+
184
+ /** Clip text to a UTF-8 byte budget without splitting a character. */
185
+ function clip(text: string, maxBytes: number): { text: string; clipped: boolean } {
186
+ if (maxBytes <= 0) return { text: "", clipped: text !== "" };
187
+ if (Buffer.byteLength(text, "utf-8") <= maxBytes) return { text, clipped: false };
188
+ let end = Math.min(text.length, maxBytes);
189
+ while (end > 0 && Buffer.byteLength(text.slice(0, end), "utf-8") > maxBytes) end -= 1;
190
+ return { text: text.slice(0, end), clipped: true };
191
+ }
192
+
193
+ /** A path the tool may read: inside the working directory, and not a known secret file. */
194
+ export function askPath(raw: string, cwd: string): { full: string } | { refused: string } {
195
+ const full = isAbsolute(raw) ? resolve(raw) : resolve(cwd, raw);
196
+ // Judge the symlink-resolved target too: a link inside cwd may point anywhere.
197
+ // A missing path has no target to leak; the read reports "not found".
198
+ const checks: Array<[string, string]> = [[full, resolve(cwd)]];
199
+ try {
200
+ checks.push([realpathSync(full), realpathSync(cwd)]);
201
+ } catch {}
202
+ for (const [path, root] of checks) {
203
+ const inside = relative(root, path);
204
+ if (inside === "" || inside.startsWith("..") || isAbsolute(inside)) {
205
+ return { refused: "outside the working directory" };
206
+ }
207
+ if (SECRET_BASENAME.test(basename(path))) return { refused: "looks like a secret file" };
208
+ }
209
+ return { full };
210
+ }
211
+
212
+ /** Read at most `maxBytes` of a file without loading the rest of it, or say why it could not be read. */
213
+ function readBounded(path: string, maxBytes: number): { text: string; clipped: boolean } | { error: string } {
214
+ let size: number;
215
+ try {
216
+ const stats = statSync(path);
217
+ if (!stats.isFile()) return { error: "not a file" };
218
+ size = stats.size;
219
+ } catch (error) {
220
+ return { error: isRecord(error) && error.code === "ENOENT" ? "not found" : "unreadable" };
221
+ }
222
+ const length = Math.max(0, Math.min(size, maxBytes));
223
+ const buffer = Buffer.alloc(length);
224
+ try {
225
+ const fd = openSync(path, "r");
226
+ try {
227
+ const read = readSync(fd, buffer, 0, length, 0);
228
+ return { text: buffer.subarray(0, read).toString("utf-8"), clipped: size > read };
229
+ } finally {
230
+ closeSync(fd);
231
+ }
232
+ } catch {
233
+ return { error: "unreadable" };
234
+ }
235
+ }
236
+
237
+ /** One shell command through the same local operations the bash tool uses. */
238
+ export async function runAskCommand(
239
+ command: string,
240
+ cwd: string,
241
+ signal: AbortSignal | undefined,
242
+ ): Promise<{ output: string; exitCode: number | null; clipped: boolean; failed: boolean }> {
243
+ const chunks: string[] = [];
244
+ let bytes = 0;
245
+ let clipped = false;
246
+ let failed = false;
247
+ let exitCode: number | null = null;
248
+ try {
249
+ const code = await createLocalBashOperations().exec(command, cwd, {
250
+ onData: (data) => {
251
+ if (clipped) return;
252
+ const text = data.toString("utf-8");
253
+ chunks.push(text);
254
+ bytes += Buffer.byteLength(text, "utf-8");
255
+ if (bytes >= MAX_ASK_COMMAND_BYTES) clipped = true;
256
+ },
257
+ signal,
258
+ timeout: ASK_COMMAND_TIMEOUT_SECONDS,
259
+ });
260
+ exitCode = code.exitCode;
261
+ } catch {
262
+ // A failed, timed-out, or cancelled command still reports what it printed.
263
+ failed = true;
264
+ }
265
+ return { output: chunks.join(""), exitCode, clipped, failed };
266
+ }
267
+
268
+ // ---------------------------------------------------------------------------
269
+ // Answers
270
+ // ---------------------------------------------------------------------------
271
+
272
+ function formatValue(value: unknown): string {
273
+ if (typeof value === "number") return Number.isInteger(value) ? String(value) : value.toFixed(2);
274
+ if (typeof value === "string") return value;
275
+ return JSON.stringify(value);
276
+ }
277
+
278
+ /** A score answer as the agent reads it: the nearest level label, then the raw position. */
279
+ export function formatScore(answer: Record<string, unknown>, levels: unknown): string | undefined {
280
+ const score = answer.score;
281
+ if (typeof score !== "number") return undefined;
282
+ const legend = isRecord(answer.legend) ? answer.legend : undefined;
283
+ const nearest = Math.round(score);
284
+ const label = legend?.[String(nearest)] ?? (Array.isArray(levels) ? levels[nearest] : undefined);
285
+ return typeof label === "string" ? `${label} (${formatValue(score)})` : formatValue(score);
286
+ }
287
+
288
+ /**
289
+ * Render answers for the agent: each value and its confidence, never the raw
290
+ * probability distribution, the file contents, or the command output.
291
+ */
292
+ export function renderAskAnswers(input: {
293
+ questions: readonly AskJevQuestion[];
294
+ answers: Record<string, unknown>;
295
+ rejected: Record<string, JevAbstainReason>;
296
+ failure?: JevAbstainReason;
297
+ model: string;
298
+ elapsedMs: number;
299
+ refused: readonly string[];
300
+ notes: readonly string[];
301
+ /** How many paths the agent passed, and the directory they resolve against. */
302
+ pathsRequested?: number;
303
+ cwd?: string;
304
+ }): string {
305
+ const lines: string[] = [];
306
+ let answered = 0;
307
+ for (const question of input.questions) {
308
+ const answer = input.answers[question.name];
309
+ if (!isRecord(answer)) {
310
+ lines.push(`- ${question.name}: no answer (${input.rejected[question.name] ?? input.failure ?? "missing"})`);
311
+ continue;
312
+ }
313
+ answered += 1;
314
+ // The type is already in the label, the legend is folded into the score, and raw
315
+ // distributions stay out of the agent's context.
316
+ const fields = Object.entries(answer)
317
+ .filter(([key]) => !["probabilities", "type", "legend"].includes(key))
318
+ .map(([key, value]) =>
319
+ key === "score" ? `score=${formatScore(answer, question.criteria)}` : `${key}=${formatValue(value)}`,
320
+ );
321
+ lines.push(`- ${question.name} (${question.type}): ${fields.length > 0 ? fields.join(", ") : "answered"}`);
322
+ }
323
+ const header = [`ask_jev: ${answered}/${input.questions.length} answered via ${input.model} in ${input.elapsedMs}ms`];
324
+ if (answered === 0 && input.failure) header.push(`Jev was unavailable or abstained: ${input.failure}.`);
325
+ // Missing material changes what the answers mean, so it leads rather than trails them.
326
+ const requested = input.pathsRequested ?? 0;
327
+ if (requested > 0 && input.refused.length >= requested) {
328
+ header.push(
329
+ `WARNING: Jev answered without any of the ${requested} files you passed (${input.refused.join("; ")}). ` +
330
+ `Paths resolve inside ${input.cwd ?? "the working directory"}; treat these answers as judged from your prose alone.`,
331
+ );
332
+ } else if (input.refused.length > 0) {
333
+ header.push(`Not sent to Jev: ${input.refused.join("; ")}.`);
334
+ }
335
+ const footer = [
336
+ "Only these answers are returned: the file contents and command output you supplied were not sent back to you.",
337
+ ...(input.notes.length > 0 ? [`Truncated to fit the request budget: ${input.notes.join("; ")}.`] : []),
338
+ ];
339
+ return [...header, ...lines, ...footer].join("\n");
340
+ }
341
+
342
+ // ---------------------------------------------------------------------------
343
+ // Tool
344
+ // ---------------------------------------------------------------------------
345
+
346
+ const QuestionSpec = Type.Object({
347
+ type: StringEnum(ASK_JEV_QUESTION_TYPES, {
348
+ description: "choice: pick one option. score: place the state on a rubric. noul: probability a statement holds.",
349
+ }),
350
+ instructions: Type.String({ description: "The question, stated in one sentence." }),
351
+ criteria: Type.Optional(
352
+ Type.Unknown({
353
+ description:
354
+ 'choice: {"option": "what the option means"} (at least two). score: an ordered array of level labels. noul: {"true": "...", "false": "..."}.',
355
+ }),
356
+ ),
357
+ focus: Type.Optional(Type.String({ description: "Extra judging guidance for this question." })),
358
+ });
359
+
360
+ const AskJevParams = Type.Object({
361
+ state: Type.Optional(Type.String({ description: "Your own prose: the situation and what you are deciding." })),
362
+ paths: Type.Optional(
363
+ Type.Array(Type.String(), {
364
+ description: `Code to include as state, read-only (max ${MAX_ASK_PATHS}). Relative paths resolve against the working directory; paths outside it are refused.`,
365
+ }),
366
+ ),
367
+ command: Type.Optional(Type.String({ description: "One shell command whose output becomes state." })),
368
+ questions: Type.Record(Type.String(), QuestionSpec, { description: "Question name -> spec." }),
369
+ });
370
+
371
+ export default function jevAskToolExtension(pi: ExtensionAPI): void {
372
+ pi.registerTool({
373
+ name: ASK_JEV_TOOL,
374
+ label: "Ask Jev",
375
+ description: [
376
+ "Ask Jev (a typed decisions model, not a chat model) a block of questions about a state you assemble here:",
377
+ "your own prose in `state`, code read from `paths`, and the output of one shell `command`. Code makes one",
378
+ "bounded call and returns answers only — the file contents and command output are never sent back to you.",
379
+ "Anything you put in the state is sent to Jev, so pass only code you are willing to share.",
380
+ '`questions` is an object keyed by question name; each value is {"type": "choice" | "score" | "noul",',
381
+ '"instructions": <the question>, "criteria": <see below>, "focus"?: <extra guidance>}.',
382
+ 'choice criteria: {"option": "what the option means"}; score criteria: an ordered array of level labels;',
383
+ 'noul criteria: {"true": "...", "false": "..."} (optional for noul).',
384
+ "Each answer returns the chosen value, score, or probability plus a confidence; raw probability",
385
+ "distributions are omitted.",
386
+ "Use it to triage a failure before touching anything (run the failing command, ask what kind of failure it is),",
387
+ "judge a diff before shipping (a risk score and a needs-review flag), classify a request before planning, or",
388
+ "find which of several files matters without reading them into your own context.",
389
+ `Bounds: ${MAX_ASK_QUESTIONS} questions, ${MAX_ASK_PATHS} paths per call; each file is clipped to about ${MAX_ASK_FILE_BYTES / 1024} KB and marked when clipped.`,
390
+ ].join(" "),
391
+ promptSnippet: "Ask Jev a typed question about state you assemble from prose, files, or one command",
392
+ promptGuidelines: [
393
+ "Use ask_jev whenever a judgment would otherwise be a guess — what kind of failure this is, how risky a diff is, what a request is really asking for. It returns typed answers with confidences for a fraction of a cent, so it validates assumptions before they cost a wrong edit.",
394
+ "Prefer ask_jev over reading files when you only need a verdict about them (which of these files handles X, does this code already do Y, is this test output a code bug or a flaky test): pass them as `paths` or the command as `command`, and only the answer enters your context. Read a file yourself when you need its exact text, such as before editing it.",
395
+ ],
396
+ discovery: {
397
+ summary: "Ask Jev one block of typed choice/score/noul questions over prose, code, and one command's output",
398
+ aliases: ["jev", "triage", "classify", "judge", "score"],
399
+ category: "Decisions",
400
+ },
401
+ parameters: AskJevParams,
402
+
403
+ async execute(_toolCallId, params, signal, _onUpdate, ctx: ExtensionContext) {
404
+ const text = (value: string) => ({ content: [{ type: "text" as const, text: value }], details: {} });
405
+
406
+ const config = readJevAdvisoryConfig(getSettingsPath());
407
+ const route = config.routes.ask;
408
+ if (!route.enabled) {
409
+ return text(
410
+ `ask_jev is disabled (jevAdvisory.routes.ask.enabled is false in ${getSettingsPath()}). Remove that setting or set it to true.`,
411
+ );
412
+ }
413
+
414
+ const parsed = parseAskQuestions(params.questions as Record<string, unknown> | undefined);
415
+ if (parsed.error !== undefined) {
416
+ return text(`ask_jev needs a usable question block: ${parsed.error}`);
417
+ }
418
+ const questions = parsed.questions;
419
+
420
+ // Reading files and running the command exist only to feed Jev: without a reachable
421
+ // Jev, do neither (an `npm test` would otherwise run in full and be thrown away).
422
+ const connection = jevConnection(config, route);
423
+ const unreachable = await jevUnavailable(ctx, connection);
424
+ if (unreachable) {
425
+ if (ctx.hasUI) warnJevUnavailableOnce(ctx.ui, config.provider);
426
+ emitJevTelemetry(pi.events, "decision", {
427
+ route: "ask",
428
+ outcome: "fallback",
429
+ candidates: questions.length,
430
+ confidence: confidenceBucket(undefined),
431
+ elapsedMs: 0,
432
+ reason: unreachable,
433
+ });
434
+ return {
435
+ content: [
436
+ {
437
+ type: "text" as const,
438
+ text: `${renderAskAnswers({
439
+ questions,
440
+ answers: {},
441
+ rejected: {},
442
+ failure: unreachable,
443
+ model: config.model,
444
+ elapsedMs: 0,
445
+ refused: [],
446
+ notes: [],
447
+ })}\nNothing was read, run, or sent. Decide without Jev: read the files or run the command yourself.`,
448
+ },
449
+ ],
450
+ details: { answered: 0, questions: questions.length, model: config.model, elapsedMs: 0, failure: unreachable },
451
+ };
452
+ }
453
+
454
+ const questionBytes = Buffer.byteLength(
455
+ JSON.stringify(buildAskPayload(questions, {}).questions),
456
+ "utf-8",
457
+ );
458
+ const requested = (params.paths ?? []).slice(0, MAX_ASK_PATHS);
459
+ const budget = askPartBudget(route.payloadBytes, requested.length, questionBytes);
460
+
461
+ const refused: string[] = [];
462
+ const notes: string[] = [];
463
+
464
+ const prose = clip((params.state ?? "").trim(), budget.stateChars);
465
+ if (prose.clipped) notes.push("state");
466
+
467
+ const files: Record<string, string> = {};
468
+ for (const raw of requested) {
469
+ const path = askPath(raw, ctx.cwd);
470
+ if ("refused" in path) {
471
+ refused.push(`${raw} (${path.refused})`);
472
+ continue;
473
+ }
474
+ const file = readBounded(path.full, Math.max(0, budget.fileBytes - Buffer.byteLength(TRUNCATION_MARKER, "utf-8")));
475
+ if ("error" in file) {
476
+ refused.push(`${raw} (${file.error})`);
477
+ continue;
478
+ }
479
+ if (file.clipped) notes.push(raw);
480
+ files[relative(ctx.cwd, path.full)] = file.clipped ? `${file.text}${TRUNCATION_MARKER}` : file.text;
481
+ }
482
+
483
+ let command: Record<string, unknown> | undefined;
484
+ if (params.command?.trim()) {
485
+ const result = await runAskCommand(params.command.trim(), ctx.cwd, signal);
486
+ const output = clip(result.output, budget.commandBytes);
487
+ if (output.clipped || result.clipped) notes.push("command output");
488
+ command = {
489
+ command: params.command.trim(),
490
+ output: output.text,
491
+ exitCode: result.exitCode,
492
+ ...(result.failed ? { failed: true } : {}),
493
+ };
494
+ }
495
+
496
+ const payload = buildAskPayload(questions, {
497
+ ...(prose.text ? { request: prose.text } : {}),
498
+ ...(Object.keys(files).length > 0 ? { files } : {}),
499
+ ...(command ? { command } : {}),
500
+ });
501
+ if (serializeJevRequest(payload, route.payloadBytes) === undefined) {
502
+ return text("ask_jev could not fit this request in its budget; pass fewer paths or a shorter command.");
503
+ }
504
+
505
+ const result = await askJevAnswers(ctx, connection, {
506
+ payload,
507
+ maxBytes: route.payloadBytes,
508
+ });
509
+ const answers = result.answers ?? {};
510
+ const rejected: Record<string, JevAbstainReason> = {};
511
+ let answered = 0;
512
+ let topConfidence: number | undefined;
513
+ for (const question of questions) {
514
+ const answer = answers[question.name];
515
+ if (!isRecord(answer)) {
516
+ rejected[question.name] = result.failure ?? "missing";
517
+ continue;
518
+ }
519
+ answered += 1;
520
+ if (typeof answer.confidence === "number") {
521
+ topConfidence = topConfidence === undefined ? answer.confidence : Math.max(topConfidence, answer.confidence);
522
+ }
523
+ }
524
+
525
+ // Content-free: no state, answers, or criteria travel, only shape and outcome.
526
+ emitJevTelemetry(pi.events, "decision", {
527
+ route: "ask",
528
+ outcome: answered > 0 ? "jev" : "fallback",
529
+ candidates: questions.length,
530
+ confidence: confidenceBucket(topConfidence),
531
+ elapsedMs: result.elapsedMs,
532
+ ...(answered > 0 ? {} : { reason: result.failure ?? "missing" }),
533
+ });
534
+
535
+ return {
536
+ content: [
537
+ {
538
+ type: "text" as const,
539
+ text: renderAskAnswers({
540
+ questions,
541
+ answers,
542
+ rejected,
543
+ failure: result.failure,
544
+ model: config.model,
545
+ elapsedMs: result.elapsedMs,
546
+ refused,
547
+ notes,
548
+ pathsRequested: requested.length,
549
+ cwd: ctx.cwd,
550
+ }),
551
+ },
552
+ ],
553
+ // Deliberately shape only: tool details are persisted in the session.
554
+ details: {
555
+ answered,
556
+ questions: questions.length,
557
+ model: config.model,
558
+ elapsedMs: result.elapsedMs,
559
+ ...(result.failure ? { failure: result.failure } : {}),
560
+ },
561
+ };
562
+ },
563
+ });
564
+
565
+ // Offer the tool only while Jev can answer: a tool that always fails costs the agent a turn
566
+ // every time it reaches for it. Checked before every run, so `/tokenin add` brings it back
567
+ // without a reload. Silent on purpose: the one warning comes from whichever Jev path first
568
+ // needs the missing credential, not from every session of a user without a subscription.
569
+ // Only a removal made here is ever undone, so a loadout the user trimmed stays trimmed.
570
+ let hidden = false;
571
+ pi.on("before_agent_start", async (_event, ctx) => {
572
+ const config = readJevAdvisoryConfig(getSettingsPath());
573
+ const route = config.routes.ask;
574
+ const offline = !route.enabled || (await jevUnavailable(ctx, jevConnection(config, route))) !== undefined;
575
+ const active = pi.getActiveTools();
576
+ if (offline) {
577
+ if (active.includes(ASK_JEV_TOOL)) {
578
+ pi.setActiveTools(active.filter((name) => name !== ASK_JEV_TOOL));
579
+ hidden = true;
580
+ }
581
+ } else if (hidden) {
582
+ if (!active.includes(ASK_JEV_TOOL)) pi.setActiveTools([...active, ASK_JEV_TOOL]);
583
+ hidden = false;
584
+ }
585
+ return undefined;
586
+ });
587
+ }
@@ -19,6 +19,7 @@
19
19
  "./herdr-agent-state.ts",
20
20
  "./inline-skills.ts",
21
21
  "./jev-advisory-routing.ts",
22
+ "./jev-ask-tool.ts",
22
23
  "./rtk.ts",
23
24
  "./tokenin-onboarding.ts",
24
25
  "./pi-graft",
@@ -483,45 +483,19 @@ Move behavior:
483
483
 
484
484
  ## Configuration
485
485
 
486
- Create `~/.selesai/agent/hermes-memory-config.json`:
486
+ Set extension options under `hermesMemory` in `~/.selesai/agent/settings.json`:
487
487
 
488
488
  ```json
489
489
  {
490
- "memoryMode": "policy-only",
491
- "memoryPolicyStyle": "full",
492
- "memoryCharLimit": 5000,
493
- "userCharLimit": 5000,
494
- "projectCharLimit": 5000,
495
- "memoryDir": "~/.selesai/agent/pi-hermes-memory",
496
- "projectsMemoryDir": "projects-memory",
497
- "sessionSearch": { "variant": "legacy" },
498
- "sessionRetentionDays": 0,
499
- "quickCheckOnOpen": true,
500
- "llmModelOverride": "openrouter/deepseek/deepseek-v4-flash",
501
- "llmThinkingOverride": "off",
502
- "childExtensionPaths": ["~/.selesai/agent/git/github.com/example/custom-provider-extension/index.ts"],
503
- "nudgeInterval": 10,
504
- "nudgeToolCalls": 15,
505
- "reviewRecentMessages": 0,
506
- "reviewEnabled": true,
507
- "reviewTransport": "direct",
508
- "memoryOverflowStrategy": "auto-consolidate",
509
- "autoConsolidate": true,
510
- "correctionDetection": true,
511
- "failureInjectionEnabled": true,
512
- "failureInjectionMaxAgeDays": 7,
513
- "failureInjectionMaxEntries": 5,
514
- "consolidationTimeoutMs": 180000,
515
- "overflowGraceMs": 180000,
516
- "autoConsolidationWarnOnFailure": true,
517
- "flushOnCompact": true,
518
- "flushOnShutdown": true,
519
- "flushMinTurns": 6,
520
- "flushRecentMessages": 0,
521
- "standingInstructionsEnabled": true
490
+ "hermesMemory": {
491
+ "llmThinkingOverride": "off",
492
+ "consolidationTimeoutMs": 300000
493
+ }
522
494
  }
523
495
  ```
524
496
 
497
+ The extension reads these options at startup. Values in `settings.json` override the legacy `hermes-memory-config.json`, which remains a fallback for existing installations. All settings below are keys inside `hermesMemory`. `llmModelOverride` is optional and uses `provider/model` format (for example, `tokenin/deepseek-v4.1-flash` if your account has access); leave it unset to use the active model.
498
+
525
499
  | Setting | Default | Description |
526
500
  |---|---|---|
527
501
  | `memoryMode` | `policy-only` | Prompt behavior: `policy-only` injects only memory policy; `legacy-inject` restores full memory prompt injection |
@@ -536,7 +510,7 @@ Create `~/.selesai/agent/hermes-memory-config.json`:
536
510
  | `sessionSearch` | `{ "variant": "legacy" }` | Session search implementation: `legacy` keeps the existing SQLite/FTS snippet search; `anchors` uses the opt-in Markdown request surface and returns compact JSONL line-range anchors from `~/.selesai/agent/sessions/` |
537
511
  | `sessionRetentionDays` | `0` | Opt-in SQLite session retention, in days. `0` (default) disables pruning entirely and keeps the legacy count-only backfill preflight. When positive, sessions whose JSONL source file was last modified longer ago than the window are pruned from SQLite at startup — **rows only; the JSONL files in `~/.selesai/agent/sessions/` are never deleted** — and both the deferred backfill and `/memory-index-sessions` skip files outside the window, so pruned sessions stay pruned instead of being re-indexed |
538
512
  | `quickCheckOnOpen` | `true` | Run a full SQLite integrity check asynchronously after opening the database; set to `false` to skip the startup scan (operation-time recovery remains enabled) |
539
- | `llmModelOverride` | unset | Optional model override for background review (direct and subprocess), correction save, session flush, and consolidation |
513
+ | `llmModelOverride` | unset | Optional `provider/model` override for background review (direct and subprocess), correction save, session flush, and consolidation; unset uses the active model |
540
514
  | `llmThinkingOverride` | unset | Optional thinking override for those LLM calls; valid values are `off`, `minimal`, `low`, `medium`, `high`, and `xhigh`. If `llmModelOverride` is set and this is omitted, review/child calls default to `off` |
541
515
  | `childExtensionPaths` | unset | Trusted provider/auth extension sources explicitly allowed in isolated child Pi processes. Values are passed to Pi's standard `-e` resolver, so absolute paths, `~/...`, paths relative to the child working directory, and `git:`/`npm:` package sources are supported. Sibling packages matching the `*-oauth-adapter`/`*-auth-adapter` naming convention (including scoped packages, via their `package.json` `pi.extensions` manifest) are detected automatically. This setting is only needed for custom providers or adapters that are not detected. In-process direct transport (the default for review/flush/correction/consolidation) doesn't need it, since it reads whatever provider auth is already registered. |
542
516
  | `nudgeInterval` | `10` | Turns between auto-reviews |
@@ -546,7 +520,7 @@ Create `~/.selesai/agent/hermes-memory-config.json`:
546
520
  | `reviewTransport` | `direct` | LLM transport for background review, session flush, correction save, and manual consolidation: `direct` uses in-process `completeSimple()` with subprocess fallback; `subprocess` forces legacy `pi -p` only |
547
521
  | `memoryOverflowStrategy` | `auto-consolidate` | Legacy-inject behavior when a Markdown memory file reaches its character limit: `auto-consolidate` runs the existing consolidation flow; `reject` returns an error; `fifo-evict` rotates older entries in file order until the new entry fits |
548
522
  | `autoConsolidate` | `true` | Legacy alias for `memoryOverflowStrategy` when `memoryOverflowStrategy` is not set (`true` = `auto-consolidate`, `false` = `reject`) |
549
- | `consolidationTimeoutMs` | `180000` | Maximum time in milliseconds for a consolidation run (auto and `/memory-consolidate` alike). Configured values are used verbatim; a consolidation pays child-process boot plus a full LLM turn, so values below the default are frequently killed mid-run and log a warning at startup |
523
+ | `consolidationTimeoutMs` | `300000` | Maximum time in milliseconds for a consolidation run (auto and `/memory-consolidate` alike). Configured values are used verbatim; a consolidation pays child-process boot plus a full LLM turn, so values below the default are frequently killed mid-run and log a warning at startup |
550
524
  | `overflowGraceMs` | `180000` | Wall-clock grace period after a memory overflow before automatic consolidation is retried; this gives the active agent time to consolidate manually. Set to `0` to disable the grace period |
551
525
  | `autoConsolidationWarnOnFailure` | `true` | Log failed automatic consolidation attempts to the session console. Set to `false` to suppress only this warning; the memory tool result still reports the failure reason |
552
526
  | `correctionDetection` | `true` | Detect user corrections and save immediately |
@@ -602,7 +576,8 @@ The deferred backfill, live-index, and integrity-check spans may appear after st
602
576
  │ │ └── SKILL.md
603
577
  │ └── another-project/
604
578
  │ └── MEMORY.md
605
- ├── hermes-memory-config.json
579
+ ├── settings.json ← Global settings; extension options are under hermesMemory
580
+ ├── hermes-memory-config.json ← Legacy fallback for older extension builds
606
581
  └── ...
607
582
  ```
608
583