specpi 0.26.0 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +37 -3
  3. package/SECURITY_MODEL.md +44 -4
  4. package/THIRD_PARTY.md +9 -1
  5. package/extensions/jev-advisor/broker.mjs +277 -0
  6. package/extensions/jev-advisor/client.mjs +172 -0
  7. package/extensions/jev-advisor/config.mjs +270 -0
  8. package/extensions/jev-advisor/consent.mjs +133 -0
  9. package/extensions/jev-advisor/gate.mjs +263 -0
  10. package/extensions/jev-advisor/index.ts +999 -0
  11. package/extensions/jev-advisor/key-source.mjs +252 -0
  12. package/extensions/jev-advisor/layer.mjs +169 -0
  13. package/extensions/jev-advisor/ledger.mjs +138 -0
  14. package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
  15. package/extensions/jev-advisor/questions/compaction.mjs +153 -0
  16. package/extensions/jev-advisor/questions/gap.mjs +140 -0
  17. package/extensions/jev-advisor/questions/guard.mjs +168 -0
  18. package/extensions/jev-advisor/questions/progress.mjs +195 -0
  19. package/extensions/jev-advisor/questions/retention.mjs +188 -0
  20. package/extensions/jev-advisor/questions/sources.mjs +91 -0
  21. package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
  22. package/extensions/jev-advisor/risk.mjs +442 -0
  23. package/extensions/jev-advisor/sanitize.mjs +0 -0
  24. package/extensions/jev-advisor/usage.mjs +92 -0
  25. package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
  26. package/extensions/tool-wishlist/index.ts +11 -0
  27. package/extensions/workflow-controls/capabilities.mjs +26 -0
  28. package/extensions/workflow-controls/index.ts +2 -2
  29. package/package.json +1 -1
  30. package/scripts/packages.mjs +56 -0
  31. package/scripts/specpi.mjs +73 -4
@@ -0,0 +1,999 @@
1
+ import fs from "node:fs";
2
+ import { type ExtensionAPI, type ExtensionContext } from "@earendil-works/pi-coding-agent";
3
+ import { SYSTEM_NAMES, loadSettings, saveSettings, settingsPath } from "./config.mjs";
4
+ import { keySources } from "./key-source.mjs";
5
+ import { applyLayer, guardWarning, layerScopeLine, layerToPersist, startupToPersist } from "./layer.mjs";
6
+ import { consentPath, granted, revokeConsent } from "./consent.mjs";
7
+ import { createBroker } from "./broker.mjs";
8
+ import { ledgerPath, read as readLedger } from "./ledger.mjs";
9
+ import { usagePath } from "./usage.mjs";
10
+ import { GATED_TOOLS, SHELL_TOOLS, callTargets, classifyCall, commandText } from "./risk.mjs";
11
+ import * as retention from "./questions/retention.mjs";
12
+ import * as compaction from "./questions/compaction.mjs";
13
+ import * as gap from "./questions/gap.mjs";
14
+ import * as sources from "./questions/sources.mjs";
15
+ import * as progress from "./questions/progress.mjs";
16
+ import * as untrusted from "./questions/untrusted.mjs";
17
+ import * as capabilities from "./questions/capabilities.mjs";
18
+ import * as guard from "./questions/guard.mjs";
19
+
20
+ const MAX_RECENT = 8;
21
+
22
+ /** One line of a call, for a notification or a block reason. Never a digest; never sent anywhere. */
23
+ function short(value: string, limit: number) {
24
+ const text = String(value ?? "")
25
+ .replace(/\s+/gu, " ")
26
+ .trim();
27
+
28
+ return text.length > limit ? `${text.slice(0, limit - 1)}…` : text;
29
+ }
30
+
31
+ function safeMessage(error: unknown) {
32
+ return String((error as any)?.message ?? error ?? "unknown error").slice(0, 200);
33
+ }
34
+
35
+ /**
36
+ * Tell the person something, and never let the telling change what happens.
37
+ *
38
+ * `ctx.ui.notify` reaches the host over RPC and can throw -- a disconnected client, a torn-down UI,
39
+ * a host without the method. Called inline inside the guard's fail-open catch, one such throw
40
+ * unwound a decided refusal into an allow, so the announcement is isolated from the decision here.
41
+ */
42
+ function announce(ctx: ExtensionContext, message: string) {
43
+ if (!ctx.hasUI) {
44
+ return;
45
+ }
46
+
47
+ try {
48
+ ctx.ui.notify(message, "error");
49
+ } catch {
50
+ // A failed notification is not a reason to run a command, or not to.
51
+ }
52
+ }
53
+
54
+ export default function jevAdvisor(pi: ExtensionAPI) {
55
+ // Session switches live in memory. A session toggle must never write the startup preference,
56
+ // so the saved file is read once per session and only /jev startup ever writes it.
57
+ let settings = loadSettings();
58
+ const broker = createBroker({ loadSettings: () => settings });
59
+ const recent: { tool: string; outcome: string }[] = [];
60
+ let objective = "";
61
+
62
+ // System 5's local state. Every field here is something the session already knows; it exists so
63
+ // that "ask local state first" has something to ask. Local state cannot answer whether a session
64
+ // is stuck, but it answers cheaply whether that question is worth 300ms and a call.
65
+ const HISTORY_WINDOW = 12;
66
+ const history = {
67
+ turn: 0,
68
+ signatures: [] as string[],
69
+ tools: [] as string[],
70
+ errors: [] as string[],
71
+ consecutiveErrors: 0,
72
+ turnsSinceChange: 0,
73
+ filesChanged: 0,
74
+ changedThisTurn: false,
75
+ nudged: false,
76
+ askedAtTurn: undefined as number | undefined,
77
+ };
78
+ const resetHistory = () => {
79
+ history.turn = 0;
80
+ history.signatures.length = 0;
81
+ history.tools.length = 0;
82
+ history.errors.length = 0;
83
+ history.consecutiveErrors = 0;
84
+ history.turnsSinceChange = 0;
85
+ history.filesChanged = 0;
86
+ history.changedThisTurn = false;
87
+ history.nudged = false;
88
+ history.askedAtTurn = undefined;
89
+ };
90
+
91
+ const enabled = (system: string) => settings.master && settings.systems[system] === true;
92
+
93
+ // System 6: decide once, before the first provider request, whether this session will need a
94
+ // withdrawn tool group -- and offer it now rather than at turn 6.
95
+ //
96
+ // Phase 7 is why this exists and why it is shaped like this. Flipping Browser QA on mid-session
97
+ // collapsed cached tokens to 3,200 at the next request in three attempts out of three and cost
98
+ // 20% of the attempt to re-warm; arming the same group from turn 1 cost 16% against 47%. The
99
+ // whole value here is moving one decision earlier, so it happens exactly once and only before
100
+ // the first request.
101
+ let capabilityAsked = false;
102
+ const capabilityDeclined = new Set<string>();
103
+
104
+ /** Persist the session's switches, or report that it could not be done. */
105
+ const persistLayer = (result: { settings: any }) => {
106
+ try {
107
+ return saveSettings(layerToPersist(result, loadSettings()));
108
+ } catch {
109
+ return undefined;
110
+ }
111
+ };
112
+
113
+ pi.on("session_start", () => {
114
+ settings = loadSettings();
115
+ if (!settings.startup) {
116
+ settings = { ...settings, master: false };
117
+ }
118
+
119
+ broker.reset();
120
+ recent.length = 0;
121
+ resetHistory();
122
+ capabilityAsked = false;
123
+ capabilityDeclined.clear();
124
+ });
125
+
126
+ pi.on("session_shutdown", () => {
127
+ // finish, not reset: the counts are published once more as an ended session so anything
128
+ // reading them from outside -- SpecPi Chat's panel, most of all -- shows what the session
129
+ // actually spent rather than a zeroed live one.
130
+ broker.finish();
131
+ recent.length = 0;
132
+ resetHistory();
133
+ });
134
+
135
+ pi.on("turn_start", (event: any) => {
136
+ history.turn = typeof event?.turnIndex === "number" ? event.turnIndex : history.turn + 1;
137
+ history.changedThisTurn = false;
138
+ });
139
+
140
+ // The task objective is the one piece of context every system wants, and it is already in the
141
+ // system prompt, so reading it here costs nothing extra.
142
+ pi.on("before_agent_start", (event: any) => {
143
+ const match = /\[SPECPI TASK CONTRACT\]\n([^\n]{0,200})/u.exec(event?.systemPrompt ?? "");
144
+ if (match) {
145
+ objective = match[1];
146
+ }
147
+ });
148
+
149
+ // Once per session, whatever the answer: asking later would be the mid-session flip the probe
150
+ // priced at three times the cost of doing it now.
151
+ pi.on("before_agent_start", async (event: any, ctx: ExtensionContext) => {
152
+ if (capabilityAsked || !enabled("capability")) {
153
+ return;
154
+ }
155
+
156
+ // No interactive human means no proposal at all, exactly as `request_capability` refuses
157
+ // without one. An unattended run must never be the thing that arms a capability.
158
+ if (!ctx.hasUI) {
159
+ return;
160
+ }
161
+
162
+ capabilityAsked = true;
163
+ try {
164
+ // Dynamically imported so the advisor never hard-depends on workflow-controls: a
165
+ // core-only install fails this import and the system degrades to off rather than
166
+ // taking the whole extension down with it.
167
+ const table = await import("../workflow-controls/capabilities.mjs");
168
+ const { syncActiveTools } = await import("../workflow-controls/web-access.mjs");
169
+ const active = typeof pi.getActiveTools === "function" ? pi.getActiveTools() : [];
170
+ const registered =
171
+ typeof pi.getAllTools === "function" ? pi.getAllTools().map((tool: any) => tool.name) : [];
172
+ // Only groups that are installed and withdrawn. Proposing one that is already on, or
173
+ // one whose package is absent, is a confirmation dialog that can only waste a person's
174
+ // attention.
175
+ const available = table.capabilityNames().filter((id: string) => {
176
+ const capability = table.findCapability(id);
177
+
178
+ return (
179
+ capability &&
180
+ table.capabilityInstalled(registered, capability) &&
181
+ !table.capabilityActive(active, capability)
182
+ );
183
+ });
184
+ if (available.length === 0) {
185
+ return;
186
+ }
187
+
188
+ let entries: string[] = [];
189
+ try {
190
+ entries = fs.readdirSync(ctx.cwd ?? ".").slice(0, 200);
191
+ } catch {
192
+ entries = [];
193
+ }
194
+
195
+ const local = capabilities.localSignals({ prompt: event?.prompt, entries });
196
+ if (!local.ask) {
197
+ return;
198
+ }
199
+
200
+ const result = await broker.request({
201
+ system: "capability",
202
+ state: capabilities.buildInput({
203
+ prompt: event?.prompt,
204
+ reasons: local.reasons,
205
+ available,
206
+ cwdEntries: entries,
207
+ }),
208
+ questions: capabilities.questions({ available }),
209
+ ctx,
210
+ root: ctx.cwd,
211
+ decide: (answers: any) => {
212
+ const advice = capabilities.decide(answers, available);
213
+
214
+ return { applied: advice.propose.length > 0, decision: advice };
215
+ },
216
+ });
217
+ if (!result.ok) {
218
+ return;
219
+ }
220
+
221
+ const advice = result.decision;
222
+ if (advice.suggestDelegation) {
223
+ // A suggestion, never an activation: delegation binds a model and a host and has
224
+ // its own command, which is why the capability table deliberately omits it.
225
+ ctx.ui.notify(
226
+ "Jev: this looks like a question a delegated read-only session could answer over many files. Run /delegate on if you want it.",
227
+ "info",
228
+ );
229
+ }
230
+
231
+ for (const id of advice.propose) {
232
+ const capability = table.findCapability(id);
233
+ if (!capability || capabilityDeclined.has(id)) {
234
+ continue;
235
+ }
236
+
237
+ const pending = table.missingTools(pi.getActiveTools(), capability);
238
+ // The same confirmation `request_capability` shows, pre-filled and moved to turn 0.
239
+ // Authority is unchanged: the human still decides, and declining is remembered so
240
+ // nothing asks twice in one session.
241
+ const accepted = await ctx.ui.confirm(
242
+ `Allow ${capability.label} for this session?`,
243
+ `Jev expects this request to ${capability.summary}, from the request itself rather than from anything it has done yet.\n\nThis offers ${pending.length} tool${pending.length === 1 ? "" : "s"} for the rest of this session and adds ${capability.schemaCost}. Accepting now is materially cheaper than accepting later: activating it mid-session also discards the cached prompt prefix, which measured about 20% of a mid-length attempt's cost. Withdraw it with ${capability.command} off.`,
244
+ );
245
+ if (!accepted) {
246
+ capabilityDeclined.add(id);
247
+ continue;
248
+ }
249
+
250
+ syncActiveTools(pi, capability.tools, true);
251
+ }
252
+ } catch {
253
+ // Nothing here may prevent a session from starting.
254
+ }
255
+ });
256
+
257
+ // System 8: the command guard, before a shell or file call runs.
258
+ //
259
+ // Fail open at every step. Local triage settles most calls for nothing; anything else is asked
260
+ // about, and a call is blocked only on a confident verdict that the request does not account
261
+ // for. Every other outcome -- no key, no budget, a timeout, an unconfident answer, no human to
262
+ // ask -- returns the call to @gotgenes/pi-permission-system, which decides it exactly as it did
263
+ // before this layer existed. The package this replaced was fail-closed, so an outage or a
264
+ // missing key stopped work; that is the single behaviour most worth not reproducing.
265
+ pi.on("tool_call", async (event: any, ctx: ExtensionContext) => {
266
+ if (!enabled("guard") || !GATED_TOOLS.includes(event?.toolName)) {
267
+ return undefined;
268
+ }
269
+
270
+ const shell = SHELL_TOOLS.includes(event.toolName);
271
+ // Not `input.command`: `write_stdin` types into a live shell under another name, so reading
272
+ // one key classified every such call as the empty string -- spending a guard call on nothing
273
+ // while the text actually being run went unexamined.
274
+ const command = shell ? commandText(event?.input) : "";
275
+ // Every file the call names, because `multi_edit` and `apply_patch` do not carry one `path`
276
+ // and a target the guard cannot see is a target it never asks the credential question about.
277
+ const targets = callTargets(event?.input);
278
+ const local = classifyCall({ tool: event.toolName, command, targets, cwd: ctx.cwd });
279
+ const subject = shell
280
+ ? command || "(command unknown)"
281
+ : `${event.toolName} ${targets.join(", ") || "(target unknown)"}`;
282
+
283
+ // Built once, and nothing inside it may throw. A refusal that has already been decided must
284
+ // reach the harness: an exception raised while announcing it would unwind into the fail-open
285
+ // catch below and turn the layer's only blocking action into an allow.
286
+ const refuse = (reason: string) => {
287
+ announce(ctx, `Jev guard blocked ${event.toolName}: ${reason}.`);
288
+
289
+ return { block: true, reason: `Jev guard: ${reason}. Call: ${short(subject, 160)}` };
290
+ };
291
+
292
+ if (local.decision === "safe") {
293
+ return undefined;
294
+ }
295
+
296
+ if (local.decision === "dangerous") {
297
+ // Catastrophic and unambiguous, so it needs neither a network call nor a human. This is
298
+ // the one path that blocks without asking Jev, which is why its rule list is tiny.
299
+ return refuse(local.reason);
300
+ }
301
+
302
+ let verdict;
303
+ try {
304
+ const result = await broker.request({
305
+ system: "guard",
306
+ state: guard.buildInput({
307
+ tool: event.toolName,
308
+ subject,
309
+ protectedTarget: local.reason === "writes to a protected path",
310
+ objective,
311
+ recent,
312
+ cwd: ctx.cwd,
313
+ }),
314
+ questions: guard.questions({ protected: local.reason === "writes to a protected path" }),
315
+ ctx,
316
+ root: ctx.cwd,
317
+ decide: (answers: any) => {
318
+ const verdict = guard.decide(answers, { hasUI: ctx.hasUI });
319
+
320
+ return { applied: verdict.action !== "defer", decision: verdict };
321
+ },
322
+ });
323
+ if (!result.ok) {
324
+ return undefined;
325
+ }
326
+
327
+ verdict = result.decision;
328
+ } catch {
329
+ // An advisor must never be the reason a tool call fails. Anything unexpected while
330
+ // asking hands the call back to the permission system unchanged. The catch ends here, so
331
+ // that everything the verdict then decides is outside it.
332
+ return undefined;
333
+ }
334
+
335
+ if (verdict.action === "block") {
336
+ return refuse(verdict.reason);
337
+ }
338
+
339
+ if (verdict.action === "ask" && ctx.hasUI) {
340
+ let choice;
341
+ try {
342
+ choice = await ctx.ui.select({
343
+ title: "Jev guard",
344
+ message: `This looks ${verdict.reason}: ${short(subject, 300)}`,
345
+ options: [guard.CHOICES.run, guard.CHOICES.block],
346
+ });
347
+ } catch {
348
+ // The one failure in this file that does not fail open, and deliberately. Reaching
349
+ // here means the verdict already said this call needs a person's approval; a host
350
+ // that cannot ask has not obtained it, and an unanswerable question resolved as yes
351
+ // is the failure mode a confirmation dialog exists to rule out.
352
+ return refuse("this needs your approval and you could not be asked");
353
+ }
354
+
355
+ // `guard.approved` owns the rule; see it for why every non-answer is a refusal.
356
+ return guard.approved(choice) ? undefined : refuse("not approved by you");
357
+ }
358
+
359
+ return undefined;
360
+ });
361
+
362
+ // System 1: condense a spent tool result before it is appended. Doing this after the fact would
363
+ // rewrite a cached prefix; on arrival it never touches one.
364
+ pi.on("tool_result", async (event: any, ctx: ExtensionContext) => {
365
+ // Bookkeeping first, and unconditionally. Retention's own eligibility gate returns early on
366
+ // most results, and a history that only recorded the large read-only ones would be blind to
367
+ // exactly the short repeated failures system 5 exists to notice.
368
+ if (event?.isError === true) {
369
+ history.consecutiveErrors += 1;
370
+ history.errors.push(retention.resultText(event).slice(0, 200));
371
+ if (history.errors.length > HISTORY_WINDOW) {
372
+ history.errors.shift();
373
+ }
374
+ } else {
375
+ history.consecutiveErrors = 0;
376
+ if (progress.MUTATING_TOOLS.has(event?.toolName)) {
377
+ history.filesChanged += 1;
378
+ history.changedThisTurn = true;
379
+ }
380
+ }
381
+
382
+ // Every result, not only the ones retention asked about. This history is what lets the
383
+ // command guard tell a cleanup step from a first move, and it was written in one place --
384
+ // inside retention's success path -- so a session running the guard with retention off, or
385
+ // with retention's budget spent, evaluated the block rule against an empty history for its
386
+ // whole length while the question set said history was what the intent answer weighed.
387
+ recent.push({ tool: String(event?.toolName ?? ""), outcome: event?.isError === true ? "error" : "ok" });
388
+ if (recent.length > MAX_RECENT) {
389
+ recent.shift();
390
+ }
391
+
392
+ // Two systems share this hook. Retention wants large read-only results; system 7 wants
393
+ // externally fetched ones whatever their size, because an injected instruction can be two
394
+ // hundred bytes. When both want the same result they are one call: questions are evaluated
395
+ // in parallel against one state, so the second question rides the first's digest for
396
+ // nothing rather than paying for the same bytes twice at the same hook.
397
+ const wantRetention = enabled("retention") && retention.eligible(event);
398
+ const wantUntrusted = enabled("untrusted") && untrusted.applies(event);
399
+ if (!wantRetention && !wantUntrusted) {
400
+ return;
401
+ }
402
+
403
+ try {
404
+ const text = retention.resultText(event);
405
+ const bytes = retention.resultBytes(event);
406
+ const result = await broker.request({
407
+ // Charged to whichever system is driving, which is retention whenever retention is
408
+ // interested. System 7's own budget therefore only binds when retention is off or
409
+ // the result was too small for it.
410
+ system: wantRetention ? "retention" : "untrusted",
411
+ state: retention.buildInput({ event, objective, recent }),
412
+ questions: {
413
+ ...(wantRetention ? retention.questions() : {}),
414
+ ...(wantUntrusted ? untrusted.questions() : {}),
415
+ },
416
+ ctx,
417
+ root: ctx.cwd,
418
+ // The gate runs inside the call so the ledger line can say what the advice did
419
+ // rather than only that it was asked. The replacement text is built here too,
420
+ // because its length is the saving: computing it a second time to measure it would
421
+ // be the measurement inventing its own number.
422
+ decide: (answers: any) => {
423
+ const verdict = wantRetention
424
+ ? retention.decide(answers)
425
+ : { elide: false, reason: "retention-off" };
426
+ const flagged = wantUntrusted && untrusted.decide(answers).banner;
427
+ // Order matters: shorten first, then mark. A banner belongs at the top of
428
+ // whatever the model is actually going to read.
429
+ const body = verdict.elide ? retention.digest(text, { tool: event.toolName, bytes }) : text;
430
+ const replacement = flagged ? untrusted.mark(body) : body;
431
+
432
+ return {
433
+ applied: verdict.elide || flagged,
434
+ savedBytes: verdict.elide ? bytes - Buffer.byteLength(replacement, "utf8") : 0,
435
+ // retention.decide already names why it declined; carrying that into the
436
+ // ledger is what makes "asked and did nothing" diagnosable later.
437
+ reason: flagged ? `${verdict.reason}+flagged` : verdict.reason,
438
+ decision: {
439
+ verdict,
440
+ flagged,
441
+ replacement: verdict.elide || flagged ? replacement : undefined,
442
+ },
443
+ };
444
+ },
445
+ });
446
+ if (!result.ok) {
447
+ return;
448
+ }
449
+
450
+ const { verdict, replacement } = result.decision;
451
+ // Retention knows something the bookkeeping above does not -- whether the result was
452
+ // spent -- so it refines its own entry rather than appending a second one for the same
453
+ // call. If anything has been recorded since, the entry is gone and so is the chance.
454
+ const latest = recent[recent.length - 1];
455
+ if (wantRetention && latest?.tool === event.toolName) {
456
+ latest.outcome = verdict.elide ? "spent" : "kept";
457
+ }
458
+
459
+ if (replacement === undefined) {
460
+ return;
461
+ }
462
+
463
+ // Replace only the text parts. An image part carries no cheap digest and is left whole.
464
+ const images = (event.content ?? []).filter((part: any) => part?.type !== "text");
465
+
466
+ return { content: [{ type: "text" as const, text: replacement }, ...images] };
467
+ } catch {
468
+ // An advisor that throws must not fail the tool call that produced the result.
469
+ return;
470
+ }
471
+ });
472
+
473
+ // System 1b: steer the summary at the one boundary where the prompt cache is discarded anyway.
474
+ // Only customInstructions is supplied; the preparation's own cut and budget are left alone.
475
+ pi.on("session_before_compact", async (event: any, ctx: ExtensionContext) => {
476
+ if (!enabled("compaction")) {
477
+ return;
478
+ }
479
+
480
+ try {
481
+ const result = await broker.request({
482
+ system: "compaction",
483
+ state: compaction.buildInput({ preparation: event.preparation, objective }),
484
+ questions: compaction.questions(),
485
+ ctx,
486
+ root: ctx.cwd,
487
+ signal: event.signal,
488
+ decide: (answers: any) => {
489
+ const built = compaction.decide(answers);
490
+
491
+ // Nothing is shortened here, so savedBytes stays 0 and `applied` is the whole
492
+ // record: either a sentence reached the summariser or Pi's own prompt ran.
493
+ return { applied: Boolean(built.customInstructions), decision: built };
494
+ },
495
+ });
496
+ if (!result.ok) {
497
+ return;
498
+ }
499
+
500
+ const advice = result.decision;
501
+ if (!advice.customInstructions) {
502
+ return;
503
+ }
504
+
505
+ const existing = typeof event.customInstructions === "string" ? event.customInstructions.trim() : "";
506
+
507
+ return {
508
+ customInstructions: existing
509
+ ? `${existing}\n\n${advice.customInstructions}`
510
+ : advice.customInstructions,
511
+ };
512
+ } catch {
513
+ return;
514
+ }
515
+ });
516
+
517
+ // System 1b, second hook. Branch summarisation is the same problem at the same boundary --
518
+ // something is about to be reduced to a summary and the prefix is being rebuilt regardless --
519
+ // and it was simply unserved. It shares the compaction switch rather than adding a fifth
520
+ // system, because a user who has decided the advisor may steer a summary has decided that once.
521
+ //
522
+ // `label` is the part worth having. Pi's `/tree` can filter to labelled entries, so a branch
523
+ // that says what it was is the difference between a navigable tree and a list of timestamps,
524
+ // and the enum is fixed so no model-written text reaches the session file.
525
+ pi.on("session_before_tree", async (event: any, ctx: ExtensionContext) => {
526
+ if (!enabled("compaction")) {
527
+ return;
528
+ }
529
+
530
+ try {
531
+ const entries = event?.preparation?.entriesToSummarize ?? [];
532
+ if (entries.length === 0) {
533
+ return;
534
+ }
535
+
536
+ const result = await broker.request({
537
+ system: "compaction",
538
+ state: compaction.buildBranchInput({ preparation: event.preparation, objective }),
539
+ questions: compaction.questions({ branch: true }),
540
+ ctx,
541
+ root: ctx.cwd,
542
+ signal: event.signal,
543
+ decide: (answers: any) => {
544
+ const built = compaction.decide(answers);
545
+ const branchLabel = compaction.label(answers);
546
+
547
+ return {
548
+ applied: Boolean(branchLabel || built.customInstructions),
549
+ decision: { ...built, label: branchLabel },
550
+ };
551
+ },
552
+ });
553
+ if (!result.ok) {
554
+ return;
555
+ }
556
+
557
+ const advice = result.decision;
558
+ const patch: Record<string, unknown> = {};
559
+ if (advice.label) {
560
+ patch.label = advice.label;
561
+ }
562
+
563
+ // Only when a summary is actually going to be generated. Instructions for a summariser
564
+ // that will not run are bytes nobody reads, and `replaceInstructions` is left alone so
565
+ // Pi's own branch prompt still frames the result.
566
+ if (advice.customInstructions && event.preparation?.userWantsSummary === true) {
567
+ const existing =
568
+ typeof event.preparation?.customInstructions === "string"
569
+ ? event.preparation.customInstructions.trim()
570
+ : "";
571
+ patch.customInstructions = existing
572
+ ? `${existing}\n\n${advice.customInstructions}`
573
+ : advice.customInstructions;
574
+ }
575
+
576
+ return Object.keys(patch).length > 0 ? patch : undefined;
577
+ } catch {
578
+ return;
579
+ }
580
+ });
581
+
582
+ pi.on("tool_call", async (event: any, ctx: ExtensionContext) => {
583
+ history.signatures.push(progress.signature(event.toolName, event.input));
584
+ history.tools.push(event.toolName);
585
+ if (history.signatures.length > HISTORY_WINDOW) {
586
+ history.signatures.shift();
587
+ }
588
+
589
+ if (history.tools.length > HISTORY_WINDOW) {
590
+ history.tools.shift();
591
+ }
592
+
593
+ // System 2: triage a capability gap before tool-wishlist writes it. `event.input` is
594
+ // documented as mutable, so this patches the report in place rather than duplicating any
595
+ // of the wishlist's authority logic. Nothing here records a decision.
596
+ if (event.toolName === "report_capability_gap" && enabled("gap")) {
597
+ try {
598
+ const result = await broker.request({
599
+ system: "gap",
600
+ state: gap.buildInput({ gap: event.input, existing: [] }),
601
+ questions: gap.questions({ gap: event.input, existing: [] }),
602
+ ctx,
603
+ root: ctx.cwd,
604
+ decide: (answers: any) => {
605
+ const built = gap.decide(answers);
606
+ // Exactly the conditions the caller applies below, so the ledger line says
607
+ // what happened rather than what was available. A gated answer that
608
+ // duplicates a field the model already filled in changed nothing.
609
+ const changes =
610
+ (built.blockForSanitization ? 1 : 0) +
611
+ (built.canonicalKey && typeof event.input?.canonicalKey !== "string" ? 1 : 0) +
612
+ (built.suggestedFix && !event.input?.suggestedFix ? 1 : 0) +
613
+ (built.independentImpact ? 1 : 0);
614
+
615
+ return { applied: changes > 0, decision: built };
616
+ },
617
+ });
618
+ if (!result.ok) {
619
+ return;
620
+ }
621
+
622
+ const advice = result.decision;
623
+ if (advice.blockForSanitization) {
624
+ return {
625
+ block: true,
626
+ reason: "This report appears to contain a credential, an absolute path or other machine-specific detail. Rewrite it with the specifics removed and report it again.",
627
+ };
628
+ }
629
+
630
+ if (advice.canonicalKey && typeof event.input?.canonicalKey !== "string") {
631
+ event.input.canonicalKey = advice.canonicalKey;
632
+ }
633
+
634
+ if (advice.suggestedFix && !event.input?.suggestedFix) {
635
+ event.input.suggestedFix = advice.suggestedFix;
636
+ }
637
+
638
+ // Recorded alongside the model's own claim, never over it: a human reading the
639
+ // wishlist should still see what was originally reported.
640
+ if (advice.independentImpact) {
641
+ event.input.independentImpact = advice.independentImpact;
642
+ }
643
+ } catch {
644
+ return;
645
+ }
646
+
647
+ return;
648
+ }
649
+
650
+ // System 4: order the sources a delegation batch will snapshot. Ordering only — the same
651
+ // set is frozen either way, but a child pages through `list_sources` in this order.
652
+ if (event.toolName === "delegate" && enabled("sources") && Array.isArray(event.input?.sources)) {
653
+ try {
654
+ const candidates = event.input.sources
655
+ .filter((item: unknown) => typeof item === "string")
656
+ .map((item: string) => ({ path: item }));
657
+ if (candidates.length < 2) {
658
+ return;
659
+ }
660
+
661
+ const result = await broker.request({
662
+ system: "sources",
663
+ state: sources.buildInput({ question: event.input?.question ?? objective, candidates }),
664
+ questions: sources.questions({ candidates }),
665
+ ctx,
666
+ root: ctx.cwd,
667
+ decide: (answers: any) => {
668
+ const built = sources.decide(answers, candidates);
669
+ const order = built.ordered.map((item: any) => item.path);
670
+ // An ungated run returns the caller's own order, which is not a change and
671
+ // must not be recorded as one.
672
+ const moved = order.some((item: string, index: number) => item !== candidates[index]?.path);
673
+
674
+ return { applied: moved, decision: { ranked: built, ordered: order } };
675
+ },
676
+ });
677
+ if (!result.ok) {
678
+ return;
679
+ }
680
+
681
+ const { ranked, ordered } = result.decision;
682
+ const missing = event.input.sources.filter((item: string) => !ordered.includes(item));
683
+ event.input.sources = [...ordered, ...missing];
684
+
685
+ // Two answers the same batch already computed and nothing read. Output is free, so
686
+ // they were paid for whether or not anyone looked. A confident "this is not a
687
+ // self-contained evidence question" is worth surfacing before specpi-delegation
688
+ // freezes up to 200 files and 8 MiB for a child that then cannot answer it.
689
+ //
690
+ // Advisory only, and deliberately so: the batch still runs, the ceilings are
691
+ // unchanged, and with no UI this says nothing rather than blocking.
692
+ if (ranked.notWorthDelegating && ctx.hasUI) {
693
+ ctx.ui.notify(
694
+ `Jev rates this a poor fit for delegation${ranked.jobMode ? ` (it reads as ${ranked.jobMode} work)` : ""}. Running anyway; ${ordered.length + missing.length} sources will be frozen for the child.`,
695
+ "warning",
696
+ );
697
+ }
698
+ } catch {
699
+ return;
700
+ }
701
+ }
702
+ });
703
+
704
+ // System 5: notice a session that has stopped making progress, while it can still be helped.
705
+ //
706
+ // THE ONE HANDLER THAT IS NOT AWAITED. Everything else in this file mutates what it inspects --
707
+ // a tool result, a compaction patch, a tool's input -- so the session has to wait for the
708
+ // answer. This one acts on the next turn, and at roughly 300ms a call, awaiting it on a
709
+ // thrashing session would add seconds to an attempt to deliver advice that could not have
710
+ // changed the turn it was asked during.
711
+ pi.on("turn_end", (event: any, ctx: ExtensionContext) => {
712
+ // Kept whether or not the system is on, so switching it on mid-session does not start from
713
+ // a blank history and immediately look healthy.
714
+ history.turnsSinceChange = history.changedThisTurn ? 0 : history.turnsSinceChange + 1;
715
+ if (!enabled("progress") || history.nudged) {
716
+ return;
717
+ }
718
+
719
+ const local = progress.suspicious(history);
720
+ if (!local.ask) {
721
+ return;
722
+ }
723
+
724
+ // Recorded before the call rather than after, so a slow answer cannot let the next turn ask
725
+ // again while this one is still in flight.
726
+ history.askedAtTurn = history.turn;
727
+ void (async () => {
728
+ try {
729
+ const result = await broker.request({
730
+ system: "progress",
731
+ state: progress.buildInput({ history, objective, reasons: local.reasons }),
732
+ questions: progress.questions(),
733
+ ctx,
734
+ root: ctx.cwd,
735
+ decide: (answers: any) => {
736
+ const advice = progress.decide(answers);
737
+
738
+ return {
739
+ applied: Boolean(advice.nudge),
740
+ reason: advice.nudge ? advice.mode : advice.stuck ? "stuck-but-mode-ungated" : "not-stuck",
741
+ decision: advice,
742
+ };
743
+ },
744
+ });
745
+ if (!result.ok || !result.decision?.nudge) {
746
+ return;
747
+ }
748
+
749
+ // Write-once, per the standing rule. A second nudge would either repeat a line the
750
+ // model already has or contradict it, and neither can be withdrawn: it was appended
751
+ // to a prefix that is cached behind it by the time anyone regrets it.
752
+ history.nudged = true;
753
+ if (ctx.hasUI) {
754
+ ctx.ui.notify(
755
+ result.decision.needsHuman
756
+ ? `Jev progress check: this session looks blocked on something only you can answer. ${result.decision.nudge}`
757
+ : `Jev progress check: ${result.decision.nudge}`,
758
+ "warning",
759
+ );
760
+ }
761
+
762
+ // The plan specified `deliverAs: "nextTurn"`, which is documented as "queued for
763
+ // next user prompt, does not interrupt or trigger anything". An unattended session
764
+ // has exactly one user prompt, so a nextTurn message would never be delivered -- in
765
+ // precisely the case the argument for this system rests on, a headless attempt
766
+ // burning its wall clock. "steer" is delivered after the current tool calls finish
767
+ // and before the next model request, which is the same append at the same boundary
768
+ // and is actually read. triggerTurn is left off so this can never add a turn.
769
+ // Suppressed when the session is blocked on something only a person can answer:
770
+ // steering a model past a missing credential costs a turn to say nothing.
771
+ if (
772
+ settings.progressNudge === "message" &&
773
+ !result.decision.needsHuman &&
774
+ typeof pi.sendMessage === "function"
775
+ ) {
776
+ pi.sendMessage(
777
+ {
778
+ customType: "specpi-jev-progress",
779
+ content: result.decision.nudge,
780
+ display: true,
781
+ details: { mode: result.decision.mode, reasons: local.reasons },
782
+ },
783
+ { deliverAs: "steer" },
784
+ );
785
+ }
786
+ } catch {
787
+ // The turn has already ended. An advisor must not be able to fail it retroactively.
788
+ }
789
+ })();
790
+ });
791
+
792
+ pi.registerCommand("jev", {
793
+ description: "Show or change the Jev advisor: master switch, per-system switches and the transmission ledger",
794
+ getArgumentCompletions: (prefix: string) =>
795
+ ["status", "on", "off", "startup", "enable", "disable", "ledger", "forget"]
796
+ .filter((value) => value.startsWith(prefix.trim().toLowerCase()))
797
+ .map((value) => ({ value, label: value })),
798
+ handler: async (args: string, ctx: ExtensionContext) => {
799
+ const [actionRaw = "status", ...rest] = args.trim().split(/\s+/u).filter(Boolean);
800
+ const action = actionRaw.toLowerCase();
801
+ try {
802
+ if (action === "on" || action === "off") {
803
+ const on = action === "on";
804
+ // `--session` is the old behaviour, kept for the case it was the right one: a
805
+ // one-off try that must not change what the next session does.
806
+ const sessionOnly = rest.some((value) => /^--?(session|once)$/u.test(value.toLowerCase()));
807
+ const unknown = rest.filter((value) => !/^--?(session|once)$/u.test(value.toLowerCase()));
808
+ if (unknown.length > 0) {
809
+ throw new Error(`Usage: /jev ${action} [--session]`);
810
+ }
811
+
812
+ const result = applyLayer({ on }, { settings }, { keySources: () => keySources() });
813
+ settings = result.settings;
814
+ // Persisting is the default because a switch that forgets is not a switch. The
815
+ // old rule -- that only /jev startup may write -- protected against a session
816
+ // toggle silently changing tomorrow's sessions, but the cost of that protection
817
+ // was a layer people turned on repeatedly and never actually ran.
818
+ const persisted = !sessionOnly && ctx.hasUI ? persistLayer(result) : undefined;
819
+ const lines = [
820
+ ...result.lines,
821
+ layerScopeLine({
822
+ sessionOnly,
823
+ interactive: ctx.hasUI,
824
+ persisted,
825
+ stored: loadSettings(),
826
+ settingsFile: settingsPath(),
827
+ }),
828
+ ];
829
+ ctx.ui.notify(lines.join("\n"), "info");
830
+
831
+ return;
832
+ }
833
+
834
+ if (action === "enable" || action === "disable") {
835
+ const names = rest.map((name) => name.toLowerCase());
836
+ const unknown = names.filter((name) => !SYSTEM_NAMES.includes(name));
837
+ if (names.length === 0 || unknown.length > 0) {
838
+ throw new Error(`Usage: /jev ${action} <${SYSTEM_NAMES.join("|")}>`);
839
+ }
840
+
841
+ const changes = Object.fromEntries(names.map((name) => [name, action === "enable"]));
842
+ const systems = { ...settings.systems, ...changes };
843
+ // Disabling the last system while the layer is on leaves it running and doing
844
+ // nothing -- the state `enableSystems`, `startupToPersist`, `couple` and the
845
+ // Chat panel's save check all exist to prevent, reachable through the one path
846
+ // that did not check it. Switching the layer off is the honest reading of
847
+ // "disable everything", and it is announced rather than inferred.
848
+ const emptied = settings.master && SYSTEM_NAMES.every((name) => !systems[name]);
849
+ settings = { ...settings, systems, master: emptied ? false : settings.master };
850
+ // Persisted like every other switch here, and merged into the stored map rather
851
+ // than overwriting it: this session's copy may predate systems enabled on disk
852
+ // since it started, and writing it whole turned those back off silently.
853
+ const kept = ctx.hasUI
854
+ ? (() => {
855
+ try {
856
+ const current = loadSettings();
857
+ const merged = { ...current.systems, ...changes };
858
+ const dead = current.master && SYSTEM_NAMES.every((name) => !merged[name]);
859
+
860
+ return saveSettings({
861
+ ...current,
862
+ systems: merged,
863
+ master: dead ? false : current.master,
864
+ startup: dead ? false : current.startup,
865
+ });
866
+ } catch {
867
+ return undefined;
868
+ }
869
+ })()
870
+ : undefined;
871
+ // The same disclosure `/jev on` makes, on the path that arms the guard by name.
872
+ // Learning from a blocked call that calls can be blocked is the outcome that
873
+ // rule exists to prevent, and which command did the arming does not change it.
874
+ const armedGuard = action === "enable" && names.includes("guard") && settings.master;
875
+ ctx.ui.notify(
876
+ `${action === "enable" ? "Enabled" : "Disabled"}: ${names.join(", ")}.` +
877
+ `${kept ? " Remembered for new sessions." : " This session only."}` +
878
+ `${emptied ? " That was the last system, so the layer was switched off; it would otherwise run and do nothing." : ""}` +
879
+ `${!emptied && !settings.master ? " The layer is still off; run /jev on." : ""}` +
880
+ `${armedGuard ? `\n${guardWarning()}` : ""}`,
881
+ "info",
882
+ );
883
+
884
+ return;
885
+ }
886
+
887
+ if (action === "startup") {
888
+ const [choice] = rest;
889
+ if (!choice) {
890
+ const current = loadSettings();
891
+ ctx.ui.notify(
892
+ `Jev starts ${current.startup && current.master ? "on" : "off"} in new sessions. Preference: ${settingsPath()}`,
893
+ "info",
894
+ );
895
+
896
+ return;
897
+ }
898
+
899
+ if (!ctx.hasUI) {
900
+ throw new Error("Startup changes require a human interactive command");
901
+ }
902
+
903
+ if (!["on", "off"].includes(choice.toLowerCase())) {
904
+ throw new Error("Usage: /jev startup [on|off]");
905
+ }
906
+
907
+ // Both keys, and the systems with them. Writing `startup` alone was the whole
908
+ // two-keys-for-one-intention trap, left in the command named after it: the
909
+ // advisor's session_start keeps a stored `master` only when `startup` is true,
910
+ // so `startup: true, master: false` starts every future session with the layer
911
+ // off while this command cheerfully reported it would start on. And a layer
912
+ // that starts on with no system enabled runs and does nothing, so the same rule
913
+ // `/jev on` uses applies here: fill them in only when none are chosen.
914
+ const wanted = choice.toLowerCase() === "on";
915
+ const saved = saveSettings(startupToPersist(wanted, loadSettings()));
916
+ const enabled = SYSTEM_NAMES.filter((name) => saved.systems[name]);
917
+ ctx.ui.notify(
918
+ saved.startup && saved.master
919
+ ? `New Pi sessions will start with the Jev layer on, with ${enabled.length} of ${SYSTEM_NAMES.length} systems: ${enabled.join(", ")}. This session is unchanged; run /jev on to switch it on now.` +
920
+ `${saved.systems.guard ? `\n${guardWarning()}` : ""}`
921
+ : "New Pi sessions will start with the Jev layer off. This session is unchanged.",
922
+ "info",
923
+ );
924
+
925
+ return;
926
+ }
927
+
928
+ if (action === "forget") {
929
+ if (!ctx.hasUI) {
930
+ throw new Error("Revoking consent requires a human interactive command");
931
+ }
932
+
933
+ revokeConsent();
934
+ ctx.ui.notify(
935
+ "Forgot the Jev transmission consent. The next system that would send anything will ask again.",
936
+ "info",
937
+ );
938
+
939
+ return;
940
+ }
941
+
942
+ if (action === "ledger") {
943
+ const limit = Number.parseInt(rest[0] ?? "10", 10);
944
+ const entries = readLedger(Number.isInteger(limit) ? limit : 10);
945
+ if (entries.length === 0) {
946
+ ctx.ui.notify(`No Jev transmissions recorded. Ledger: ${ledgerPath()}`, "info");
947
+
948
+ return;
949
+ }
950
+
951
+ const lines = entries.map(
952
+ (entry: any) =>
953
+ `${entry.at} ${entry.system} ${entry.stateBytes}B ${entry.ok ? `${entry.latencyMs}ms` : entry.reason} ${String(entry.payloadSha256 ?? "").slice(0, 12)} [${(entry.questionKeys ?? []).join(", ")}]`,
954
+ );
955
+ ctx.ui.notify(`${lines.join("\n")}\n\nLedger: ${ledgerPath()}`, "info");
956
+
957
+ return;
958
+ }
959
+
960
+ if (action !== "status") {
961
+ throw new Error(
962
+ "Usage: /jev [status|on [--session]|off [--session]|startup [on|off]|enable <system>|disable <system>|ledger [n]|forget]",
963
+ );
964
+ }
965
+
966
+ const state = broker.status();
967
+ const sources = keySources();
968
+ const activeSource = sources.find((source: { present: boolean }) => source.present)?.name;
969
+ const stored = loadSettings();
970
+ const lines = [
971
+ `master: ${settings.master ? "on" : "off"} (new sessions start ${stored.startup && stored.master ? "on" : "off"})`,
972
+ ...SYSTEM_NAMES.map((name) => ` ${name}: ${settings.systems[name] ? "on" : "off"}`),
973
+ // Every place a key could come from, in the order they are consulted, with the
974
+ // one in force marked. A bare "missing" was actively misleading here: it is
975
+ // what someone saw who had a perfectly good OpenRouter key stored by /login,
976
+ // and it gave them nothing to act on. Names only -- no key is ever printed.
977
+ `key: ${activeSource ? `in use from ${activeSource}` : "none found"}`,
978
+ ...sources.map(
979
+ (source: { name: string; label: string; detail: string; present: boolean }) =>
980
+ ` ${source.present ? "found" : " - "} ${source.label} (${source.detail})`,
981
+ ),
982
+ `consent: ${granted() ? "granted" : "not granted"}`,
983
+ `calls this session: ${state.callsUsed}/${state.budgets.total} total`,
984
+ ...SYSTEM_NAMES.map((name) => ` ${name}: ${state.usedBySystem[name] ?? 0}/${state.budgets[name]}`),
985
+ `settings: ${settingsPath()}`,
986
+ `consent file: ${consentPath()}`,
987
+ `ledger: ${ledgerPath()}`,
988
+ // Named here because it is the one file another process is meant to read, and
989
+ // SpecPi Chat showing a number nobody can find is how a number stops being
990
+ // checkable.
991
+ `session counts: ${usagePath()}`,
992
+ ];
993
+ ctx.ui.notify(lines.join("\n"), "info");
994
+ } catch (error) {
995
+ ctx.ui.notify(safeMessage(error), "error");
996
+ }
997
+ },
998
+ });
999
+ }