specpi 0.26.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,849 @@
1
+ import fs from "node:fs";
2
+ import { type ExtensionAPI, type ExtensionContext } from "@earendil-works/pi-coding-agent";
3
+ import { SYSTEM_NAMES, keyPresent, loadSettings, saveSettings, settingsPath } from "./config.mjs";
4
+ import { consentPath, granted, revokeConsent } from "./consent.mjs";
5
+ import { createBroker } from "./broker.mjs";
6
+ import { ledgerPath, read as readLedger } from "./ledger.mjs";
7
+ import { usagePath } from "./usage.mjs";
8
+ import { applyConfig as applyGuardConfig, statusLine as guardStatusLine } from "./guard.mjs";
9
+ import * as retention from "./questions/retention.mjs";
10
+ import * as compaction from "./questions/compaction.mjs";
11
+ import * as gap from "./questions/gap.mjs";
12
+ import * as sources from "./questions/sources.mjs";
13
+ import * as progress from "./questions/progress.mjs";
14
+ import * as untrusted from "./questions/untrusted.mjs";
15
+ import * as capabilities from "./questions/capabilities.mjs";
16
+
17
+ const MAX_RECENT = 8;
18
+
19
+ function safeMessage(error: unknown) {
20
+ return String((error as any)?.message ?? error ?? "unknown error").slice(0, 200);
21
+ }
22
+
23
+ export default function jevAdvisor(pi: ExtensionAPI) {
24
+ // Session switches live in memory. A session toggle must never write the startup preference,
25
+ // so the saved file is read once per session and only /jev startup ever writes it.
26
+ let settings = loadSettings();
27
+ const broker = createBroker({ loadSettings: () => settings });
28
+ const recent: { tool: string; outcome: string }[] = [];
29
+ let objective = "";
30
+
31
+ // System 5's local state. Every field here is something the session already knows; it exists so
32
+ // that "ask local state first" has something to ask. Local state cannot answer whether a session
33
+ // is stuck, but it answers cheaply whether that question is worth 300ms and a call.
34
+ const HISTORY_WINDOW = 12;
35
+ const history = {
36
+ turn: 0,
37
+ signatures: [] as string[],
38
+ tools: [] as string[],
39
+ errors: [] as string[],
40
+ consecutiveErrors: 0,
41
+ turnsSinceChange: 0,
42
+ filesChanged: 0,
43
+ changedThisTurn: false,
44
+ nudged: false,
45
+ askedAtTurn: undefined as number | undefined,
46
+ };
47
+ const resetHistory = () => {
48
+ history.turn = 0;
49
+ history.signatures.length = 0;
50
+ history.tools.length = 0;
51
+ history.errors.length = 0;
52
+ history.consecutiveErrors = 0;
53
+ history.turnsSinceChange = 0;
54
+ history.filesChanged = 0;
55
+ history.changedThisTurn = false;
56
+ history.nudged = false;
57
+ history.askedAtTurn = undefined;
58
+ };
59
+
60
+ const enabled = (system: string) => settings.master && settings.systems[system] === true;
61
+
62
+ // System 6: decide once, before the first provider request, whether this session will need a
63
+ // withdrawn tool group -- and offer it now rather than at turn 6.
64
+ //
65
+ // Phase 7 is why this exists and why it is shaped like this. Flipping Browser QA on mid-session
66
+ // collapsed cached tokens to 3,200 at the next request in three attempts out of three and cost
67
+ // 20% of the attempt to re-warm; arming the same group from turn 1 cost 16% against 47%. The
68
+ // whole value here is moving one decision earlier, so it happens exactly once and only before
69
+ // the first request.
70
+ let capabilityAsked = false;
71
+ const capabilityDeclined = new Set<string>();
72
+
73
+ let guardEnabled = false;
74
+ const syncGuard = () => {
75
+ try {
76
+ return applyGuardConfig(guardEnabled);
77
+ } catch {
78
+ // A guard that cannot be reconfigured keeps whatever posture it has, which
79
+ // /jev status reports rather than hides.
80
+ return { applied: false, reason: "unwritable" };
81
+ }
82
+ };
83
+
84
+ pi.on("session_start", () => {
85
+ settings = loadSettings();
86
+ if (!settings.startup) {
87
+ settings = { ...settings, master: false };
88
+ }
89
+
90
+ broker.reset();
91
+ recent.length = 0;
92
+ resetHistory();
93
+ capabilityAsked = false;
94
+ capabilityDeclined.clear();
95
+ guardEnabled = settings.guard.startup === true;
96
+ // Deliberately outside the master switch. The guard is a separate package with its own
97
+ // gate, and whether it is inert is a property of the install rather than a feature of the
98
+ // advisor, so its configuration is rewritten every session either way. Off is the default
99
+ // and is a real written configuration, not an absence of one.
100
+ syncGuard();
101
+ });
102
+
103
+ pi.on("session_shutdown", () => {
104
+ // finish, not reset: the counts are published once more as an ended session so anything
105
+ // reading them from outside -- SpecPi Chat's panel, most of all -- shows what the session
106
+ // actually spent rather than a zeroed live one.
107
+ broker.finish();
108
+ recent.length = 0;
109
+ resetHistory();
110
+ });
111
+
112
+ pi.on("turn_start", (event: any) => {
113
+ history.turn = typeof event?.turnIndex === "number" ? event.turnIndex : history.turn + 1;
114
+ history.changedThisTurn = false;
115
+ });
116
+
117
+ // The task objective is the one piece of context every system wants, and it is already in the
118
+ // system prompt, so reading it here costs nothing extra.
119
+ pi.on("before_agent_start", (event: any) => {
120
+ const match = /\[SPECPI TASK CONTRACT\]\n([^\n]{0,200})/u.exec(event?.systemPrompt ?? "");
121
+ if (match) {
122
+ objective = match[1];
123
+ }
124
+ });
125
+
126
+ // Once per session, whatever the answer: asking later would be the mid-session flip the probe
127
+ // priced at three times the cost of doing it now.
128
+ pi.on("before_agent_start", async (event: any, ctx: ExtensionContext) => {
129
+ if (capabilityAsked || !enabled("capability")) {
130
+ return;
131
+ }
132
+
133
+ // No interactive human means no proposal at all, exactly as `request_capability` refuses
134
+ // without one. An unattended run must never be the thing that arms a capability.
135
+ if (!ctx.hasUI) {
136
+ return;
137
+ }
138
+
139
+ capabilityAsked = true;
140
+ try {
141
+ // Dynamically imported so the advisor never hard-depends on workflow-controls: a
142
+ // core-only install fails this import and the system degrades to off rather than
143
+ // taking the whole extension down with it.
144
+ const table = await import("../workflow-controls/capabilities.mjs");
145
+ const { syncActiveTools } = await import("../workflow-controls/web-access.mjs");
146
+ const active = typeof pi.getActiveTools === "function" ? pi.getActiveTools() : [];
147
+ const registered =
148
+ typeof pi.getAllTools === "function" ? pi.getAllTools().map((tool: any) => tool.name) : [];
149
+ // Only groups that are installed and withdrawn. Proposing one that is already on, or
150
+ // one whose package is absent, is a confirmation dialog that can only waste a person's
151
+ // attention.
152
+ const available = table.capabilityNames().filter((id: string) => {
153
+ const capability = table.findCapability(id);
154
+
155
+ return (
156
+ capability &&
157
+ table.capabilityInstalled(registered, capability) &&
158
+ !table.capabilityActive(active, capability)
159
+ );
160
+ });
161
+ if (available.length === 0) {
162
+ return;
163
+ }
164
+
165
+ let entries: string[] = [];
166
+ try {
167
+ entries = fs.readdirSync(ctx.cwd ?? ".").slice(0, 200);
168
+ } catch {
169
+ entries = [];
170
+ }
171
+
172
+ const local = capabilities.localSignals({ prompt: event?.prompt, entries });
173
+ if (!local.ask) {
174
+ return;
175
+ }
176
+
177
+ const result = await broker.request({
178
+ system: "capability",
179
+ state: capabilities.buildInput({
180
+ prompt: event?.prompt,
181
+ reasons: local.reasons,
182
+ available,
183
+ cwdEntries: entries,
184
+ }),
185
+ questions: capabilities.questions({ available }),
186
+ ctx,
187
+ root: ctx.cwd,
188
+ decide: (answers: any) => {
189
+ const advice = capabilities.decide(answers, available);
190
+
191
+ return { applied: advice.propose.length > 0, decision: advice };
192
+ },
193
+ });
194
+ if (!result.ok) {
195
+ return;
196
+ }
197
+
198
+ const advice = result.decision;
199
+ if (advice.suggestDelegation) {
200
+ // A suggestion, never an activation: delegation binds a model and a host and has
201
+ // its own command, which is why the capability table deliberately omits it.
202
+ ctx.ui.notify(
203
+ "Jev: this looks like a question a delegated read-only session could answer over many files. Run /delegate on if you want it.",
204
+ "info",
205
+ );
206
+ }
207
+
208
+ for (const id of advice.propose) {
209
+ const capability = table.findCapability(id);
210
+ if (!capability || capabilityDeclined.has(id)) {
211
+ continue;
212
+ }
213
+
214
+ const pending = table.missingTools(pi.getActiveTools(), capability);
215
+ // The same confirmation `request_capability` shows, pre-filled and moved to turn 0.
216
+ // Authority is unchanged: the human still decides, and declining is remembered so
217
+ // nothing asks twice in one session.
218
+ const accepted = await ctx.ui.confirm(
219
+ `Allow ${capability.label} for this session?`,
220
+ `Jev expects this request to ${capability.summary}, from the request itself rather than from anything it has done yet.\n\nThis offers ${pending.length} tool${pending.length === 1 ? "" : "s"} for the rest of this session and adds ${capability.schemaCost}. Accepting now is materially cheaper than accepting later: activating it mid-session also discards the cached prompt prefix, which measured about 20% of a mid-length attempt's cost. Withdraw it with ${capability.command} off.`,
221
+ );
222
+ if (!accepted) {
223
+ capabilityDeclined.add(id);
224
+ continue;
225
+ }
226
+
227
+ syncActiveTools(pi, capability.tools, true);
228
+ }
229
+ } catch {
230
+ // Nothing here may prevent a session from starting.
231
+ }
232
+ });
233
+
234
+ // System 1: condense a spent tool result before it is appended. Doing this after the fact would
235
+ // rewrite a cached prefix; on arrival it never touches one.
236
+ pi.on("tool_result", async (event: any, ctx: ExtensionContext) => {
237
+ // Bookkeeping first, and unconditionally. Retention's own eligibility gate returns early on
238
+ // most results, and a history that only recorded the large read-only ones would be blind to
239
+ // exactly the short repeated failures system 5 exists to notice.
240
+ if (event?.isError === true) {
241
+ history.consecutiveErrors += 1;
242
+ history.errors.push(retention.resultText(event).slice(0, 200));
243
+ if (history.errors.length > HISTORY_WINDOW) {
244
+ history.errors.shift();
245
+ }
246
+ } else {
247
+ history.consecutiveErrors = 0;
248
+ if (progress.MUTATING_TOOLS.has(event?.toolName)) {
249
+ history.filesChanged += 1;
250
+ history.changedThisTurn = true;
251
+ }
252
+ }
253
+
254
+ // Two systems share this hook. Retention wants large read-only results; system 7 wants
255
+ // externally fetched ones whatever their size, because an injected instruction can be two
256
+ // hundred bytes. When both want the same result they are one call: questions are evaluated
257
+ // in parallel against one state, so the second question rides the first's digest for
258
+ // nothing rather than paying for the same bytes twice at the same hook.
259
+ const wantRetention = enabled("retention") && retention.eligible(event);
260
+ const wantUntrusted = enabled("untrusted") && untrusted.applies(event);
261
+ if (!wantRetention && !wantUntrusted) {
262
+ return;
263
+ }
264
+
265
+ try {
266
+ const text = retention.resultText(event);
267
+ const bytes = retention.resultBytes(event);
268
+ const result = await broker.request({
269
+ // Charged to whichever system is driving, which is retention whenever retention is
270
+ // interested. System 7's own budget therefore only binds when retention is off or
271
+ // the result was too small for it.
272
+ system: wantRetention ? "retention" : "untrusted",
273
+ state: retention.buildInput({ event, objective, recent }),
274
+ questions: {
275
+ ...(wantRetention ? retention.questions() : {}),
276
+ ...(wantUntrusted ? untrusted.questions() : {}),
277
+ },
278
+ ctx,
279
+ root: ctx.cwd,
280
+ // The gate runs inside the call so the ledger line can say what the advice did
281
+ // rather than only that it was asked. The replacement text is built here too,
282
+ // because its length is the saving: computing it a second time to measure it would
283
+ // be the measurement inventing its own number.
284
+ decide: (answers: any) => {
285
+ const verdict = wantRetention
286
+ ? retention.decide(answers)
287
+ : { elide: false, reason: "retention-off" };
288
+ const flagged = wantUntrusted && untrusted.decide(answers).banner;
289
+ // Order matters: shorten first, then mark. A banner belongs at the top of
290
+ // whatever the model is actually going to read.
291
+ const body = verdict.elide ? retention.digest(text, { tool: event.toolName, bytes }) : text;
292
+ const replacement = flagged ? untrusted.mark(body) : body;
293
+
294
+ return {
295
+ applied: verdict.elide || flagged,
296
+ savedBytes: verdict.elide ? bytes - Buffer.byteLength(replacement, "utf8") : 0,
297
+ // retention.decide already names why it declined; carrying that into the
298
+ // ledger is what makes "asked and did nothing" diagnosable later.
299
+ reason: flagged ? `${verdict.reason}+flagged` : verdict.reason,
300
+ decision: {
301
+ verdict,
302
+ flagged,
303
+ replacement: verdict.elide || flagged ? replacement : undefined,
304
+ },
305
+ };
306
+ },
307
+ });
308
+ if (!result.ok) {
309
+ return;
310
+ }
311
+
312
+ const { verdict, replacement } = result.decision;
313
+ if (wantRetention) {
314
+ recent.push({ tool: event.toolName, outcome: verdict.elide ? "spent" : "kept" });
315
+ if (recent.length > MAX_RECENT) {
316
+ recent.shift();
317
+ }
318
+ }
319
+
320
+ if (replacement === undefined) {
321
+ return;
322
+ }
323
+
324
+ // Replace only the text parts. An image part carries no cheap digest and is left whole.
325
+ const images = (event.content ?? []).filter((part: any) => part?.type !== "text");
326
+
327
+ return { content: [{ type: "text" as const, text: replacement }, ...images] };
328
+ } catch {
329
+ // An advisor that throws must not fail the tool call that produced the result.
330
+ return;
331
+ }
332
+ });
333
+
334
+ // System 1b: steer the summary at the one boundary where the prompt cache is discarded anyway.
335
+ // Only customInstructions is supplied; the preparation's own cut and budget are left alone.
336
+ pi.on("session_before_compact", async (event: any, ctx: ExtensionContext) => {
337
+ if (!enabled("compaction")) {
338
+ return;
339
+ }
340
+
341
+ try {
342
+ const result = await broker.request({
343
+ system: "compaction",
344
+ state: compaction.buildInput({ preparation: event.preparation, objective }),
345
+ questions: compaction.questions(),
346
+ ctx,
347
+ root: ctx.cwd,
348
+ signal: event.signal,
349
+ decide: (answers: any) => {
350
+ const built = compaction.decide(answers);
351
+
352
+ // Nothing is shortened here, so savedBytes stays 0 and `applied` is the whole
353
+ // record: either a sentence reached the summariser or Pi's own prompt ran.
354
+ return { applied: Boolean(built.customInstructions), decision: built };
355
+ },
356
+ });
357
+ if (!result.ok) {
358
+ return;
359
+ }
360
+
361
+ const advice = result.decision;
362
+ if (!advice.customInstructions) {
363
+ return;
364
+ }
365
+
366
+ const existing = typeof event.customInstructions === "string" ? event.customInstructions.trim() : "";
367
+
368
+ return {
369
+ customInstructions: existing
370
+ ? `${existing}\n\n${advice.customInstructions}`
371
+ : advice.customInstructions,
372
+ };
373
+ } catch {
374
+ return;
375
+ }
376
+ });
377
+
378
+ // System 1b, second hook. Branch summarisation is the same problem at the same boundary --
379
+ // something is about to be reduced to a summary and the prefix is being rebuilt regardless --
380
+ // and it was simply unserved. It shares the compaction switch rather than adding a fifth
381
+ // system, because a user who has decided the advisor may steer a summary has decided that once.
382
+ //
383
+ // `label` is the part worth having. Pi's `/tree` can filter to labelled entries, so a branch
384
+ // that says what it was is the difference between a navigable tree and a list of timestamps,
385
+ // and the enum is fixed so no model-written text reaches the session file.
386
+ pi.on("session_before_tree", async (event: any, ctx: ExtensionContext) => {
387
+ if (!enabled("compaction")) {
388
+ return;
389
+ }
390
+
391
+ try {
392
+ const entries = event?.preparation?.entriesToSummarize ?? [];
393
+ if (entries.length === 0) {
394
+ return;
395
+ }
396
+
397
+ const result = await broker.request({
398
+ system: "compaction",
399
+ state: compaction.buildBranchInput({ preparation: event.preparation, objective }),
400
+ questions: compaction.questions({ branch: true }),
401
+ ctx,
402
+ root: ctx.cwd,
403
+ signal: event.signal,
404
+ decide: (answers: any) => {
405
+ const built = compaction.decide(answers);
406
+ const branchLabel = compaction.label(answers);
407
+
408
+ return {
409
+ applied: Boolean(branchLabel || built.customInstructions),
410
+ decision: { ...built, label: branchLabel },
411
+ };
412
+ },
413
+ });
414
+ if (!result.ok) {
415
+ return;
416
+ }
417
+
418
+ const advice = result.decision;
419
+ const patch: Record<string, unknown> = {};
420
+ if (advice.label) {
421
+ patch.label = advice.label;
422
+ }
423
+
424
+ // Only when a summary is actually going to be generated. Instructions for a summariser
425
+ // that will not run are bytes nobody reads, and `replaceInstructions` is left alone so
426
+ // Pi's own branch prompt still frames the result.
427
+ if (advice.customInstructions && event.preparation?.userWantsSummary === true) {
428
+ const existing =
429
+ typeof event.preparation?.customInstructions === "string"
430
+ ? event.preparation.customInstructions.trim()
431
+ : "";
432
+ patch.customInstructions = existing
433
+ ? `${existing}\n\n${advice.customInstructions}`
434
+ : advice.customInstructions;
435
+ }
436
+
437
+ return Object.keys(patch).length > 0 ? patch : undefined;
438
+ } catch {
439
+ return;
440
+ }
441
+ });
442
+
443
+ pi.on("tool_call", async (event: any, ctx: ExtensionContext) => {
444
+ history.signatures.push(progress.signature(event.toolName, event.input));
445
+ history.tools.push(event.toolName);
446
+ if (history.signatures.length > HISTORY_WINDOW) {
447
+ history.signatures.shift();
448
+ }
449
+
450
+ if (history.tools.length > HISTORY_WINDOW) {
451
+ history.tools.shift();
452
+ }
453
+
454
+ // System 2: triage a capability gap before tool-wishlist writes it. `event.input` is
455
+ // documented as mutable, so this patches the report in place rather than duplicating any
456
+ // of the wishlist's authority logic. Nothing here records a decision.
457
+ if (event.toolName === "report_capability_gap" && enabled("gap")) {
458
+ try {
459
+ const result = await broker.request({
460
+ system: "gap",
461
+ state: gap.buildInput({ gap: event.input, existing: [] }),
462
+ questions: gap.questions({ gap: event.input, existing: [] }),
463
+ ctx,
464
+ root: ctx.cwd,
465
+ decide: (answers: any) => {
466
+ const built = gap.decide(answers);
467
+ // Exactly the conditions the caller applies below, so the ledger line says
468
+ // what happened rather than what was available. A gated answer that
469
+ // duplicates a field the model already filled in changed nothing.
470
+ const changes =
471
+ (built.blockForSanitization ? 1 : 0) +
472
+ (built.canonicalKey && typeof event.input?.canonicalKey !== "string" ? 1 : 0) +
473
+ (built.suggestedFix && !event.input?.suggestedFix ? 1 : 0) +
474
+ (built.independentImpact ? 1 : 0);
475
+
476
+ return { applied: changes > 0, decision: built };
477
+ },
478
+ });
479
+ if (!result.ok) {
480
+ return;
481
+ }
482
+
483
+ const advice = result.decision;
484
+ if (advice.blockForSanitization) {
485
+ return {
486
+ block: true,
487
+ reason: "This report appears to contain a credential, an absolute path or other machine-specific detail. Rewrite it with the specifics removed and report it again.",
488
+ };
489
+ }
490
+
491
+ if (advice.canonicalKey && typeof event.input?.canonicalKey !== "string") {
492
+ event.input.canonicalKey = advice.canonicalKey;
493
+ }
494
+
495
+ if (advice.suggestedFix && !event.input?.suggestedFix) {
496
+ event.input.suggestedFix = advice.suggestedFix;
497
+ }
498
+
499
+ // Recorded alongside the model's own claim, never over it: a human reading the
500
+ // wishlist should still see what was originally reported.
501
+ if (advice.independentImpact) {
502
+ event.input.independentImpact = advice.independentImpact;
503
+ }
504
+ } catch {
505
+ return;
506
+ }
507
+
508
+ return;
509
+ }
510
+
511
+ // System 4: order the sources a delegation batch will snapshot. Ordering only — the same
512
+ // set is frozen either way, but a child pages through `list_sources` in this order.
513
+ if (event.toolName === "delegate" && enabled("sources") && Array.isArray(event.input?.sources)) {
514
+ try {
515
+ const candidates = event.input.sources
516
+ .filter((item: unknown) => typeof item === "string")
517
+ .map((item: string) => ({ path: item }));
518
+ if (candidates.length < 2) {
519
+ return;
520
+ }
521
+
522
+ const result = await broker.request({
523
+ system: "sources",
524
+ state: sources.buildInput({ question: event.input?.question ?? objective, candidates }),
525
+ questions: sources.questions({ candidates }),
526
+ ctx,
527
+ root: ctx.cwd,
528
+ decide: (answers: any) => {
529
+ const built = sources.decide(answers, candidates);
530
+ const order = built.ordered.map((item: any) => item.path);
531
+ // An ungated run returns the caller's own order, which is not a change and
532
+ // must not be recorded as one.
533
+ const moved = order.some((item: string, index: number) => item !== candidates[index]?.path);
534
+
535
+ return { applied: moved, decision: { ranked: built, ordered: order } };
536
+ },
537
+ });
538
+ if (!result.ok) {
539
+ return;
540
+ }
541
+
542
+ const { ranked, ordered } = result.decision;
543
+ const missing = event.input.sources.filter((item: string) => !ordered.includes(item));
544
+ event.input.sources = [...ordered, ...missing];
545
+
546
+ // Two answers the same batch already computed and nothing read. Output is free, so
547
+ // they were paid for whether or not anyone looked. A confident "this is not a
548
+ // self-contained evidence question" is worth surfacing before specpi-delegation
549
+ // freezes up to 200 files and 8 MiB for a child that then cannot answer it.
550
+ //
551
+ // Advisory only, and deliberately so: the batch still runs, the ceilings are
552
+ // unchanged, and with no UI this says nothing rather than blocking.
553
+ if (ranked.notWorthDelegating && ctx.hasUI) {
554
+ ctx.ui.notify(
555
+ `Jev rates this a poor fit for delegation${ranked.jobMode ? ` (it reads as ${ranked.jobMode} work)` : ""}. Running anyway; ${ordered.length + missing.length} sources will be frozen for the child.`,
556
+ "warning",
557
+ );
558
+ }
559
+ } catch {
560
+ return;
561
+ }
562
+ }
563
+ });
564
+
565
+ // System 5: notice a session that has stopped making progress, while it can still be helped.
566
+ //
567
+ // THE ONE HANDLER THAT IS NOT AWAITED. Everything else in this file mutates what it inspects --
568
+ // a tool result, a compaction patch, a tool's input -- so the session has to wait for the
569
+ // answer. This one acts on the next turn, and at roughly 300ms a call, awaiting it on a
570
+ // thrashing session would add seconds to an attempt to deliver advice that could not have
571
+ // changed the turn it was asked during.
572
+ pi.on("turn_end", (event: any, ctx: ExtensionContext) => {
573
+ // Kept whether or not the system is on, so switching it on mid-session does not start from
574
+ // a blank history and immediately look healthy.
575
+ history.turnsSinceChange = history.changedThisTurn ? 0 : history.turnsSinceChange + 1;
576
+ if (!enabled("progress") || history.nudged) {
577
+ return;
578
+ }
579
+
580
+ const local = progress.suspicious(history);
581
+ if (!local.ask) {
582
+ return;
583
+ }
584
+
585
+ // Recorded before the call rather than after, so a slow answer cannot let the next turn ask
586
+ // again while this one is still in flight.
587
+ history.askedAtTurn = history.turn;
588
+ void (async () => {
589
+ try {
590
+ const result = await broker.request({
591
+ system: "progress",
592
+ state: progress.buildInput({ history, objective, reasons: local.reasons }),
593
+ questions: progress.questions(),
594
+ ctx,
595
+ root: ctx.cwd,
596
+ decide: (answers: any) => {
597
+ const advice = progress.decide(answers);
598
+
599
+ return {
600
+ applied: Boolean(advice.nudge),
601
+ reason: advice.nudge ? advice.mode : advice.stuck ? "stuck-but-mode-ungated" : "not-stuck",
602
+ decision: advice,
603
+ };
604
+ },
605
+ });
606
+ if (!result.ok || !result.decision?.nudge) {
607
+ return;
608
+ }
609
+
610
+ // Write-once, per the standing rule. A second nudge would either repeat a line the
611
+ // model already has or contradict it, and neither can be withdrawn: it was appended
612
+ // to a prefix that is cached behind it by the time anyone regrets it.
613
+ history.nudged = true;
614
+ if (ctx.hasUI) {
615
+ ctx.ui.notify(
616
+ result.decision.needsHuman
617
+ ? `Jev progress check: this session looks blocked on something only you can answer. ${result.decision.nudge}`
618
+ : `Jev progress check: ${result.decision.nudge}`,
619
+ "warning",
620
+ );
621
+ }
622
+
623
+ // The plan specified `deliverAs: "nextTurn"`, which is documented as "queued for
624
+ // next user prompt, does not interrupt or trigger anything". An unattended session
625
+ // has exactly one user prompt, so a nextTurn message would never be delivered -- in
626
+ // precisely the case the argument for this system rests on, a headless attempt
627
+ // burning its wall clock. "steer" is delivered after the current tool calls finish
628
+ // and before the next model request, which is the same append at the same boundary
629
+ // and is actually read. triggerTurn is left off so this can never add a turn.
630
+ // Suppressed when the session is blocked on something only a person can answer:
631
+ // steering a model past a missing credential costs a turn to say nothing.
632
+ if (
633
+ settings.progressNudge === "message" &&
634
+ !result.decision.needsHuman &&
635
+ typeof pi.sendMessage === "function"
636
+ ) {
637
+ pi.sendMessage(
638
+ {
639
+ customType: "specpi-jev-progress",
640
+ content: result.decision.nudge,
641
+ display: true,
642
+ details: { mode: result.decision.mode, reasons: local.reasons },
643
+ },
644
+ { deliverAs: "steer" },
645
+ );
646
+ }
647
+ } catch {
648
+ // The turn has already ended. An advisor must not be able to fail it retroactively.
649
+ }
650
+ })();
651
+ });
652
+
653
+ pi.registerCommand("jev", {
654
+ description: "Show or change the Jev advisor: master switch, per-system switches and the transmission ledger",
655
+ getArgumentCompletions: (prefix: string) =>
656
+ ["status", "on", "off", "startup", "enable", "disable", "guard", "ledger", "forget"]
657
+ .filter((value) => value.startsWith(prefix.trim().toLowerCase()))
658
+ .map((value) => ({ value, label: value })),
659
+ handler: async (args: string, ctx: ExtensionContext) => {
660
+ const [actionRaw = "status", ...rest] = args.trim().split(/\s+/u).filter(Boolean);
661
+ const action = actionRaw.toLowerCase();
662
+ try {
663
+ if (action === "on" || action === "off") {
664
+ settings = { ...settings, master: action === "on" };
665
+ const active = SYSTEM_NAMES.filter((name) => settings.systems[name]);
666
+ ctx.ui.notify(
667
+ action === "on"
668
+ ? `Jev advisor on for this session with ${active.length} of ${SYSTEM_NAMES.length} systems enabled${active.length === 0 ? " (enable one with /jev enable <system>)" : `: ${active.join(", ")}`}.`
669
+ : "Jev advisor off for this session. No state leaves this machine.",
670
+ "info",
671
+ );
672
+
673
+ return;
674
+ }
675
+
676
+ if (action === "enable" || action === "disable") {
677
+ const names = rest.map((name) => name.toLowerCase());
678
+ const unknown = names.filter((name) => !SYSTEM_NAMES.includes(name));
679
+ if (names.length === 0 || unknown.length > 0) {
680
+ throw new Error(`Usage: /jev ${action} <${SYSTEM_NAMES.join("|")}>`);
681
+ }
682
+
683
+ const systems = { ...settings.systems };
684
+ for (const name of names) {
685
+ systems[name] = action === "enable";
686
+ }
687
+
688
+ settings = { ...settings, systems };
689
+ ctx.ui.notify(
690
+ `${action === "enable" ? "Enabled" : "Disabled"} for this session: ${names.join(", ")}.${settings.master ? "" : " The master switch is still off; run /jev on."}`,
691
+ "info",
692
+ );
693
+
694
+ return;
695
+ }
696
+
697
+ if (action === "startup") {
698
+ const [choice] = rest;
699
+ if (!choice) {
700
+ ctx.ui.notify(
701
+ `Jev starts ${loadSettings().startup ? "on" : "off"} in new sessions. Preference: ${settingsPath()}`,
702
+ "info",
703
+ );
704
+
705
+ return;
706
+ }
707
+
708
+ if (!ctx.hasUI) {
709
+ throw new Error("Startup changes require a human interactive command");
710
+ }
711
+
712
+ if (!["on", "off"].includes(choice.toLowerCase())) {
713
+ throw new Error("Usage: /jev startup [on|off]");
714
+ }
715
+
716
+ const saved = saveSettings({ ...loadSettings(), startup: choice.toLowerCase() === "on" });
717
+ ctx.ui.notify(
718
+ saved.startup
719
+ ? "New Pi sessions will start with the Jev advisor on. This session is unchanged."
720
+ : "New Pi sessions will start with the Jev advisor off. This session is unchanged.",
721
+ "info",
722
+ );
723
+
724
+ return;
725
+ }
726
+
727
+ if (action === "guard") {
728
+ const [verb, choice] = rest.map((value) => value.toLowerCase());
729
+ if (!verb) {
730
+ ctx.ui.notify(guardStatusLine(), "info");
731
+
732
+ return;
733
+ }
734
+
735
+ if (verb === "on" || verb === "off") {
736
+ guardEnabled = verb === "on";
737
+ const result = syncGuard();
738
+ ctx.ui.notify(
739
+ result.reason === "not-installed"
740
+ ? "specpi-jev-guard is not installed, so there is nothing to switch. Command policy stays with the permission system."
741
+ : guardEnabled
742
+ ? "Jev guard on for this session. It scores shell and file calls and defers to the permission system whenever Jev is unavailable or unconfident."
743
+ : "Jev guard off for this session. Every tool call goes straight to the permission system.",
744
+ "info",
745
+ );
746
+
747
+ return;
748
+ }
749
+
750
+ if (verb !== "startup") {
751
+ throw new Error("Usage: /jev guard [on|off|startup [on|off]]");
752
+ }
753
+
754
+ if (!choice) {
755
+ ctx.ui.notify(
756
+ `The Jev guard starts ${loadSettings().guard.startup ? "on" : "off"} in new sessions.`,
757
+ "info",
758
+ );
759
+
760
+ return;
761
+ }
762
+
763
+ if (!ctx.hasUI) {
764
+ throw new Error("Startup changes require a human interactive command");
765
+ }
766
+
767
+ if (!["on", "off"].includes(choice)) {
768
+ throw new Error("Usage: /jev guard startup [on|off]");
769
+ }
770
+
771
+ const stored = loadSettings();
772
+ const saved = saveSettings({
773
+ ...stored,
774
+ guard: { ...stored.guard, startup: choice === "on" },
775
+ });
776
+ ctx.ui.notify(
777
+ saved.guard.startup
778
+ ? "New Pi sessions will start with the Jev guard on. This session is unchanged."
779
+ : "New Pi sessions will start with the Jev guard off. This session is unchanged.",
780
+ "info",
781
+ );
782
+
783
+ return;
784
+ }
785
+
786
+ if (action === "forget") {
787
+ if (!ctx.hasUI) {
788
+ throw new Error("Revoking consent requires a human interactive command");
789
+ }
790
+
791
+ revokeConsent();
792
+ ctx.ui.notify(
793
+ "Forgot the Jev transmission consent. The next system that would send anything will ask again.",
794
+ "info",
795
+ );
796
+
797
+ return;
798
+ }
799
+
800
+ if (action === "ledger") {
801
+ const limit = Number.parseInt(rest[0] ?? "10", 10);
802
+ const entries = readLedger(Number.isInteger(limit) ? limit : 10);
803
+ if (entries.length === 0) {
804
+ ctx.ui.notify(`No Jev transmissions recorded. Ledger: ${ledgerPath()}`, "info");
805
+
806
+ return;
807
+ }
808
+
809
+ const lines = entries.map(
810
+ (entry: any) =>
811
+ `${entry.at} ${entry.system} ${entry.stateBytes}B ${entry.ok ? `${entry.latencyMs}ms` : entry.reason} ${String(entry.payloadSha256 ?? "").slice(0, 12)} [${(entry.questionKeys ?? []).join(", ")}]`,
812
+ );
813
+ ctx.ui.notify(`${lines.join("\n")}\n\nLedger: ${ledgerPath()}`, "info");
814
+
815
+ return;
816
+ }
817
+
818
+ if (action !== "status") {
819
+ throw new Error(
820
+ "Usage: /jev [status|on|off|startup [on|off]|enable <system>|disable <system>|guard [on|off|startup [on|off]]|ledger [n]|forget]",
821
+ );
822
+ }
823
+
824
+ const state = broker.status();
825
+ const lines = [
826
+ `master: ${settings.master ? "on" : "off"} (new sessions start ${loadSettings().startup ? "on" : "off"})`,
827
+ ...SYSTEM_NAMES.map((name) => ` ${name}: ${settings.systems[name] ? "on" : "off"}`),
828
+ // Both names, because the default route is OpenRouter and naming only the
829
+ // other one sends a reader to set the key that returns a bare 401.
830
+ `key: ${keyPresent() ? "present" : "missing"} (OPENROUTER_API_KEY, or TYPESAFE_API_KEY with JEV_BACKEND=typesafe)`,
831
+ `consent: ${granted() ? "granted" : "not granted"}`,
832
+ `calls this session: ${state.callsUsed}/${state.budgets.total} total`,
833
+ ...SYSTEM_NAMES.map((name) => ` ${name}: ${state.usedBySystem[name] ?? 0}/${state.budgets[name]}`),
834
+ guardStatusLine(),
835
+ `settings: ${settingsPath()}`,
836
+ `consent file: ${consentPath()}`,
837
+ `ledger: ${ledgerPath()}`,
838
+ // Named here because it is the one file another process is meant to read, and
839
+ // SpecPi Chat showing a number nobody can find is how a number stops being
840
+ // checkable.
841
+ `session counts: ${usagePath()}`,
842
+ ];
843
+ ctx.ui.notify(lines.join("\n"), "info");
844
+ } catch (error) {
845
+ ctx.ui.notify(safeMessage(error), "error");
846
+ }
847
+ },
848
+ });
849
+ }