specpi 0.28.0 → 0.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +45 -0
- package/NPM_RELEASE.md +1 -1
- package/README.md +22 -23
- package/SECURITY_MODEL.md +14 -12
- package/THIRD_PARTY.md +6 -5
- package/extensions/jev-advisor/broker.mjs +2 -4
- package/extensions/jev-advisor/config.mjs +91 -62
- package/extensions/jev-advisor/gate.mjs +5 -30
- package/extensions/jev-advisor/index.ts +8 -260
- package/extensions/jev-advisor/key-source.mjs +1 -1
- package/extensions/jev-advisor/layer.mjs +8 -31
- package/package.json +2 -1
- package/scripts/jev-guard.mjs +154 -0
- package/scripts/packages.mjs +0 -56
- package/scripts/specpi.mjs +24 -52
- package/templates/settings.json +2 -1
- package/extensions/jev-advisor/questions/compaction.mjs +0 -153
- package/extensions/jev-advisor/questions/guard.mjs +0 -168
- package/extensions/jev-advisor/risk.mjs +0 -442
|
@@ -5,6 +5,10 @@
|
|
|
5
5
|
// name and never by value, because "how do I configure this" is a question asked before enabling
|
|
6
6
|
// anything -- see key-source.mjs.
|
|
7
7
|
//
|
|
8
|
+
// The command guard is not here at all. It is a separate pinned package that keeps its own
|
|
9
|
+
// configuration and its own switch, and nothing in this file tracks it -- see scripts/jev-guard.mjs
|
|
10
|
+
// for the one place SpecPi touches it, which is install time.
|
|
11
|
+
//
|
|
8
12
|
// The file is SpecPi's own, hardened the same way as web-access and capability-policy: atomic
|
|
9
13
|
// write, mode 0600, symlinks refused, and a missing or unreadable file read as off.
|
|
10
14
|
|
|
@@ -14,23 +18,7 @@ import path from "node:path";
|
|
|
14
18
|
import { randomUUID } from "node:crypto";
|
|
15
19
|
|
|
16
20
|
/** Systems that may run inside a session. Offline scripts are not gated here. */
|
|
17
|
-
export const SYSTEM_NAMES = Object.freeze([
|
|
18
|
-
"retention",
|
|
19
|
-
"compaction",
|
|
20
|
-
"gap",
|
|
21
|
-
"sources",
|
|
22
|
-
"progress",
|
|
23
|
-
"untrusted",
|
|
24
|
-
"capability",
|
|
25
|
-
// The command guard, native since schema 3. It used to be a separate pinned package with its own
|
|
26
|
-
// global configuration file, which is why it used to carry its own switch here: a second switch,
|
|
27
|
-
// outside the master, with its own startup preference and its own persistence rules. Those rules
|
|
28
|
-
// disagreed with the layer's often enough to be their own source of defects -- a preference
|
|
29
|
-
// erased by a command that had decided nothing about the guard, a state written by one command
|
|
30
|
-
// and reverted by the next session. As a system it is gated, budgeted, reported and toggled by
|
|
31
|
-
// exactly the same code as the other seven.
|
|
32
|
-
"guard",
|
|
33
|
-
]);
|
|
21
|
+
export const SYSTEM_NAMES = Object.freeze(["retention", "gap", "sources", "progress", "untrusted", "capability"]);
|
|
34
22
|
|
|
35
23
|
/**
|
|
36
24
|
* What a confident stuck verdict is allowed to do. `notify` tells the person and cannot be wrong in
|
|
@@ -52,8 +40,8 @@ export const SYSTEM_NAMES = Object.freeze([
|
|
|
52
40
|
export const NUDGE_MODES = Object.freeze(["notify", "message"]);
|
|
53
41
|
|
|
54
42
|
const MAX_SETTINGS_BYTES = 4096;
|
|
55
|
-
const MAX_CALL_BUDGET =
|
|
56
|
-
const MAX_TOTAL_BUDGET =
|
|
43
|
+
const MAX_CALL_BUDGET = 1024;
|
|
44
|
+
const MAX_TOTAL_BUDGET = 2048;
|
|
57
45
|
|
|
58
46
|
/**
|
|
59
47
|
* One shared budget could not survive a turn-level system. A system that fires once per turn would
|
|
@@ -61,16 +49,19 @@ const MAX_TOTAL_BUDGET = 512;
|
|
|
61
49
|
* session, and which one won would be decided by event ordering rather than by anyone's policy.
|
|
62
50
|
*
|
|
63
51
|
* So the ceiling is two-level: each system gets its own, and the total is a real constraint because
|
|
64
|
-
* it is deliberately less than their sum
|
|
65
|
-
*
|
|
52
|
+
* it is deliberately less than their sum -- 2048 against 2274. That relationship is the invariant,
|
|
53
|
+
* not either number: raising the total without raising the per-system ceilings would leave a total
|
|
54
|
+
* no combination of systems could ever reach, which is a limit that reads as a limit and is not
|
|
55
|
+
* one. `tests/jev-advisor.test.mjs` pins the inequality so a future change to one has to consider
|
|
56
|
+
* the other. Running out of one system's budget stops that system and nothing else.
|
|
66
57
|
*
|
|
67
58
|
* The per-system numbers follow how often each one can fire: retention on every large read-only
|
|
68
|
-
* result,
|
|
59
|
+
* result, gap per report, sources per delegation batch.
|
|
69
60
|
*/
|
|
70
61
|
export const DEFAULT_BUDGETS = Object.freeze({
|
|
71
62
|
// A backstop, not a working limit, and the number says which. Measured, a full tier-3 task -- a
|
|
72
63
|
// 120-step repair chain over about 25 model requests -- spends 4 to 7 calls, and the busiest
|
|
73
|
-
// attempt ever recorded spent 12. A session would have to run for days before
|
|
64
|
+
// attempt ever recorded spent 12. A session would have to run for days before 2048 bound
|
|
74
65
|
// anything a person was actually doing, which is the point: the ceiling should only ever be hit
|
|
75
66
|
// by a loop, and hitting it should therefore be information rather than an inconvenience.
|
|
76
67
|
//
|
|
@@ -78,35 +69,36 @@ export const DEFAULT_BUDGETS = Object.freeze({
|
|
|
78
69
|
// runs for two minutes; an interactive session runs for a day, and a turn-level system at one
|
|
79
70
|
// call every four turns reaches 120 somewhere in the afternoon and then goes quiet without
|
|
80
71
|
// having found anything wrong. A ceiling that a normal long session reaches is not protecting
|
|
81
|
-
// anyone, it is just failing later than it looks.
|
|
72
|
+
// anyone, it is just failing later than it looks. 512 was the same mistake one order of
|
|
73
|
+
// magnitude further out, and the long-session evals are what showed it: a session long enough
|
|
74
|
+
// to compact several times spends in the hundreds, so the backstop sat close enough to real
|
|
75
|
+
// use to be reachable by a session that was working correctly.
|
|
82
76
|
//
|
|
83
|
-
// Cost is not what these are for. A call is about $0.00003, so the whole total is about
|
|
84
|
-
//
|
|
85
|
-
//
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
total:
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
77
|
+
// Cost is not what these are for. A call is about $0.00003, so the whole total is about six
|
|
78
|
+
// cents. They bound two things that do not get cheaper with scale: how much digest leaves the
|
|
79
|
+
// machine for a third party, at up to 1 KB a call, and how much awaited latency a runaway loop
|
|
80
|
+
// can add before something stops it. Two megabytes of digest and an announced stop is the shape
|
|
81
|
+
// of the trade.
|
|
82
|
+
total: 2048,
|
|
83
|
+
// The per-system numbers move with the total, because a total the per-system ceilings can never
|
|
84
|
+
// add up to is not a constraint at all -- see below. They are scaled rather than re-derived:
|
|
85
|
+
// each one's rationale is a firing frequency, and none of those frequencies changed.
|
|
86
|
+
retention: 832,
|
|
87
|
+
gap: 192,
|
|
88
|
+
sources: 128,
|
|
93
89
|
// Turn-level, but gated behind local signals and a four-turn cooldown, so it only spends on
|
|
94
90
|
// sessions that already look wrong. The ceiling is what stops a genuinely thrashing session
|
|
95
91
|
// from spending the total on being told it is thrashing.
|
|
96
|
-
progress:
|
|
92
|
+
progress: 704,
|
|
97
93
|
// Usually free: when retention is on, system 7's question rides the call retention was already
|
|
98
94
|
// making against the same state. This ceiling only binds when retention is off, or when the
|
|
99
95
|
// fetched result is too small for retention to be interested in it.
|
|
100
|
-
untrusted:
|
|
96
|
+
untrusted: 416,
|
|
101
97
|
// Once per session by construction, and only when local signals already suggest it. Two rather
|
|
102
|
-
// than one so a retried first turn is not silently un-served.
|
|
98
|
+
// than one so a retried first turn is not silently un-served. The one number here that does not
|
|
99
|
+
// scale with the total, because it is not sized by a frequency: the system asks once and then
|
|
100
|
+
// never again, so a larger ceiling would buy nothing and would only misdescribe what it does.
|
|
103
101
|
capability: 2,
|
|
104
|
-
// Per gated tool call that local rules could not settle, so its frequency is retention's rather
|
|
105
|
-
// than compaction's -- and like retention, most calls never reach it: read-only commands and
|
|
106
|
-
// ordinary project writes are answered locally for nothing. Running out means the guard defers
|
|
107
|
-
// to the permission system for the rest of the session, which is what it does for every other
|
|
108
|
-
// kind of unavailability.
|
|
109
|
-
guard: 208,
|
|
110
102
|
});
|
|
111
103
|
|
|
112
104
|
/** 0 is a real budget meaning no calls. Switching a system off is what `systems[name] = false` is for. */
|
|
@@ -143,7 +135,7 @@ export function regularFile(file, label) {
|
|
|
143
135
|
/** Every unknown shape collapses to the same all-off default rather than a partial enable. */
|
|
144
136
|
export function defaultSettings() {
|
|
145
137
|
return {
|
|
146
|
-
schema:
|
|
138
|
+
schema: 5,
|
|
147
139
|
master: false,
|
|
148
140
|
startup: false,
|
|
149
141
|
systems: Object.fromEntries(SYSTEM_NAMES.map((name) => [name, false])),
|
|
@@ -187,36 +179,73 @@ function migrate(raw) {
|
|
|
187
179
|
}
|
|
188
180
|
|
|
189
181
|
/**
|
|
190
|
-
* Schema
|
|
191
|
-
*
|
|
192
|
-
*
|
|
182
|
+
* Schema 4 removes every trace of the command guard from this file, from either shape it took.
|
|
183
|
+
*
|
|
184
|
+
* Schema 2 kept a `guard: { enabled, startup }` pair, from when SpecPi re-asserted the package's
|
|
185
|
+
* switch at every session start; schema 3 made the guard the layer's eighth system under
|
|
186
|
+
* `systems.guard`. Both keys go, because neither answer is SpecPi's to hold any more: the package
|
|
187
|
+
* owns its switch in its own file, and a stale copy here could only ever disagree with it.
|
|
188
|
+
*
|
|
189
|
+
* Dropping them changes nothing about whether anyone's guard is on. A schema-2 file's preference
|
|
190
|
+
* had already been written through to the package's own configuration by the last session that
|
|
191
|
+
* read it, and schema 3 shipped with no guard package installed at all.
|
|
192
|
+
*/
|
|
193
|
+
function migrateToFour(raw) {
|
|
194
|
+
const { guard: _pair, ...rest } = raw ?? {};
|
|
195
|
+
const { guard: _system, ...systems } = rest.systems ?? {};
|
|
196
|
+
|
|
197
|
+
return { ...rest, schema: 4, systems };
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Schema 5 removes compaction guidance, which was withdrawn rather than fixed.
|
|
202
|
+
*
|
|
203
|
+
* Two tier-6 runs, on SpecPi 0.28.0 and 0.29.0, both measured the arm carrying it solving fewer
|
|
204
|
+
* long-session tasks than plain SpecPi: 14/16 against 8/16 pooled, Fisher exact p = 0.054. Over the
|
|
205
|
+
* same attempts compaction was 55 of 56 applied verdicts, so the arm's behaviour was almost entirely
|
|
206
|
+
* this system's, and no other system in the layer applied enough to be a candidate.
|
|
193
207
|
*
|
|
194
|
-
*
|
|
195
|
-
*
|
|
196
|
-
*
|
|
197
|
-
*
|
|
198
|
-
*
|
|
199
|
-
*
|
|
208
|
+
* The evidence never reached significance and the mechanism was never isolated. The system was
|
|
209
|
+
* removed anyway, because a system that steers a summary has to earn the risk it takes, and one
|
|
210
|
+
* whose only measurement says it costs solve rate has not.
|
|
211
|
+
*
|
|
212
|
+
* Both the `systems.compaction` switch and the per-system `budgets.compaction` ceiling go. Dropping
|
|
213
|
+
* them turns nothing off that a user had on in any meaningful sense: the hooks they gated no longer
|
|
214
|
+
* exist, so a retained preference could only describe a system that cannot run.
|
|
200
215
|
*/
|
|
201
|
-
function
|
|
202
|
-
const
|
|
216
|
+
function migrateToFive(raw) {
|
|
217
|
+
const { compaction: _system, ...systems } = raw?.systems ?? {};
|
|
218
|
+
const { compaction: _budget, ...budgets } = raw?.budgets ?? {};
|
|
203
219
|
|
|
204
|
-
|
|
220
|
+
return { ...raw, schema: 5, systems, budgets };
|
|
221
|
+
}
|
|
205
222
|
|
|
206
|
-
|
|
223
|
+
/**
|
|
224
|
+
* What the advisor will read, given a settings object, without writing it anywhere.
|
|
225
|
+
*
|
|
226
|
+
* Exported for callers that compose a settings file for somewhere other than this process's own
|
|
227
|
+
* agent directory -- the eval harness writes one into a disposable home -- and need to check what
|
|
228
|
+
* they composed. Building a literal and trusting it is how the harness came to run every published
|
|
229
|
+
* tier with a system it believed it had enabled switched off: it hardcoded a schema number, the
|
|
230
|
+
* migration for that number read a preference from a key that shape does not have, and nothing ever
|
|
231
|
+
* compared the result to the ask. Compose, then normalize, then assert -- not compose and trust.
|
|
232
|
+
*/
|
|
233
|
+
export function normalizeSettings(raw) {
|
|
234
|
+
return normalize(raw);
|
|
207
235
|
}
|
|
208
236
|
|
|
209
237
|
function normalize(raw) {
|
|
210
238
|
const one = raw?.schema === 1 ? migrate(raw) : raw;
|
|
211
|
-
const
|
|
212
|
-
|
|
239
|
+
const four = one?.schema === 2 || one?.schema === 3 ? migrateToFour(one) : one;
|
|
240
|
+
const source = four?.schema === 4 ? migrateToFive(four) : four;
|
|
241
|
+
if (source?.schema !== 5) {
|
|
213
242
|
return defaultSettings();
|
|
214
243
|
}
|
|
215
244
|
|
|
216
245
|
const systems = Object.fromEntries(SYSTEM_NAMES.map((name) => [name, source.systems?.[name] === true]));
|
|
217
246
|
|
|
218
247
|
return {
|
|
219
|
-
schema:
|
|
248
|
+
schema: 5,
|
|
220
249
|
master: source.master === true,
|
|
221
250
|
startup: source.startup === true,
|
|
222
251
|
systems,
|
|
@@ -252,9 +281,9 @@ export function writeFileAtomic(file, contents) {
|
|
|
252
281
|
}
|
|
253
282
|
|
|
254
283
|
export function saveSettings(settings) {
|
|
255
|
-
// A caller handing back
|
|
284
|
+
// A caller handing back an older shape is migrated rather than reset, so a round trip through
|
|
256
285
|
// an old reader cannot quietly disable the layer.
|
|
257
|
-
const next = normalize(
|
|
286
|
+
const next = normalize([1, 2, 3, 4].includes(settings?.schema) ? settings : { ...settings, schema: 5 });
|
|
258
287
|
const file = settingsFile();
|
|
259
288
|
if (fs.existsSync(file)) {
|
|
260
289
|
regularFile(file, "Jev settings");
|
|
@@ -50,12 +50,11 @@
|
|
|
50
50
|
// was unreachable, and running the layer could never have revealed it, because a system that
|
|
51
51
|
// never fires looks exactly like a system whose advice was always to do nothing.
|
|
52
52
|
//
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
//
|
|
57
|
-
//
|
|
58
|
-
// and because a firing there blocks a write.
|
|
53
|
+
// Compaction guidance had the same problem and was lowered to 0.60 to clear it. That system has
|
|
54
|
+
// since been withdrawn -- two tier-6 runs measured the arm carrying it solving fewer long-session
|
|
55
|
+
// tasks than plain SpecPi -- so the loosest gate in the layer is gone with it. gap keeps 0.85
|
|
56
|
+
// because its Noul reaches 0.96 on the case that matters and because a firing there blocks a
|
|
57
|
+
// write.
|
|
59
58
|
|
|
60
59
|
/**
|
|
61
60
|
* The figures the comment above cites, in a form a test can check against the artifact. A citation
|
|
@@ -91,16 +90,6 @@ export const THRESHOLDS = Object.freeze({
|
|
|
91
90
|
// Measured 0.06-0.07 on the spent case, so this clears with room.
|
|
92
91
|
low: 0.1,
|
|
93
92
|
}),
|
|
94
|
-
compaction: Object.freeze({
|
|
95
|
-
scoreConfidence: 0.6,
|
|
96
|
-
boundary: 0.3,
|
|
97
|
-
choiceConfidence: 0.75,
|
|
98
|
-
margin: 0.25,
|
|
99
|
-
// Lowered from 0.85: the clearest open investigation scores 0.63-0.65, and the action is
|
|
100
|
-
// one sentence added to a prompt that is being rebuilt regardless.
|
|
101
|
-
high: 0.6,
|
|
102
|
-
low: 0.15,
|
|
103
|
-
}),
|
|
104
93
|
gap: Object.freeze({
|
|
105
94
|
scoreConfidence: 0.6,
|
|
106
95
|
boundary: 0.3,
|
|
@@ -156,20 +145,6 @@ export const THRESHOLDS = Object.freeze({
|
|
|
156
145
|
high: 0.85,
|
|
157
146
|
low: 0.15,
|
|
158
147
|
}),
|
|
159
|
-
guard: Object.freeze({
|
|
160
|
-
// The same measured Score operating point as every other system, written out rather than
|
|
161
|
-
// inherited by falling through `thresholdsFor`'s default. The guard asked under the name
|
|
162
|
-
// "gap" for exactly as long as it took to notice that adding a `guard` entry would then have
|
|
163
|
-
// changed nothing -- a silent no-op on the one system whose action takes a tool call away.
|
|
164
|
-
scoreConfidence: 0.6,
|
|
165
|
-
boundary: 0.3,
|
|
166
|
-
choiceConfidence: 0.75,
|
|
167
|
-
margin: 0.2,
|
|
168
|
-
// Unused: this system reads only the Score side. Kept so every system has a full set, and
|
|
169
|
-
// so the calibration tests cover this entry like any other.
|
|
170
|
-
high: 0.85,
|
|
171
|
-
low: 0.15,
|
|
172
|
-
}),
|
|
173
148
|
sources: Object.freeze({
|
|
174
149
|
scoreConfidence: 0.6,
|
|
175
150
|
boundary: 0.3,
|
|
@@ -2,55 +2,24 @@ import fs from "node:fs";
|
|
|
2
2
|
import { type ExtensionAPI, type ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
3
3
|
import { SYSTEM_NAMES, loadSettings, saveSettings, settingsPath } from "./config.mjs";
|
|
4
4
|
import { keySources } from "./key-source.mjs";
|
|
5
|
-
import { applyLayer,
|
|
5
|
+
import { applyLayer, layerScopeLine, layerToPersist, startupToPersist } from "./layer.mjs";
|
|
6
6
|
import { consentPath, granted, revokeConsent } from "./consent.mjs";
|
|
7
7
|
import { createBroker } from "./broker.mjs";
|
|
8
8
|
import { ledgerPath, read as readLedger } from "./ledger.mjs";
|
|
9
9
|
import { usagePath } from "./usage.mjs";
|
|
10
|
-
import { GATED_TOOLS, SHELL_TOOLS, callTargets, classifyCall, commandText } from "./risk.mjs";
|
|
11
10
|
import * as retention from "./questions/retention.mjs";
|
|
12
|
-
import * as compaction from "./questions/compaction.mjs";
|
|
13
11
|
import * as gap from "./questions/gap.mjs";
|
|
14
12
|
import * as sources from "./questions/sources.mjs";
|
|
15
13
|
import * as progress from "./questions/progress.mjs";
|
|
16
14
|
import * as untrusted from "./questions/untrusted.mjs";
|
|
17
15
|
import * as capabilities from "./questions/capabilities.mjs";
|
|
18
|
-
import * as guard from "./questions/guard.mjs";
|
|
19
16
|
|
|
20
17
|
const MAX_RECENT = 8;
|
|
21
18
|
|
|
22
|
-
/** One line of a call, for a notification or a block reason. Never a digest; never sent anywhere. */
|
|
23
|
-
function short(value: string, limit: number) {
|
|
24
|
-
const text = String(value ?? "")
|
|
25
|
-
.replace(/\s+/gu, " ")
|
|
26
|
-
.trim();
|
|
27
|
-
|
|
28
|
-
return text.length > limit ? `${text.slice(0, limit - 1)}…` : text;
|
|
29
|
-
}
|
|
30
|
-
|
|
31
19
|
function safeMessage(error: unknown) {
|
|
32
20
|
return String((error as any)?.message ?? error ?? "unknown error").slice(0, 200);
|
|
33
21
|
}
|
|
34
22
|
|
|
35
|
-
/**
|
|
36
|
-
* Tell the person something, and never let the telling change what happens.
|
|
37
|
-
*
|
|
38
|
-
* `ctx.ui.notify` reaches the host over RPC and can throw -- a disconnected client, a torn-down UI,
|
|
39
|
-
* a host without the method. Called inline inside the guard's fail-open catch, one such throw
|
|
40
|
-
* unwound a decided refusal into an allow, so the announcement is isolated from the decision here.
|
|
41
|
-
*/
|
|
42
|
-
function announce(ctx: ExtensionContext, message: string) {
|
|
43
|
-
if (!ctx.hasUI) {
|
|
44
|
-
return;
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
try {
|
|
48
|
-
ctx.ui.notify(message, "error");
|
|
49
|
-
} catch {
|
|
50
|
-
// A failed notification is not a reason to run a command, or not to.
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
|
|
54
23
|
export default function jevAdvisor(pi: ExtensionAPI) {
|
|
55
24
|
// Session switches live in memory. A session toggle must never write the startup preference,
|
|
56
25
|
// so the saved file is read once per session and only /jev startup ever writes it.
|
|
@@ -254,111 +223,6 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
254
223
|
}
|
|
255
224
|
});
|
|
256
225
|
|
|
257
|
-
// System 8: the command guard, before a shell or file call runs.
|
|
258
|
-
//
|
|
259
|
-
// Fail open at every step. Local triage settles most calls for nothing; anything else is asked
|
|
260
|
-
// about, and a call is blocked only on a confident verdict that the request does not account
|
|
261
|
-
// for. Every other outcome -- no key, no budget, a timeout, an unconfident answer, no human to
|
|
262
|
-
// ask -- returns the call to @gotgenes/pi-permission-system, which decides it exactly as it did
|
|
263
|
-
// before this layer existed. The package this replaced was fail-closed, so an outage or a
|
|
264
|
-
// missing key stopped work; that is the single behaviour most worth not reproducing.
|
|
265
|
-
pi.on("tool_call", async (event: any, ctx: ExtensionContext) => {
|
|
266
|
-
if (!enabled("guard") || !GATED_TOOLS.includes(event?.toolName)) {
|
|
267
|
-
return undefined;
|
|
268
|
-
}
|
|
269
|
-
|
|
270
|
-
const shell = SHELL_TOOLS.includes(event.toolName);
|
|
271
|
-
// Not `input.command`: `write_stdin` types into a live shell under another name, so reading
|
|
272
|
-
// one key classified every such call as the empty string -- spending a guard call on nothing
|
|
273
|
-
// while the text actually being run went unexamined.
|
|
274
|
-
const command = shell ? commandText(event?.input) : "";
|
|
275
|
-
// Every file the call names, because `multi_edit` and `apply_patch` do not carry one `path`
|
|
276
|
-
// and a target the guard cannot see is a target it never asks the credential question about.
|
|
277
|
-
const targets = callTargets(event?.input);
|
|
278
|
-
const local = classifyCall({ tool: event.toolName, command, targets, cwd: ctx.cwd });
|
|
279
|
-
const subject = shell
|
|
280
|
-
? command || "(command unknown)"
|
|
281
|
-
: `${event.toolName} ${targets.join(", ") || "(target unknown)"}`;
|
|
282
|
-
|
|
283
|
-
// Built once, and nothing inside it may throw. A refusal that has already been decided must
|
|
284
|
-
// reach the harness: an exception raised while announcing it would unwind into the fail-open
|
|
285
|
-
// catch below and turn the layer's only blocking action into an allow.
|
|
286
|
-
const refuse = (reason: string) => {
|
|
287
|
-
announce(ctx, `Jev guard blocked ${event.toolName}: ${reason}.`);
|
|
288
|
-
|
|
289
|
-
return { block: true, reason: `Jev guard: ${reason}. Call: ${short(subject, 160)}` };
|
|
290
|
-
};
|
|
291
|
-
|
|
292
|
-
if (local.decision === "safe") {
|
|
293
|
-
return undefined;
|
|
294
|
-
}
|
|
295
|
-
|
|
296
|
-
if (local.decision === "dangerous") {
|
|
297
|
-
// Catastrophic and unambiguous, so it needs neither a network call nor a human. This is
|
|
298
|
-
// the one path that blocks without asking Jev, which is why its rule list is tiny.
|
|
299
|
-
return refuse(local.reason);
|
|
300
|
-
}
|
|
301
|
-
|
|
302
|
-
let verdict;
|
|
303
|
-
try {
|
|
304
|
-
const result = await broker.request({
|
|
305
|
-
system: "guard",
|
|
306
|
-
state: guard.buildInput({
|
|
307
|
-
tool: event.toolName,
|
|
308
|
-
subject,
|
|
309
|
-
protectedTarget: local.reason === "writes to a protected path",
|
|
310
|
-
objective,
|
|
311
|
-
recent,
|
|
312
|
-
cwd: ctx.cwd,
|
|
313
|
-
}),
|
|
314
|
-
questions: guard.questions({ protected: local.reason === "writes to a protected path" }),
|
|
315
|
-
ctx,
|
|
316
|
-
root: ctx.cwd,
|
|
317
|
-
decide: (answers: any) => {
|
|
318
|
-
const verdict = guard.decide(answers, { hasUI: ctx.hasUI });
|
|
319
|
-
|
|
320
|
-
return { applied: verdict.action !== "defer", decision: verdict };
|
|
321
|
-
},
|
|
322
|
-
});
|
|
323
|
-
if (!result.ok) {
|
|
324
|
-
return undefined;
|
|
325
|
-
}
|
|
326
|
-
|
|
327
|
-
verdict = result.decision;
|
|
328
|
-
} catch {
|
|
329
|
-
// An advisor must never be the reason a tool call fails. Anything unexpected while
|
|
330
|
-
// asking hands the call back to the permission system unchanged. The catch ends here, so
|
|
331
|
-
// that everything the verdict then decides is outside it.
|
|
332
|
-
return undefined;
|
|
333
|
-
}
|
|
334
|
-
|
|
335
|
-
if (verdict.action === "block") {
|
|
336
|
-
return refuse(verdict.reason);
|
|
337
|
-
}
|
|
338
|
-
|
|
339
|
-
if (verdict.action === "ask" && ctx.hasUI) {
|
|
340
|
-
let choice;
|
|
341
|
-
try {
|
|
342
|
-
choice = await ctx.ui.select({
|
|
343
|
-
title: "Jev guard",
|
|
344
|
-
message: `This looks ${verdict.reason}: ${short(subject, 300)}`,
|
|
345
|
-
options: [guard.CHOICES.run, guard.CHOICES.block],
|
|
346
|
-
});
|
|
347
|
-
} catch {
|
|
348
|
-
// The one failure in this file that does not fail open, and deliberately. Reaching
|
|
349
|
-
// here means the verdict already said this call needs a person's approval; a host
|
|
350
|
-
// that cannot ask has not obtained it, and an unanswerable question resolved as yes
|
|
351
|
-
// is the failure mode a confirmation dialog exists to rule out.
|
|
352
|
-
return refuse("this needs your approval and you could not be asked");
|
|
353
|
-
}
|
|
354
|
-
|
|
355
|
-
// `guard.approved` owns the rule; see it for why every non-answer is a refusal.
|
|
356
|
-
return guard.approved(choice) ? undefined : refuse("not approved by you");
|
|
357
|
-
}
|
|
358
|
-
|
|
359
|
-
return undefined;
|
|
360
|
-
});
|
|
361
|
-
|
|
362
226
|
// System 1: condense a spent tool result before it is appended. Doing this after the fact would
|
|
363
227
|
// rewrite a cached prefix; on arrival it never touches one.
|
|
364
228
|
pi.on("tool_result", async (event: any, ctx: ExtensionContext) => {
|
|
@@ -379,11 +243,10 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
379
243
|
}
|
|
380
244
|
}
|
|
381
245
|
|
|
382
|
-
// Every result, not only the ones retention asked about.
|
|
383
|
-
//
|
|
384
|
-
//
|
|
385
|
-
//
|
|
386
|
-
// whole length while the question set said history was what the intent answer weighed.
|
|
246
|
+
// Every result, not only the ones retention asked about. It was written in one place --
|
|
247
|
+
// inside retention's success path -- so a session with retention off, or with retention's
|
|
248
|
+
// budget spent, handed every other system an empty history for its whole length while
|
|
249
|
+
// their question sets said history was what they weighed.
|
|
387
250
|
recent.push({ tool: String(event?.toolName ?? ""), outcome: event?.isError === true ? "error" : "ok" });
|
|
388
251
|
if (recent.length > MAX_RECENT) {
|
|
389
252
|
recent.shift();
|
|
@@ -470,115 +333,6 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
470
333
|
}
|
|
471
334
|
});
|
|
472
335
|
|
|
473
|
-
// System 1b: steer the summary at the one boundary where the prompt cache is discarded anyway.
|
|
474
|
-
// Only customInstructions is supplied; the preparation's own cut and budget are left alone.
|
|
475
|
-
pi.on("session_before_compact", async (event: any, ctx: ExtensionContext) => {
|
|
476
|
-
if (!enabled("compaction")) {
|
|
477
|
-
return;
|
|
478
|
-
}
|
|
479
|
-
|
|
480
|
-
try {
|
|
481
|
-
const result = await broker.request({
|
|
482
|
-
system: "compaction",
|
|
483
|
-
state: compaction.buildInput({ preparation: event.preparation, objective }),
|
|
484
|
-
questions: compaction.questions(),
|
|
485
|
-
ctx,
|
|
486
|
-
root: ctx.cwd,
|
|
487
|
-
signal: event.signal,
|
|
488
|
-
decide: (answers: any) => {
|
|
489
|
-
const built = compaction.decide(answers);
|
|
490
|
-
|
|
491
|
-
// Nothing is shortened here, so savedBytes stays 0 and `applied` is the whole
|
|
492
|
-
// record: either a sentence reached the summariser or Pi's own prompt ran.
|
|
493
|
-
return { applied: Boolean(built.customInstructions), decision: built };
|
|
494
|
-
},
|
|
495
|
-
});
|
|
496
|
-
if (!result.ok) {
|
|
497
|
-
return;
|
|
498
|
-
}
|
|
499
|
-
|
|
500
|
-
const advice = result.decision;
|
|
501
|
-
if (!advice.customInstructions) {
|
|
502
|
-
return;
|
|
503
|
-
}
|
|
504
|
-
|
|
505
|
-
const existing = typeof event.customInstructions === "string" ? event.customInstructions.trim() : "";
|
|
506
|
-
|
|
507
|
-
return {
|
|
508
|
-
customInstructions: existing
|
|
509
|
-
? `${existing}\n\n${advice.customInstructions}`
|
|
510
|
-
: advice.customInstructions,
|
|
511
|
-
};
|
|
512
|
-
} catch {
|
|
513
|
-
return;
|
|
514
|
-
}
|
|
515
|
-
});
|
|
516
|
-
|
|
517
|
-
// System 1b, second hook. Branch summarisation is the same problem at the same boundary --
|
|
518
|
-
// something is about to be reduced to a summary and the prefix is being rebuilt regardless --
|
|
519
|
-
// and it was simply unserved. It shares the compaction switch rather than adding a fifth
|
|
520
|
-
// system, because a user who has decided the advisor may steer a summary has decided that once.
|
|
521
|
-
//
|
|
522
|
-
// `label` is the part worth having. Pi's `/tree` can filter to labelled entries, so a branch
|
|
523
|
-
// that says what it was is the difference between a navigable tree and a list of timestamps,
|
|
524
|
-
// and the enum is fixed so no model-written text reaches the session file.
|
|
525
|
-
pi.on("session_before_tree", async (event: any, ctx: ExtensionContext) => {
|
|
526
|
-
if (!enabled("compaction")) {
|
|
527
|
-
return;
|
|
528
|
-
}
|
|
529
|
-
|
|
530
|
-
try {
|
|
531
|
-
const entries = event?.preparation?.entriesToSummarize ?? [];
|
|
532
|
-
if (entries.length === 0) {
|
|
533
|
-
return;
|
|
534
|
-
}
|
|
535
|
-
|
|
536
|
-
const result = await broker.request({
|
|
537
|
-
system: "compaction",
|
|
538
|
-
state: compaction.buildBranchInput({ preparation: event.preparation, objective }),
|
|
539
|
-
questions: compaction.questions({ branch: true }),
|
|
540
|
-
ctx,
|
|
541
|
-
root: ctx.cwd,
|
|
542
|
-
signal: event.signal,
|
|
543
|
-
decide: (answers: any) => {
|
|
544
|
-
const built = compaction.decide(answers);
|
|
545
|
-
const branchLabel = compaction.label(answers);
|
|
546
|
-
|
|
547
|
-
return {
|
|
548
|
-
applied: Boolean(branchLabel || built.customInstructions),
|
|
549
|
-
decision: { ...built, label: branchLabel },
|
|
550
|
-
};
|
|
551
|
-
},
|
|
552
|
-
});
|
|
553
|
-
if (!result.ok) {
|
|
554
|
-
return;
|
|
555
|
-
}
|
|
556
|
-
|
|
557
|
-
const advice = result.decision;
|
|
558
|
-
const patch: Record<string, unknown> = {};
|
|
559
|
-
if (advice.label) {
|
|
560
|
-
patch.label = advice.label;
|
|
561
|
-
}
|
|
562
|
-
|
|
563
|
-
// Only when a summary is actually going to be generated. Instructions for a summariser
|
|
564
|
-
// that will not run are bytes nobody reads, and `replaceInstructions` is left alone so
|
|
565
|
-
// Pi's own branch prompt still frames the result.
|
|
566
|
-
if (advice.customInstructions && event.preparation?.userWantsSummary === true) {
|
|
567
|
-
const existing =
|
|
568
|
-
typeof event.preparation?.customInstructions === "string"
|
|
569
|
-
? event.preparation.customInstructions.trim()
|
|
570
|
-
: "";
|
|
571
|
-
patch.customInstructions = existing
|
|
572
|
-
? `${existing}\n\n${advice.customInstructions}`
|
|
573
|
-
: advice.customInstructions;
|
|
574
|
-
}
|
|
575
|
-
|
|
576
|
-
return Object.keys(patch).length > 0 ? patch : undefined;
|
|
577
|
-
} catch {
|
|
578
|
-
return;
|
|
579
|
-
}
|
|
580
|
-
});
|
|
581
|
-
|
|
582
336
|
pi.on("tool_call", async (event: any, ctx: ExtensionContext) => {
|
|
583
337
|
history.signatures.push(progress.signature(event.toolName, event.input));
|
|
584
338
|
history.tools.push(event.toolName);
|
|
@@ -704,7 +458,7 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
704
458
|
// System 5: notice a session that has stopped making progress, while it can still be helped.
|
|
705
459
|
//
|
|
706
460
|
// THE ONE HANDLER THAT IS NOT AWAITED. Everything else in this file mutates what it inspects --
|
|
707
|
-
// a tool result, a
|
|
461
|
+
// a tool result, a tool's input -- so the session has to wait for the
|
|
708
462
|
// answer. This one acts on the next turn, and at roughly 300ms a call, awaiting it on a
|
|
709
463
|
// thrashing session would add seconds to an attempt to deliver advice that could not have
|
|
710
464
|
// changed the turn it was asked during.
|
|
@@ -868,16 +622,11 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
868
622
|
}
|
|
869
623
|
})()
|
|
870
624
|
: undefined;
|
|
871
|
-
// The same disclosure `/jev on` makes, on the path that arms the guard by name.
|
|
872
|
-
// Learning from a blocked call that calls can be blocked is the outcome that
|
|
873
|
-
// rule exists to prevent, and which command did the arming does not change it.
|
|
874
|
-
const armedGuard = action === "enable" && names.includes("guard") && settings.master;
|
|
875
625
|
ctx.ui.notify(
|
|
876
626
|
`${action === "enable" ? "Enabled" : "Disabled"}: ${names.join(", ")}.` +
|
|
877
627
|
`${kept ? " Remembered for new sessions." : " This session only."}` +
|
|
878
628
|
`${emptied ? " That was the last system, so the layer was switched off; it would otherwise run and do nothing." : ""}` +
|
|
879
|
-
`${!emptied && !settings.master ? " The layer is still off; run /jev on." : ""}
|
|
880
|
-
`${armedGuard ? `\n${guardWarning()}` : ""}`,
|
|
629
|
+
`${!emptied && !settings.master ? " The layer is still off; run /jev on." : ""}`,
|
|
881
630
|
"info",
|
|
882
631
|
);
|
|
883
632
|
|
|
@@ -916,8 +665,7 @@ export default function jevAdvisor(pi: ExtensionAPI) {
|
|
|
916
665
|
const enabled = SYSTEM_NAMES.filter((name) => saved.systems[name]);
|
|
917
666
|
ctx.ui.notify(
|
|
918
667
|
saved.startup && saved.master
|
|
919
|
-
? `New Pi sessions will start with the Jev layer on, with ${enabled.length} of ${SYSTEM_NAMES.length} systems: ${enabled.join(", ")}. This session is unchanged; run /jev on to switch it on now.`
|
|
920
|
-
`${saved.systems.guard ? `\n${guardWarning()}` : ""}`
|
|
668
|
+
? `New Pi sessions will start with the Jev layer on, with ${enabled.length} of ${SYSTEM_NAMES.length} systems: ${enabled.join(", ")}. This session is unchanged; run /jev on to switch it on now.`
|
|
921
669
|
: "New Pi sessions will start with the Jev layer off. This session is unchanged.",
|
|
922
670
|
"info",
|
|
923
671
|
);
|
|
@@ -162,7 +162,7 @@ function environmentOnly() {
|
|
|
162
162
|
|
|
163
163
|
/**
|
|
164
164
|
* Jev is reached through OpenRouter by default: that is where it is published, it is what
|
|
165
|
-
*
|
|
165
|
+
* specpi-jev-guard uses, and an OpenRouter key (`sk-or-...`) is rejected by the direct
|
|
166
166
|
* TypeSafe API with a bare 401. `JEV_BACKEND=typesafe` selects the direct API for a TypeSafe key.
|
|
167
167
|
*
|
|
168
168
|
* It lives here rather than in client.mjs because everything below has to bind it. A parameter
|