specpi 0.26.0 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/README.md +37 -3
- package/SECURITY_MODEL.md +44 -4
- package/THIRD_PARTY.md +9 -1
- package/extensions/jev-advisor/broker.mjs +277 -0
- package/extensions/jev-advisor/client.mjs +172 -0
- package/extensions/jev-advisor/config.mjs +270 -0
- package/extensions/jev-advisor/consent.mjs +133 -0
- package/extensions/jev-advisor/gate.mjs +263 -0
- package/extensions/jev-advisor/index.ts +999 -0
- package/extensions/jev-advisor/key-source.mjs +252 -0
- package/extensions/jev-advisor/layer.mjs +169 -0
- package/extensions/jev-advisor/ledger.mjs +138 -0
- package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
- package/extensions/jev-advisor/questions/compaction.mjs +153 -0
- package/extensions/jev-advisor/questions/gap.mjs +140 -0
- package/extensions/jev-advisor/questions/guard.mjs +168 -0
- package/extensions/jev-advisor/questions/progress.mjs +195 -0
- package/extensions/jev-advisor/questions/retention.mjs +188 -0
- package/extensions/jev-advisor/questions/sources.mjs +91 -0
- package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
- package/extensions/jev-advisor/risk.mjs +442 -0
- package/extensions/jev-advisor/sanitize.mjs +0 -0
- package/extensions/jev-advisor/usage.mjs +92 -0
- package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
- package/extensions/tool-wishlist/index.ts +11 -0
- package/extensions/workflow-controls/capabilities.mjs +26 -0
- package/extensions/workflow-controls/index.ts +2 -2
- package/package.json +1 -1
- package/scripts/packages.mjs +56 -0
- package/scripts/specpi.mjs +73 -4
|
@@ -0,0 +1,442 @@
|
|
|
1
|
+
// Local triage for shell and file calls, before anything is sent anywhere.
|
|
2
|
+
//
|
|
3
|
+
// This is the first half of the native command guard. Jev is the second half and the one that does
|
|
4
|
+
// the analysis; everything here exists to decide what Jev is asked about. Three answers:
|
|
5
|
+
//
|
|
6
|
+
// safe -- settled locally. Jev is never asked, and nothing else will look at this call.
|
|
7
|
+
// dangerous -- blocked locally, with no call and no human.
|
|
8
|
+
// unknown -- Jev is asked.
|
|
9
|
+
//
|
|
10
|
+
// THE THREE ARE NOT SYMMETRIC, and that asymmetry is the whole design:
|
|
11
|
+
//
|
|
12
|
+
// a wrong `safe` is a silent, permanent hole -- the one verdict with no second reader
|
|
13
|
+
// a wrong `unknown` costs one call out of a per-session budget of 208, and Jev decides
|
|
14
|
+
// a wrong `dangerous` blocks real work with no recourse but switching the guard off
|
|
15
|
+
//
|
|
16
|
+
// So the two lists below are maintained under opposite pressures, and the mistake worth naming is
|
|
17
|
+
// treating them as one thing called "the guard" and hardening both the same way.
|
|
18
|
+
//
|
|
19
|
+
// READ_ONLY is a BUDGET mechanism, not a safety one. The guard has 208 calls and each is a few
|
|
20
|
+
// hundred milliseconds awaited on the tool path, so a session that greps and cats a few hundred
|
|
21
|
+
// times would spend the lot and then defer everything for the rest of its life -- a guard that runs
|
|
22
|
+
// out is weaker than one with a fast path. The admission test is therefore NOT "is this harmless".
|
|
23
|
+
// It is: is this binary simple whatever flags it is given, and common enough to be worth it? A
|
|
24
|
+
// binary that can launch a program, write a file or change machine state under any flag is not
|
|
25
|
+
// simple, however harmless its name reads, and Jev parses it. `env` and `fd` launch things,
|
|
26
|
+
// `find` has -delete, `rg` has --pre, `date -s` sets the clock, `hostname` sets the hostname,
|
|
27
|
+
// `file -C` writes a compiled magic file. Every one of those was on this list, and every one was a
|
|
28
|
+
// general bypass that cost almost no budget to keep.
|
|
29
|
+
//
|
|
30
|
+
// CATASTROPHIC is deliberately tiny, and its only real job is the case where Jev is NOT THERE: no
|
|
31
|
+
// key, budget spent, a timeout. Jev catches everything this list would, and weighs intent besides,
|
|
32
|
+
// which a pattern cannot. So resist adding to it. A rule here fires with no model and no human, and
|
|
33
|
+
// a false positive is a blocked session whose only remedy is switching the whole guard off -- and a
|
|
34
|
+
// guard people switch off protects nobody. Ask whether you would stake "this is never legitimate"
|
|
35
|
+
// on it; if not, it belongs in the question set, where being wrong costs one call.
|
|
36
|
+
//
|
|
37
|
+
// And nothing here throws. A classifier that can fail is a classifier that can take a session down,
|
|
38
|
+
// so an unparseable command reads as `unknown` -- ask about it -- rather than as an error.
|
|
39
|
+
|
|
40
|
+
import path from "node:path";
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Shell tools, under every name a harness gives them.
|
|
44
|
+
*
|
|
45
|
+
* `scripts/eval-proxy.mjs` already folds `pwsh`, `shell`, `exec`, `exec_command` and `write_stdin`
|
|
46
|
+
* into `bash`, which means those names reach real sessions. A guard that only knew `bash` would be
|
|
47
|
+
* bypassed by spelling, so the alias list lives here rather than being rediscovered later.
|
|
48
|
+
*/
|
|
49
|
+
export const SHELL_TOOLS = Object.freeze([
|
|
50
|
+
"bash",
|
|
51
|
+
"powershell",
|
|
52
|
+
"pwsh",
|
|
53
|
+
"shell",
|
|
54
|
+
"exec",
|
|
55
|
+
"exec_command",
|
|
56
|
+
"write_stdin",
|
|
57
|
+
]);
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Tools that write a file. The same list `questions/progress.mjs` uses for "the worktree changed",
|
|
61
|
+
* deliberately, because a write the guard cannot see is a write the credential question is never
|
|
62
|
+
* asked about -- and `multi_edit` reaching `~/.ssh/authorized_keys` is exactly that.
|
|
63
|
+
*/
|
|
64
|
+
export const WRITE_TOOLS = Object.freeze(["write", "edit", "multi_edit", "apply_patch", "create_file", "str_replace"]);
|
|
65
|
+
|
|
66
|
+
/** The tool calls this guard looks at. Everything else passes without inspection. */
|
|
67
|
+
export const GATED_TOOLS = Object.freeze([...SHELL_TOOLS, ...WRITE_TOOLS]);
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Simple binaries: ones that change nothing whatever flags they are given.
|
|
71
|
+
*
|
|
72
|
+
* "Whatever flags" is the whole test, and it is stricter than it sounds. `git` fails it obviously
|
|
73
|
+
* (`push`, `reset`, `clean`), and special-casing `git status` would only add a subcommand allowlist
|
|
74
|
+
* to maintain beside this one. These fail it less obviously, which is what made each of them a
|
|
75
|
+
* bypass worth having:
|
|
76
|
+
*
|
|
77
|
+
* env, fd launch another program outright -- `env rm -rf build`, `fd -x rm`
|
|
78
|
+
* find has -delete and -exec
|
|
79
|
+
* rg has --pre, which runs an arbitrary preprocessor per file
|
|
80
|
+
* sort, uniq name an output file (`sort -o`, `uniq in out`)
|
|
81
|
+
* date -s sets the system clock
|
|
82
|
+
* hostname with an argument, sets the hostname
|
|
83
|
+
* file -C compiles and writes a magic file
|
|
84
|
+
* printenv changes nothing and hands over a secret, which the risk question also asks about
|
|
85
|
+
*
|
|
86
|
+
* None of them reads as dangerous, and none of them was common enough for the fast path to be
|
|
87
|
+
* buying much. That is the trade: a rare binary on this list saves almost no budget and costs a
|
|
88
|
+
* silent hole, so when in doubt it comes off and Jev parses it.
|
|
89
|
+
*
|
|
90
|
+
* Being here is not a claim that a call is harmless. It is a claim that asking about it would spend
|
|
91
|
+
* the budget without learning anything -- which is why the arguments are still checked below, and
|
|
92
|
+
* `cat ~/.ssh/id_rsa` leaves the fast path even though `cat` never belongs anywhere else.
|
|
93
|
+
*/
|
|
94
|
+
const READ_ONLY = new Set([
|
|
95
|
+
"ls",
|
|
96
|
+
"dir",
|
|
97
|
+
"pwd",
|
|
98
|
+
"cd",
|
|
99
|
+
"cat",
|
|
100
|
+
"head",
|
|
101
|
+
"tail",
|
|
102
|
+
"wc",
|
|
103
|
+
"stat",
|
|
104
|
+
"du",
|
|
105
|
+
"df",
|
|
106
|
+
"whoami",
|
|
107
|
+
"uname",
|
|
108
|
+
"echo",
|
|
109
|
+
"printf",
|
|
110
|
+
"which",
|
|
111
|
+
"type",
|
|
112
|
+
"grep",
|
|
113
|
+
"diff",
|
|
114
|
+
"cut",
|
|
115
|
+
"tr",
|
|
116
|
+
"basename",
|
|
117
|
+
"dirname",
|
|
118
|
+
"realpath",
|
|
119
|
+
"true",
|
|
120
|
+
"false",
|
|
121
|
+
]);
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Shell syntax that can turn a safe-looking command into any other command: chaining, substitution,
|
|
125
|
+
* redirection, background execution. Their presence disqualifies the fast path entirely rather than
|
|
126
|
+
* being parsed, because parsing a shell correctly is not something a guard should be attempting.
|
|
127
|
+
*/
|
|
128
|
+
const SHELL_CONTROL = /[;&|><`$(){}\n\r]|\|\||&&/u;
|
|
129
|
+
|
|
130
|
+
/** Targets that mean "everything": the filesystem root, or a bare home directory. */
|
|
131
|
+
const ROOT_TARGETS = new Set(["/", "/*", "~", "~/*", "$HOME", "$HOME/*", "%USERPROFILE%", "%USERPROFILE%\\*"]);
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* A recursive delete aimed at the root or a bare home.
|
|
135
|
+
*
|
|
136
|
+
* Written as a function rather than a regex because the regex it replaces anchored on the end of the
|
|
137
|
+
* string, so it matched `rm -rf /` -- which GNU `rm` refuses on its own -- and missed
|
|
138
|
+
* `rm -rf / --no-preserve-root`, which is the spelling that actually empties the disk. Flag order and
|
|
139
|
+
* position are not something a pattern should be asked to track.
|
|
140
|
+
*/
|
|
141
|
+
function removesEverything(value) {
|
|
142
|
+
const tokens = value.trim().split(/\s+/u);
|
|
143
|
+
const start = tokens[0] === "sudo" || tokens[0] === "doas" ? 1 : 0;
|
|
144
|
+
if (leadingBinary(tokens[start]) !== "rm") {
|
|
145
|
+
return false;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const args = tokens.slice(start + 1);
|
|
149
|
+
const recursive = args.some(
|
|
150
|
+
(token) => token === "--recursive" || (/^-[a-zA-Z]+$/u.test(token) && /[rR]/u.test(token)),
|
|
151
|
+
);
|
|
152
|
+
|
|
153
|
+
return (
|
|
154
|
+
recursive && args.some((token) => !token.startsWith("-") && ROOT_TARGETS.has(token.replace(/\/+$/u, "") || "/"))
|
|
155
|
+
);
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Catastrophic and unambiguous. Every entry here is something that destroys data or hands the
|
|
160
|
+
* machine to someone else, with no legitimate reading in an agent session.
|
|
161
|
+
*
|
|
162
|
+
* Kept short on purpose. These fire without a model and without a human, so a false positive here
|
|
163
|
+
* is a blocked session with no recourse but switching the guard off -- and a guard people switch
|
|
164
|
+
* off protects nobody. Everything requiring judgement is `unknown`.
|
|
165
|
+
*/
|
|
166
|
+
const CATASTROPHIC = Object.freeze([
|
|
167
|
+
{
|
|
168
|
+
// `rm -rf /`, `rm -rf / --no-preserve-root`, `rm -r -f ~`. Not `rm -rf ./build`.
|
|
169
|
+
test: removesEverything,
|
|
170
|
+
reason: "recursive delete of the filesystem root or home directory",
|
|
171
|
+
},
|
|
172
|
+
{
|
|
173
|
+
// Writing a raw block device: mkfs, or dd with a device destination.
|
|
174
|
+
pattern: /\b(mkfs(\.\w+)?|fdisk|diskpart)\b|\bdd\b[^\n]*\bof=\/dev\/(sd|nvme|hd|disk)/u,
|
|
175
|
+
reason: "writing directly to a disk device",
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
// Piping a downloaded script straight into a shell.
|
|
179
|
+
pattern: /\b(curl|wget|iwr|Invoke-WebRequest)\b[^\n|]*\|\s*(sudo\s+)?(ba|z|k|fi|da)?sh\b/u,
|
|
180
|
+
reason: "executing a downloaded script without inspecting it",
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
// Recursive world-writable or ownership changes over a filesystem root.
|
|
184
|
+
pattern:
|
|
185
|
+
/\bchmod\s+(-[a-zA-Z]*\s+)*(-R|--recursive)\s+777\s+\/\s*$|\bchown\s+(-R|--recursive)\s+[^\s]+\s+\/\s*$/u,
|
|
186
|
+
reason: "recursive permission or ownership change over the filesystem root",
|
|
187
|
+
},
|
|
188
|
+
{
|
|
189
|
+
pattern: /:\(\)\s*\{\s*:\|\s*:\s*&\s*\}\s*;\s*:/u,
|
|
190
|
+
reason: "fork bomb",
|
|
191
|
+
},
|
|
192
|
+
{
|
|
193
|
+
// Overwriting the shell history or a credential store with nothing is how a session hides
|
|
194
|
+
// what it did; blocking it is cheap and it is never a legitimate agent action.
|
|
195
|
+
pattern: /\b(history\s+-c|Clear-History)\b|>\s*~?\/?\.bash_history\b/u,
|
|
196
|
+
reason: "clearing shell history",
|
|
197
|
+
},
|
|
198
|
+
]);
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Paths whose contents are credentials, keys or version-control internals.
|
|
202
|
+
*
|
|
203
|
+
* Every pattern is tested against a path already resolved against the working directory and written
|
|
204
|
+
* with forward slashes, so a segment is bounded by `/` or by the end of the string. Matching a
|
|
205
|
+
* directory matters as much as matching a file: `secrets/api.txt` is a secret, and the first version
|
|
206
|
+
* of this list -- which required the secret's name to be the last segment -- called it an ordinary
|
|
207
|
+
* project file.
|
|
208
|
+
*/
|
|
209
|
+
const PROTECTED = Object.freeze([
|
|
210
|
+
/(^|\/)\.env(\.|\/|$)/u,
|
|
211
|
+
/\.env$/u,
|
|
212
|
+
/(^|\/)\.git(\/|$)/u,
|
|
213
|
+
/(^|\/)\.ssh(\/|$)/u,
|
|
214
|
+
/(^|\/)\.aws(\/|$)/u,
|
|
215
|
+
/(^|\/)\.gnupg(\/|$)/u,
|
|
216
|
+
/(^|\/)(id_rsa|id_dsa|id_ecdsa|id_ed25519)(\.|$)/u,
|
|
217
|
+
/\.(pem|key|pfx|p12|keystore|jks)$/iu,
|
|
218
|
+
/(^|\/)(credentials?|secrets?)(\.|\/|$)/iu,
|
|
219
|
+
/(^|\/)auth\.json$/u,
|
|
220
|
+
/(^|\/)\.npmrc$/u,
|
|
221
|
+
/(^|\/)\.netrc$/u,
|
|
222
|
+
/(^|\/)\.pi(\/|$)/u,
|
|
223
|
+
]);
|
|
224
|
+
|
|
225
|
+
function text(value) {
|
|
226
|
+
return typeof value === "string" ? value : "";
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/** The binary a command starts with, lowercased and stripped of any path. */
|
|
230
|
+
export function leadingBinary(command) {
|
|
231
|
+
const trimmed = text(command).trim();
|
|
232
|
+
if (trimmed.length === 0) {
|
|
233
|
+
return "";
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
const first = trimmed.split(/\s+/u)[0];
|
|
237
|
+
|
|
238
|
+
return path.basename(first.replace(/\\/gu, "/")).toLowerCase();
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* Classify a shell command without asking anyone.
|
|
243
|
+
*
|
|
244
|
+
* Returns `{ decision, reason }` where decision is "safe", "dangerous" or "unknown". The order
|
|
245
|
+
* matters: dangerous is checked before safe, so a catastrophic command hidden behind a read-only
|
|
246
|
+
* binary cannot pass on the fast path.
|
|
247
|
+
*/
|
|
248
|
+
export function classifyCommand(command, cwd = process.cwd()) {
|
|
249
|
+
const value = text(command).trim();
|
|
250
|
+
if (value.length === 0) {
|
|
251
|
+
return { decision: "unknown", reason: "empty command" };
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
for (const rule of CATASTROPHIC) {
|
|
255
|
+
if (rule.test ? rule.test(value) : rule.pattern.test(value)) {
|
|
256
|
+
return { decision: "dangerous", reason: rule.reason };
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
// A command with shell control characters is never taken on the fast path, however harmless its
|
|
261
|
+
// first word looks: `ls; rm -rf ~` begins with `ls`.
|
|
262
|
+
if (SHELL_CONTROL.test(value)) {
|
|
263
|
+
return { decision: "unknown", reason: "shell control characters" };
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
if (READ_ONLY.has(leadingBinary(value))) {
|
|
267
|
+
// Read-only is a statement about what the binary does, not about what it is pointed at, and
|
|
268
|
+
// the question this guard asks names exfiltration beside destruction. `cat ~/.ssh/id_rsa`
|
|
269
|
+
// changes nothing and hands over a private key, so the fast path has to look at the
|
|
270
|
+
// arguments too or half of the question it asks is unreachable for the commands that answer
|
|
271
|
+
// it. A protected argument costs one question; every other read stays free.
|
|
272
|
+
return commandArguments(value).some((argument) => protectedPath(argument, cwd))
|
|
273
|
+
? { decision: "unknown", reason: "reads a protected path" }
|
|
274
|
+
: { decision: "safe", reason: "read-only command" };
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
return { decision: "unknown", reason: "not a known read-only command" };
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* The non-flag arguments of a command, unquoted.
|
|
282
|
+
*
|
|
283
|
+
* Deliberately crude: this decides whether to ask a question, never whether to block, so a token it
|
|
284
|
+
* splits wrongly costs a question and nothing else.
|
|
285
|
+
*/
|
|
286
|
+
function commandArguments(value) {
|
|
287
|
+
return value
|
|
288
|
+
.split(/\s+/u)
|
|
289
|
+
.slice(1)
|
|
290
|
+
.filter((token) => !token.startsWith("-"))
|
|
291
|
+
.map((token) => token.replace(/^["']|["']$/gu, ""))
|
|
292
|
+
.filter((token) => token.length > 0);
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Whether a write or edit target holds credentials or version-control internals.
|
|
297
|
+
*
|
|
298
|
+
* Relative targets are resolved against the working directory first, so `../../.ssh/config` is seen
|
|
299
|
+
* for what it is rather than for what it is spelled as.
|
|
300
|
+
*/
|
|
301
|
+
export function protectedPath(target, cwd = process.cwd()) {
|
|
302
|
+
const value = text(target).trim();
|
|
303
|
+
if (value.length === 0) {
|
|
304
|
+
return false;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
let resolved;
|
|
308
|
+
try {
|
|
309
|
+
resolved = path.resolve(cwd, value);
|
|
310
|
+
} catch {
|
|
311
|
+
resolved = value;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
const normalized = resolved.replaceAll("\\", "/");
|
|
315
|
+
|
|
316
|
+
return PROTECTED.some((pattern) => pattern.test(normalized));
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/** Keys a harness uses for "the file this call writes", across the tools in `WRITE_TOOLS`. */
|
|
320
|
+
const TARGET_KEYS = Object.freeze(["path", "file_path", "filePath", "file", "target", "notebook_path"]);
|
|
321
|
+
|
|
322
|
+
/** Nested arrays of edits, each element carrying a target of its own. */
|
|
323
|
+
const TARGET_LISTS = Object.freeze(["edits", "files", "changes", "operations"]);
|
|
324
|
+
|
|
325
|
+
/** Header forms that name a file inside a patch body. */
|
|
326
|
+
const PATCH_TARGET = /^(?:\*\*\* (?:Add|Update|Delete) File: |--- (?:a\/)?|\+\+\+ (?:b\/)?)(.+)$/gmu;
|
|
327
|
+
|
|
328
|
+
function pushTarget(into, value) {
|
|
329
|
+
if (typeof value === "string" && value.trim().length > 0 && into.length < 64) {
|
|
330
|
+
into.push(value.trim());
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* Every file a tool call names, from a tool input whose shape this module does not control.
|
|
336
|
+
*
|
|
337
|
+
* `write` and `edit` carry one `path`; `multi_edit` carries a list; `apply_patch` carries the paths
|
|
338
|
+
* inside a patch body and nowhere else. Reading only the first of those is how a write to a
|
|
339
|
+
* credential file reaches disk without the guard ever seeing a target -- so this reads all three,
|
|
340
|
+
* and returning nothing is itself a meaningful answer to `classifyCall`.
|
|
341
|
+
*/
|
|
342
|
+
export function callTargets(input) {
|
|
343
|
+
const found = [];
|
|
344
|
+
if (input === null || typeof input !== "object") {
|
|
345
|
+
return found;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
for (const key of TARGET_KEYS) {
|
|
349
|
+
pushTarget(found, input[key]);
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
for (const key of TARGET_LISTS) {
|
|
353
|
+
const list = input[key];
|
|
354
|
+
if (!Array.isArray(list)) {
|
|
355
|
+
continue;
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
for (const item of list) {
|
|
359
|
+
if (typeof item === "string") {
|
|
360
|
+
pushTarget(found, item);
|
|
361
|
+
continue;
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
if (item !== null && typeof item === "object") {
|
|
365
|
+
for (const key2 of TARGET_KEYS) {
|
|
366
|
+
pushTarget(found, item[key2]);
|
|
367
|
+
}
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
const patch = typeof input.patch === "string" ? input.patch : typeof input.diff === "string" ? input.diff : "";
|
|
373
|
+
if (patch.length > 0) {
|
|
374
|
+
// Bounded: a patch is arbitrary size and this runs on every call.
|
|
375
|
+
for (const match of patch.slice(0, 20_000).matchAll(PATCH_TARGET)) {
|
|
376
|
+
pushTarget(found, match[1].replace(/\t.*$/u, ""));
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
return found;
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/** Keys a harness uses for "the text this call runs", across the tools in `SHELL_TOOLS`. */
|
|
384
|
+
const COMMAND_KEYS = Object.freeze(["command", "cmd", "script", "input", "text", "data", "stdin", "line"]);
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* The text a shell call will run, from a tool input whose shape this module does not control.
|
|
388
|
+
*
|
|
389
|
+
* `bash` carries `command`; `write_stdin` carries the text it types into a live shell under some
|
|
390
|
+
* other name entirely. Reading `command` alone meant every `write_stdin` was classified as an empty
|
|
391
|
+
* command -- spending a guard call on the empty string while the `rm -rf ~` being typed went
|
|
392
|
+
* unexamined -- so the tool most worth reading was the one read as blank.
|
|
393
|
+
*/
|
|
394
|
+
export function commandText(input) {
|
|
395
|
+
if (typeof input === "string") {
|
|
396
|
+
return input;
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
if (input === null || typeof input !== "object") {
|
|
400
|
+
return "";
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
for (const key of COMMAND_KEYS) {
|
|
404
|
+
if (typeof input[key] === "string" && input[key].trim().length > 0) {
|
|
405
|
+
return input[key];
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
return "";
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
/**
|
|
413
|
+
* What a gated tool call is, before Jev is involved.
|
|
414
|
+
*
|
|
415
|
+
* `write` and its siblings are only interesting when they target something protected: an ordinary
|
|
416
|
+
* source file being edited is the entire point of the agent, and asking about each one would spend a
|
|
417
|
+
* session's budget on the first directory it refactored.
|
|
418
|
+
*
|
|
419
|
+
* A write whose target could not be read is `unknown` rather than `safe`. That costs a question on a
|
|
420
|
+
* tool shape this module does not recognise, which is the right way round: the alternative is a
|
|
421
|
+
* silent hole that appears the moment a harness renames a field.
|
|
422
|
+
*/
|
|
423
|
+
export function classifyCall({ tool, command, target, targets, cwd }) {
|
|
424
|
+
if (!GATED_TOOLS.includes(tool)) {
|
|
425
|
+
return { decision: "safe", reason: "tool is not gated" };
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
if (SHELL_TOOLS.includes(tool)) {
|
|
429
|
+
return classifyCommand(command, cwd);
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
const all = [...(Array.isArray(targets) ? targets : []), ...(typeof target === "string" ? [target] : [])].filter(
|
|
433
|
+
(item) => typeof item === "string" && item.trim().length > 0,
|
|
434
|
+
);
|
|
435
|
+
if (all.length === 0) {
|
|
436
|
+
return { decision: "unknown", reason: "write target could not be read" };
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
return all.some((item) => protectedPath(item, cwd))
|
|
440
|
+
? { decision: "unknown", reason: "writes to a protected path" }
|
|
441
|
+
: { decision: "safe", reason: "ordinary project file" };
|
|
442
|
+
}
|
|
Binary file
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
// What the ledger cannot answer: how much of this session's budget is left, right now.
|
|
2
|
+
//
|
|
3
|
+
// The ledger is an audit trail -- every call ever made, append-only, rotated when it grows past
|
|
4
|
+
// eight megabytes. Counting *this session* out of it means knowing where the session began, which
|
|
5
|
+
// nothing outside the advisor's own process does. So the advisor publishes its running total to one
|
|
6
|
+
// small file, rewritten in place, and anything that wants to show a budget reads that instead of
|
|
7
|
+
// replaying a log to find the boundary.
|
|
8
|
+
//
|
|
9
|
+
// It holds counts and nothing else. No state, no questions, no answers, no payload, not even the
|
|
10
|
+
// ledger's digests: a reader learns how many calls a system made and how many changed something,
|
|
11
|
+
// and can learn nothing about what was sent. That is deliberate, because this is the one file in
|
|
12
|
+
// the layer meant to be read by another process.
|
|
13
|
+
//
|
|
14
|
+
// The file always describes the most recent session rather than being deleted at shutdown, and
|
|
15
|
+
// carries `active` to say which. "Nothing is running and the last session spent 6 calls" and "a
|
|
16
|
+
// session is running and has spent 6 calls" are different facts, and a reader that could not tell
|
|
17
|
+
// them apart would report a live budget for a session that ended yesterday.
|
|
18
|
+
|
|
19
|
+
import fs from "node:fs";
|
|
20
|
+
import path from "node:path";
|
|
21
|
+
import { SYSTEM_NAMES, jevDirectory, regularFile, writeFileAtomic } from "./config.mjs";
|
|
22
|
+
|
|
23
|
+
export function usagePath() {
|
|
24
|
+
return path.join(jevDirectory(), "usage.json");
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Replace the snapshot. Failure is swallowed for the same reason the ledger's is: an advisor that
|
|
29
|
+
* cannot write its own bookkeeping must not break the session it is advising, and a missing file
|
|
30
|
+
* reads as "no usage recorded" rather than as zero calls.
|
|
31
|
+
*/
|
|
32
|
+
export function writeUsage(snapshot) {
|
|
33
|
+
try {
|
|
34
|
+
writeFileAtomic(usagePath(), `${JSON.stringify(snapshot, null, 4)}\n`);
|
|
35
|
+
|
|
36
|
+
return true;
|
|
37
|
+
} catch {
|
|
38
|
+
return false;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** Every unknown shape reads as absent, never as a partial count. */
|
|
43
|
+
export function normalizeUsage(raw) {
|
|
44
|
+
if (raw?.schema !== 1) {
|
|
45
|
+
return undefined;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const systems = {};
|
|
49
|
+
for (const name of SYSTEM_NAMES) {
|
|
50
|
+
const bucket = raw.systems?.[name];
|
|
51
|
+
systems[name] = {
|
|
52
|
+
calls: count(bucket?.calls),
|
|
53
|
+
applied: count(bucket?.applied),
|
|
54
|
+
failed: count(bucket?.failed),
|
|
55
|
+
savedBytes: count(bucket?.savedBytes),
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
schema: 1,
|
|
61
|
+
session: typeof raw.session === "string" ? raw.session.slice(0, 64) : "",
|
|
62
|
+
startedAt: iso(raw.startedAt),
|
|
63
|
+
updatedAt: iso(raw.updatedAt),
|
|
64
|
+
active: raw.active === true,
|
|
65
|
+
calls: count(raw.calls),
|
|
66
|
+
budgets: Object.fromEntries(["total", ...SYSTEM_NAMES].map((name) => [name, count(raw.budgets?.[name])])),
|
|
67
|
+
systems,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function count(value) {
|
|
72
|
+
return Number.isSafeInteger(value) && value >= 0 ? value : 0;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function iso(value) {
|
|
76
|
+
return typeof value === "string" && !Number.isNaN(Date.parse(value)) ? value : "";
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export function readUsage() {
|
|
80
|
+
try {
|
|
81
|
+
const file = usagePath();
|
|
82
|
+
// regularFile refuses links, hard-linked files and anything over 4 KiB. A counts file for
|
|
83
|
+
// eight systems is a few hundred bytes, so the size check is a real one here.
|
|
84
|
+
if (!regularFile(file, "Jev usage")) {
|
|
85
|
+
return undefined;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return normalizeUsage(JSON.parse(fs.readFileSync(file, "utf8")));
|
|
89
|
+
} catch {
|
|
90
|
+
return undefined;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
// Phase 0 of the Jev advisor plan, which needs no model at all.
|
|
2
|
+
//
|
|
3
|
+
// Measured across the recorded eval runs, six of SpecPi's ten offered tools were offered on 31 of
|
|
4
|
+
// 31 attempts and called zero times. Two of them are these: `record_harness_contract` and
|
|
5
|
+
// `finish_harness_improvement` are *authoring* tools, usable only after a human has selected a
|
|
6
|
+
// candidate through /harness-improvement. Whether a selection exists is a fact in local state, so
|
|
7
|
+
// the answer is a boolean — no latency, no cost, no false positives, and nothing for a classifier
|
|
8
|
+
// to route.
|
|
9
|
+
//
|
|
10
|
+
// `report_capability_gap` is deliberately not in this list. It is the *observation* tool and the
|
|
11
|
+
// reason the improvement loop exists; withdrawing it would silently lose the friction reports the
|
|
12
|
+
// loop is built to capture. It stays offered at all times.
|
|
13
|
+
|
|
14
|
+
/** Only usable while an improvement is selected. */
|
|
15
|
+
export const AUTHORING_TOOL_NAMES = Object.freeze(["record_harness_contract", "finish_harness_improvement"]);
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Add or remove the authoring tools without disturbing any other extension's tools. Mirrors the
|
|
19
|
+
* web-access gate: it only ever touches the names it owns, and it does nothing when the active set
|
|
20
|
+
* already matches, so a no-op never costs the cached prompt prefix.
|
|
21
|
+
*/
|
|
22
|
+
export function syncAuthoringTools(pi, selected) {
|
|
23
|
+
if (typeof pi?.getActiveTools !== "function" || typeof pi?.setActiveTools !== "function") {
|
|
24
|
+
return false;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const owned = new Set(AUTHORING_TOOL_NAMES);
|
|
28
|
+
const active = pi.getActiveTools();
|
|
29
|
+
const present = active.filter((name) => owned.has(name));
|
|
30
|
+
if (selected && present.length === owned.size) {
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
if (!selected && present.length === 0) {
|
|
35
|
+
return false;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const others = active.filter((name) => !owned.has(name));
|
|
39
|
+
pi.setActiveTools(selected ? [...others, ...AUTHORING_TOOL_NAMES] : others);
|
|
40
|
+
|
|
41
|
+
return true;
|
|
42
|
+
}
|
|
@@ -13,6 +13,7 @@ import { getMarkdownTheme, type ExtensionAPI } from "@earendil-works/pi-coding-a
|
|
|
13
13
|
import { StringEnum } from "@earendil-works/pi-ai";
|
|
14
14
|
import { Box, Markdown, Text } from "@earendil-works/pi-tui";
|
|
15
15
|
import { Type } from "typebox";
|
|
16
|
+
import { syncAuthoringTools } from "./authoring-tools.mjs";
|
|
16
17
|
import {
|
|
17
18
|
appendWishlistDecision,
|
|
18
19
|
archiveWishlist,
|
|
@@ -498,6 +499,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
498
499
|
let activeRunId = randomUUID();
|
|
499
500
|
let activeImprovement: ActiveImprovement | undefined;
|
|
500
501
|
let improvementLifecycleGeneration = 0;
|
|
502
|
+
// The authoring tools ride every request whether or not they can be used. Local state knows
|
|
503
|
+
// when they cannot be, so the answer is a boolean rather than a prediction.
|
|
504
|
+
const syncAuthoring = () => syncAuthoringTools(pi, activeImprovement !== undefined);
|
|
501
505
|
let improvementMenuGeneration = 0;
|
|
502
506
|
let finishBusy = false;
|
|
503
507
|
let contractBusy = false;
|
|
@@ -583,15 +587,18 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
583
587
|
|
|
584
588
|
pi.on("session_start", (_event, ctx) => {
|
|
585
589
|
restoreActiveImprovement(ctx);
|
|
590
|
+
syncAuthoring();
|
|
586
591
|
});
|
|
587
592
|
|
|
588
593
|
pi.on("session_tree", (_event, ctx) => {
|
|
589
594
|
restoreActiveImprovement(ctx);
|
|
595
|
+
syncAuthoring();
|
|
590
596
|
});
|
|
591
597
|
|
|
592
598
|
pi.on("session_shutdown", () => {
|
|
593
599
|
improvementLifecycleGeneration += 1;
|
|
594
600
|
activeImprovement = undefined;
|
|
601
|
+
syncAuthoring();
|
|
595
602
|
});
|
|
596
603
|
|
|
597
604
|
const assertImprovementStillCurrent = (expected: ActiveImprovement, generation: number, ctx: any, signal?: any) => {
|
|
@@ -1131,6 +1138,7 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
1131
1138
|
});
|
|
1132
1139
|
if (activeImprovement?.selectionId === selectedImprovement.selectionId) {
|
|
1133
1140
|
activeImprovement = undefined;
|
|
1141
|
+
syncAuthoring();
|
|
1134
1142
|
}
|
|
1135
1143
|
|
|
1136
1144
|
return {
|
|
@@ -1312,6 +1320,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
1312
1320
|
...policy,
|
|
1313
1321
|
};
|
|
1314
1322
|
assertSelectionContextCurrent();
|
|
1323
|
+
// Pi offers tools added during a call from the next assistant message, so restoring
|
|
1324
|
+
// them here lands exactly when the selection they belong to becomes usable.
|
|
1325
|
+
syncAuthoring();
|
|
1315
1326
|
pi.appendEntry(TASK_CONTRACT_ENTRY, { kind: "cleared" });
|
|
1316
1327
|
leafId = ctx.sessionManager.getLeafId?.();
|
|
1317
1328
|
assertSelectionContextCurrent();
|
|
@@ -37,11 +37,35 @@ export const BROWSER_TOOL_NAMES = Object.freeze([
|
|
|
37
37
|
* the package's own state when it finishes, so a tool-name activation would undo itself.
|
|
38
38
|
* Delegation needs an activation path inside its own package.
|
|
39
39
|
*/
|
|
40
|
+
/**
|
|
41
|
+
* Restoring a group mid-session costs twice, and only one of those costs was ever stated.
|
|
42
|
+
*
|
|
43
|
+
* `schemaCost` is the standing price: those bytes ride every request until the group is withdrawn
|
|
44
|
+
* again. `activationCost` is the one-off, and it is much larger. Adding tool schemas partway
|
|
45
|
+
* through a session is not an additive change the provider can absorb -- it invalidates the cached
|
|
46
|
+
* prompt prefix, and the next request pays fresh input rates for the whole conversation so far.
|
|
47
|
+
*
|
|
48
|
+
* This was believed and reasoned about here for a long time and never measured. It is measured now.
|
|
49
|
+
* Three attempts on `t3-cascade-ledger` flipped the browser group on at turn 6: in all three,
|
|
50
|
+
* cached tokens collapsed to 3,200 at the next request while the prompt kept climbing, and the
|
|
51
|
+
* re-warm cost 14.6%, 21.6% and 23.9% of the attempt -- 20% on average, against a threshold of 10%
|
|
52
|
+
* fixed before the run. See `evals/runs/cache-probe/` and `scripts/cache-probe.mjs`.
|
|
53
|
+
*
|
|
54
|
+
* The same run says what to do about it: arming the same group from the first request cost 16% more
|
|
55
|
+
* than never arming it at all, against 47% for flipping mid-session. Paying up front is roughly
|
|
56
|
+
* three times cheaper than paying when the need appears.
|
|
57
|
+
*/
|
|
40
58
|
export const CAPABILITIES = Object.freeze({
|
|
41
59
|
web: {
|
|
42
60
|
label: "Web access",
|
|
43
61
|
tools: WEB_TOOL_NAMES,
|
|
44
62
|
schemaCost: "about 11 KB of tool schema per request",
|
|
63
|
+
// Not measured directly, and deliberately not extrapolated into a number. It is the larger
|
|
64
|
+
// schema, and pi-web-access is a third-party package that still carries promptSnippet and
|
|
65
|
+
// promptGuidelines, so activating it rebuilds the system prompt as well as the tool schema
|
|
66
|
+
// -- a second invalidation path Browser QA no longer has.
|
|
67
|
+
activationCost:
|
|
68
|
+
"and discards the cached prompt prefix once, which is not measured for this group but is at least as expensive as Browser QA's 20% of attempt cost, because its schema is larger and activating it also rebuilds the system prompt",
|
|
45
69
|
summary: "search the web and fetch page or source content",
|
|
46
70
|
command: "/webaccess",
|
|
47
71
|
},
|
|
@@ -49,6 +73,8 @@ export const CAPABILITIES = Object.freeze({
|
|
|
49
73
|
label: "Browser QA",
|
|
50
74
|
tools: BROWSER_TOOL_NAMES,
|
|
51
75
|
schemaCost: "about 8.7 KB of tool schema per request",
|
|
76
|
+
activationCost:
|
|
77
|
+
"and discards the cached prompt prefix once, measured at about 20% of a mid-length attempt's cost",
|
|
52
78
|
summary: "open pages in an isolated browser to verify rendering, behavior and accessibility",
|
|
53
79
|
command: "/browser",
|
|
54
80
|
},
|