specpi 0.26.0 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +37 -3
  3. package/SECURITY_MODEL.md +44 -4
  4. package/THIRD_PARTY.md +9 -1
  5. package/extensions/jev-advisor/broker.mjs +277 -0
  6. package/extensions/jev-advisor/client.mjs +172 -0
  7. package/extensions/jev-advisor/config.mjs +270 -0
  8. package/extensions/jev-advisor/consent.mjs +133 -0
  9. package/extensions/jev-advisor/gate.mjs +263 -0
  10. package/extensions/jev-advisor/index.ts +999 -0
  11. package/extensions/jev-advisor/key-source.mjs +252 -0
  12. package/extensions/jev-advisor/layer.mjs +169 -0
  13. package/extensions/jev-advisor/ledger.mjs +138 -0
  14. package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
  15. package/extensions/jev-advisor/questions/compaction.mjs +153 -0
  16. package/extensions/jev-advisor/questions/gap.mjs +140 -0
  17. package/extensions/jev-advisor/questions/guard.mjs +168 -0
  18. package/extensions/jev-advisor/questions/progress.mjs +195 -0
  19. package/extensions/jev-advisor/questions/retention.mjs +188 -0
  20. package/extensions/jev-advisor/questions/sources.mjs +91 -0
  21. package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
  22. package/extensions/jev-advisor/risk.mjs +442 -0
  23. package/extensions/jev-advisor/sanitize.mjs +0 -0
  24. package/extensions/jev-advisor/usage.mjs +92 -0
  25. package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
  26. package/extensions/tool-wishlist/index.ts +11 -0
  27. package/extensions/workflow-controls/capabilities.mjs +26 -0
  28. package/extensions/workflow-controls/index.ts +2 -2
  29. package/package.json +1 -1
  30. package/scripts/packages.mjs +56 -0
  31. package/scripts/specpi.mjs +73 -4
@@ -0,0 +1,442 @@
1
+ // Local triage for shell and file calls, before anything is sent anywhere.
2
+ //
3
+ // This is the first half of the native command guard. Jev is the second half and the one that does
4
+ // the analysis; everything here exists to decide what Jev is asked about. Three answers:
5
+ //
6
+ // safe -- settled locally. Jev is never asked, and nothing else will look at this call.
7
+ // dangerous -- blocked locally, with no call and no human.
8
+ // unknown -- Jev is asked.
9
+ //
10
+ // THE THREE ARE NOT SYMMETRIC, and that asymmetry is the whole design:
11
+ //
12
+ // a wrong `safe` is a silent, permanent hole -- the one verdict with no second reader
13
+ // a wrong `unknown` costs one call out of a per-session budget of 208, and Jev decides
14
+ // a wrong `dangerous` blocks real work with no recourse but switching the guard off
15
+ //
16
+ // So the two lists below are maintained under opposite pressures, and the mistake worth naming is
17
+ // treating them as one thing called "the guard" and hardening both the same way.
18
+ //
19
+ // READ_ONLY is a BUDGET mechanism, not a safety one. The guard has 208 calls and each is a few
20
+ // hundred milliseconds awaited on the tool path, so a session that greps and cats a few hundred
21
+ // times would spend the lot and then defer everything for the rest of its life -- a guard that runs
22
+ // out is weaker than one with a fast path. The admission test is therefore NOT "is this harmless".
23
+ // It is: is this binary simple whatever flags it is given, and common enough to be worth it? A
24
+ // binary that can launch a program, write a file or change machine state under any flag is not
25
+ // simple, however harmless its name reads, and Jev parses it. `env` and `fd` launch things,
26
+ // `find` has -delete, `rg` has --pre, `date -s` sets the clock, `hostname` sets the hostname,
27
+ // `file -C` writes a compiled magic file. Every one of those was on this list, and every one was a
28
+ // general bypass that cost almost no budget to keep.
29
+ //
30
+ // CATASTROPHIC is deliberately tiny, and its only real job is the case where Jev is NOT THERE: no
31
+ // key, budget spent, a timeout. Jev catches everything this list would, and weighs intent besides,
32
+ // which a pattern cannot. So resist adding to it. A rule here fires with no model and no human, and
33
+ // a false positive is a blocked session whose only remedy is switching the whole guard off -- and a
34
+ // guard people switch off protects nobody. Ask whether you would stake "this is never legitimate"
35
+ // on it; if not, it belongs in the question set, where being wrong costs one call.
36
+ //
37
+ // And nothing here throws. A classifier that can fail is a classifier that can take a session down,
38
+ // so an unparseable command reads as `unknown` -- ask about it -- rather than as an error.
39
+
40
+ import path from "node:path";
41
+
42
+ /**
43
+ * Shell tools, under every name a harness gives them.
44
+ *
45
+ * `scripts/eval-proxy.mjs` already folds `pwsh`, `shell`, `exec`, `exec_command` and `write_stdin`
46
+ * into `bash`, which means those names reach real sessions. A guard that only knew `bash` would be
47
+ * bypassed by spelling, so the alias list lives here rather than being rediscovered later.
48
+ */
49
+ export const SHELL_TOOLS = Object.freeze([
50
+ "bash",
51
+ "powershell",
52
+ "pwsh",
53
+ "shell",
54
+ "exec",
55
+ "exec_command",
56
+ "write_stdin",
57
+ ]);
58
+
59
+ /**
60
+ * Tools that write a file. The same list `questions/progress.mjs` uses for "the worktree changed",
61
+ * deliberately, because a write the guard cannot see is a write the credential question is never
62
+ * asked about -- and `multi_edit` reaching `~/.ssh/authorized_keys` is exactly that.
63
+ */
64
+ export const WRITE_TOOLS = Object.freeze(["write", "edit", "multi_edit", "apply_patch", "create_file", "str_replace"]);
65
+
66
+ /** The tool calls this guard looks at. Everything else passes without inspection. */
67
+ export const GATED_TOOLS = Object.freeze([...SHELL_TOOLS, ...WRITE_TOOLS]);
68
+
69
+ /**
70
+ * Simple binaries: ones that change nothing whatever flags they are given.
71
+ *
72
+ * "Whatever flags" is the whole test, and it is stricter than it sounds. `git` fails it obviously
73
+ * (`push`, `reset`, `clean`), and special-casing `git status` would only add a subcommand allowlist
74
+ * to maintain beside this one. These fail it less obviously, which is what made each of them a
75
+ * bypass worth having:
76
+ *
77
+ * env, fd launch another program outright -- `env rm -rf build`, `fd -x rm`
78
+ * find has -delete and -exec
79
+ * rg has --pre, which runs an arbitrary preprocessor per file
80
+ * sort, uniq name an output file (`sort -o`, `uniq in out`)
81
+ * date -s sets the system clock
82
+ * hostname with an argument, sets the hostname
83
+ * file -C compiles and writes a magic file
84
+ * printenv changes nothing and hands over a secret, which the risk question also asks about
85
+ *
86
+ * None of them reads as dangerous, and none of them was common enough for the fast path to be
87
+ * buying much. That is the trade: a rare binary on this list saves almost no budget and costs a
88
+ * silent hole, so when in doubt it comes off and Jev parses it.
89
+ *
90
+ * Being here is not a claim that a call is harmless. It is a claim that asking about it would spend
91
+ * the budget without learning anything -- which is why the arguments are still checked below, and
92
+ * `cat ~/.ssh/id_rsa` leaves the fast path even though `cat` never belongs anywhere else.
93
+ */
94
+ const READ_ONLY = new Set([
95
+ "ls",
96
+ "dir",
97
+ "pwd",
98
+ "cd",
99
+ "cat",
100
+ "head",
101
+ "tail",
102
+ "wc",
103
+ "stat",
104
+ "du",
105
+ "df",
106
+ "whoami",
107
+ "uname",
108
+ "echo",
109
+ "printf",
110
+ "which",
111
+ "type",
112
+ "grep",
113
+ "diff",
114
+ "cut",
115
+ "tr",
116
+ "basename",
117
+ "dirname",
118
+ "realpath",
119
+ "true",
120
+ "false",
121
+ ]);
122
+
123
+ /**
124
+ * Shell syntax that can turn a safe-looking command into any other command: chaining, substitution,
125
+ * redirection, background execution. Their presence disqualifies the fast path entirely rather than
126
+ * being parsed, because parsing a shell correctly is not something a guard should be attempting.
127
+ */
128
+ const SHELL_CONTROL = /[;&|><`$(){}\n\r]|\|\||&&/u;
129
+
130
+ /** Targets that mean "everything": the filesystem root, or a bare home directory. */
131
+ const ROOT_TARGETS = new Set(["/", "/*", "~", "~/*", "$HOME", "$HOME/*", "%USERPROFILE%", "%USERPROFILE%\\*"]);
132
+
133
+ /**
134
+ * A recursive delete aimed at the root or a bare home.
135
+ *
136
+ * Written as a function rather than a regex because the regex it replaces anchored on the end of the
137
+ * string, so it matched `rm -rf /` -- which GNU `rm` refuses on its own -- and missed
138
+ * `rm -rf / --no-preserve-root`, which is the spelling that actually empties the disk. Flag order and
139
+ * position are not something a pattern should be asked to track.
140
+ */
141
+ function removesEverything(value) {
142
+ const tokens = value.trim().split(/\s+/u);
143
+ const start = tokens[0] === "sudo" || tokens[0] === "doas" ? 1 : 0;
144
+ if (leadingBinary(tokens[start]) !== "rm") {
145
+ return false;
146
+ }
147
+
148
+ const args = tokens.slice(start + 1);
149
+ const recursive = args.some(
150
+ (token) => token === "--recursive" || (/^-[a-zA-Z]+$/u.test(token) && /[rR]/u.test(token)),
151
+ );
152
+
153
+ return (
154
+ recursive && args.some((token) => !token.startsWith("-") && ROOT_TARGETS.has(token.replace(/\/+$/u, "") || "/"))
155
+ );
156
+ }
157
+
158
+ /**
159
+ * Catastrophic and unambiguous. Every entry here is something that destroys data or hands the
160
+ * machine to someone else, with no legitimate reading in an agent session.
161
+ *
162
+ * Kept short on purpose. These fire without a model and without a human, so a false positive here
163
+ * is a blocked session with no recourse but switching the guard off -- and a guard people switch
164
+ * off protects nobody. Everything requiring judgement is `unknown`.
165
+ */
166
+ const CATASTROPHIC = Object.freeze([
167
+ {
168
+ // `rm -rf /`, `rm -rf / --no-preserve-root`, `rm -r -f ~`. Not `rm -rf ./build`.
169
+ test: removesEverything,
170
+ reason: "recursive delete of the filesystem root or home directory",
171
+ },
172
+ {
173
+ // Writing a raw block device: mkfs, or dd with a device destination.
174
+ pattern: /\b(mkfs(\.\w+)?|fdisk|diskpart)\b|\bdd\b[^\n]*\bof=\/dev\/(sd|nvme|hd|disk)/u,
175
+ reason: "writing directly to a disk device",
176
+ },
177
+ {
178
+ // Piping a downloaded script straight into a shell.
179
+ pattern: /\b(curl|wget|iwr|Invoke-WebRequest)\b[^\n|]*\|\s*(sudo\s+)?(ba|z|k|fi|da)?sh\b/u,
180
+ reason: "executing a downloaded script without inspecting it",
181
+ },
182
+ {
183
+ // Recursive world-writable or ownership changes over a filesystem root.
184
+ pattern:
185
+ /\bchmod\s+(-[a-zA-Z]*\s+)*(-R|--recursive)\s+777\s+\/\s*$|\bchown\s+(-R|--recursive)\s+[^\s]+\s+\/\s*$/u,
186
+ reason: "recursive permission or ownership change over the filesystem root",
187
+ },
188
+ {
189
+ pattern: /:\(\)\s*\{\s*:\|\s*:\s*&\s*\}\s*;\s*:/u,
190
+ reason: "fork bomb",
191
+ },
192
+ {
193
+ // Overwriting the shell history or a credential store with nothing is how a session hides
194
+ // what it did; blocking it is cheap and it is never a legitimate agent action.
195
+ pattern: /\b(history\s+-c|Clear-History)\b|>\s*~?\/?\.bash_history\b/u,
196
+ reason: "clearing shell history",
197
+ },
198
+ ]);
199
+
200
+ /**
201
+ * Paths whose contents are credentials, keys or version-control internals.
202
+ *
203
+ * Every pattern is tested against a path already resolved against the working directory and written
204
+ * with forward slashes, so a segment is bounded by `/` or by the end of the string. Matching a
205
+ * directory matters as much as matching a file: `secrets/api.txt` is a secret, and the first version
206
+ * of this list -- which required the secret's name to be the last segment -- called it an ordinary
207
+ * project file.
208
+ */
209
+ const PROTECTED = Object.freeze([
210
+ /(^|\/)\.env(\.|\/|$)/u,
211
+ /\.env$/u,
212
+ /(^|\/)\.git(\/|$)/u,
213
+ /(^|\/)\.ssh(\/|$)/u,
214
+ /(^|\/)\.aws(\/|$)/u,
215
+ /(^|\/)\.gnupg(\/|$)/u,
216
+ /(^|\/)(id_rsa|id_dsa|id_ecdsa|id_ed25519)(\.|$)/u,
217
+ /\.(pem|key|pfx|p12|keystore|jks)$/iu,
218
+ /(^|\/)(credentials?|secrets?)(\.|\/|$)/iu,
219
+ /(^|\/)auth\.json$/u,
220
+ /(^|\/)\.npmrc$/u,
221
+ /(^|\/)\.netrc$/u,
222
+ /(^|\/)\.pi(\/|$)/u,
223
+ ]);
224
+
225
+ function text(value) {
226
+ return typeof value === "string" ? value : "";
227
+ }
228
+
229
+ /** The binary a command starts with, lowercased and stripped of any path. */
230
+ export function leadingBinary(command) {
231
+ const trimmed = text(command).trim();
232
+ if (trimmed.length === 0) {
233
+ return "";
234
+ }
235
+
236
+ const first = trimmed.split(/\s+/u)[0];
237
+
238
+ return path.basename(first.replace(/\\/gu, "/")).toLowerCase();
239
+ }
240
+
241
+ /**
242
+ * Classify a shell command without asking anyone.
243
+ *
244
+ * Returns `{ decision, reason }` where decision is "safe", "dangerous" or "unknown". The order
245
+ * matters: dangerous is checked before safe, so a catastrophic command hidden behind a read-only
246
+ * binary cannot pass on the fast path.
247
+ */
248
+ export function classifyCommand(command, cwd = process.cwd()) {
249
+ const value = text(command).trim();
250
+ if (value.length === 0) {
251
+ return { decision: "unknown", reason: "empty command" };
252
+ }
253
+
254
+ for (const rule of CATASTROPHIC) {
255
+ if (rule.test ? rule.test(value) : rule.pattern.test(value)) {
256
+ return { decision: "dangerous", reason: rule.reason };
257
+ }
258
+ }
259
+
260
+ // A command with shell control characters is never taken on the fast path, however harmless its
261
+ // first word looks: `ls; rm -rf ~` begins with `ls`.
262
+ if (SHELL_CONTROL.test(value)) {
263
+ return { decision: "unknown", reason: "shell control characters" };
264
+ }
265
+
266
+ if (READ_ONLY.has(leadingBinary(value))) {
267
+ // Read-only is a statement about what the binary does, not about what it is pointed at, and
268
+ // the question this guard asks names exfiltration beside destruction. `cat ~/.ssh/id_rsa`
269
+ // changes nothing and hands over a private key, so the fast path has to look at the
270
+ // arguments too or half of the question it asks is unreachable for the commands that answer
271
+ // it. A protected argument costs one question; every other read stays free.
272
+ return commandArguments(value).some((argument) => protectedPath(argument, cwd))
273
+ ? { decision: "unknown", reason: "reads a protected path" }
274
+ : { decision: "safe", reason: "read-only command" };
275
+ }
276
+
277
+ return { decision: "unknown", reason: "not a known read-only command" };
278
+ }
279
+
280
+ /**
281
+ * The non-flag arguments of a command, unquoted.
282
+ *
283
+ * Deliberately crude: this decides whether to ask a question, never whether to block, so a token it
284
+ * splits wrongly costs a question and nothing else.
285
+ */
286
+ function commandArguments(value) {
287
+ return value
288
+ .split(/\s+/u)
289
+ .slice(1)
290
+ .filter((token) => !token.startsWith("-"))
291
+ .map((token) => token.replace(/^["']|["']$/gu, ""))
292
+ .filter((token) => token.length > 0);
293
+ }
294
+
295
+ /**
296
+ * Whether a write or edit target holds credentials or version-control internals.
297
+ *
298
+ * Relative targets are resolved against the working directory first, so `../../.ssh/config` is seen
299
+ * for what it is rather than for what it is spelled as.
300
+ */
301
+ export function protectedPath(target, cwd = process.cwd()) {
302
+ const value = text(target).trim();
303
+ if (value.length === 0) {
304
+ return false;
305
+ }
306
+
307
+ let resolved;
308
+ try {
309
+ resolved = path.resolve(cwd, value);
310
+ } catch {
311
+ resolved = value;
312
+ }
313
+
314
+ const normalized = resolved.replaceAll("\\", "/");
315
+
316
+ return PROTECTED.some((pattern) => pattern.test(normalized));
317
+ }
318
+
319
+ /** Keys a harness uses for "the file this call writes", across the tools in `WRITE_TOOLS`. */
320
+ const TARGET_KEYS = Object.freeze(["path", "file_path", "filePath", "file", "target", "notebook_path"]);
321
+
322
+ /** Nested arrays of edits, each element carrying a target of its own. */
323
+ const TARGET_LISTS = Object.freeze(["edits", "files", "changes", "operations"]);
324
+
325
+ /** Header forms that name a file inside a patch body. */
326
+ const PATCH_TARGET = /^(?:\*\*\* (?:Add|Update|Delete) File: |--- (?:a\/)?|\+\+\+ (?:b\/)?)(.+)$/gmu;
327
+
328
+ function pushTarget(into, value) {
329
+ if (typeof value === "string" && value.trim().length > 0 && into.length < 64) {
330
+ into.push(value.trim());
331
+ }
332
+ }
333
+
334
+ /**
335
+ * Every file a tool call names, from a tool input whose shape this module does not control.
336
+ *
337
+ * `write` and `edit` carry one `path`; `multi_edit` carries a list; `apply_patch` carries the paths
338
+ * inside a patch body and nowhere else. Reading only the first of those is how a write to a
339
+ * credential file reaches disk without the guard ever seeing a target -- so this reads all three,
340
+ * and returning nothing is itself a meaningful answer to `classifyCall`.
341
+ */
342
+ export function callTargets(input) {
343
+ const found = [];
344
+ if (input === null || typeof input !== "object") {
345
+ return found;
346
+ }
347
+
348
+ for (const key of TARGET_KEYS) {
349
+ pushTarget(found, input[key]);
350
+ }
351
+
352
+ for (const key of TARGET_LISTS) {
353
+ const list = input[key];
354
+ if (!Array.isArray(list)) {
355
+ continue;
356
+ }
357
+
358
+ for (const item of list) {
359
+ if (typeof item === "string") {
360
+ pushTarget(found, item);
361
+ continue;
362
+ }
363
+
364
+ if (item !== null && typeof item === "object") {
365
+ for (const key2 of TARGET_KEYS) {
366
+ pushTarget(found, item[key2]);
367
+ }
368
+ }
369
+ }
370
+ }
371
+
372
+ const patch = typeof input.patch === "string" ? input.patch : typeof input.diff === "string" ? input.diff : "";
373
+ if (patch.length > 0) {
374
+ // Bounded: a patch is arbitrary size and this runs on every call.
375
+ for (const match of patch.slice(0, 20_000).matchAll(PATCH_TARGET)) {
376
+ pushTarget(found, match[1].replace(/\t.*$/u, ""));
377
+ }
378
+ }
379
+
380
+ return found;
381
+ }
382
+
383
+ /** Keys a harness uses for "the text this call runs", across the tools in `SHELL_TOOLS`. */
384
+ const COMMAND_KEYS = Object.freeze(["command", "cmd", "script", "input", "text", "data", "stdin", "line"]);
385
+
386
+ /**
387
+ * The text a shell call will run, from a tool input whose shape this module does not control.
388
+ *
389
+ * `bash` carries `command`; `write_stdin` carries the text it types into a live shell under some
390
+ * other name entirely. Reading `command` alone meant every `write_stdin` was classified as an empty
391
+ * command -- spending a guard call on the empty string while the `rm -rf ~` being typed went
392
+ * unexamined -- so the tool most worth reading was the one read as blank.
393
+ */
394
+ export function commandText(input) {
395
+ if (typeof input === "string") {
396
+ return input;
397
+ }
398
+
399
+ if (input === null || typeof input !== "object") {
400
+ return "";
401
+ }
402
+
403
+ for (const key of COMMAND_KEYS) {
404
+ if (typeof input[key] === "string" && input[key].trim().length > 0) {
405
+ return input[key];
406
+ }
407
+ }
408
+
409
+ return "";
410
+ }
411
+
412
+ /**
413
+ * What a gated tool call is, before Jev is involved.
414
+ *
415
+ * `write` and its siblings are only interesting when they target something protected: an ordinary
416
+ * source file being edited is the entire point of the agent, and asking about each one would spend a
417
+ * session's budget on the first directory it refactored.
418
+ *
419
+ * A write whose target could not be read is `unknown` rather than `safe`. That costs a question on a
420
+ * tool shape this module does not recognise, which is the right way round: the alternative is a
421
+ * silent hole that appears the moment a harness renames a field.
422
+ */
423
+ export function classifyCall({ tool, command, target, targets, cwd }) {
424
+ if (!GATED_TOOLS.includes(tool)) {
425
+ return { decision: "safe", reason: "tool is not gated" };
426
+ }
427
+
428
+ if (SHELL_TOOLS.includes(tool)) {
429
+ return classifyCommand(command, cwd);
430
+ }
431
+
432
+ const all = [...(Array.isArray(targets) ? targets : []), ...(typeof target === "string" ? [target] : [])].filter(
433
+ (item) => typeof item === "string" && item.trim().length > 0,
434
+ );
435
+ if (all.length === 0) {
436
+ return { decision: "unknown", reason: "write target could not be read" };
437
+ }
438
+
439
+ return all.some((item) => protectedPath(item, cwd))
440
+ ? { decision: "unknown", reason: "writes to a protected path" }
441
+ : { decision: "safe", reason: "ordinary project file" };
442
+ }
@@ -0,0 +1,92 @@
1
+ // What the ledger cannot answer: how much of this session's budget is left, right now.
2
+ //
3
+ // The ledger is an audit trail -- every call ever made, append-only, rotated when it grows past
4
+ // eight megabytes. Counting *this session* out of it means knowing where the session began, which
5
+ // nothing outside the advisor's own process does. So the advisor publishes its running total to one
6
+ // small file, rewritten in place, and anything that wants to show a budget reads that instead of
7
+ // replaying a log to find the boundary.
8
+ //
9
+ // It holds counts and nothing else. No state, no questions, no answers, no payload, not even the
10
+ // ledger's digests: a reader learns how many calls a system made and how many changed something,
11
+ // and can learn nothing about what was sent. That is deliberate, because this is the one file in
12
+ // the layer meant to be read by another process.
13
+ //
14
+ // The file always describes the most recent session rather than being deleted at shutdown, and
15
+ // carries `active` to say which. "Nothing is running and the last session spent 6 calls" and "a
16
+ // session is running and has spent 6 calls" are different facts, and a reader that could not tell
17
+ // them apart would report a live budget for a session that ended yesterday.
18
+
19
+ import fs from "node:fs";
20
+ import path from "node:path";
21
+ import { SYSTEM_NAMES, jevDirectory, regularFile, writeFileAtomic } from "./config.mjs";
22
+
23
+ export function usagePath() {
24
+ return path.join(jevDirectory(), "usage.json");
25
+ }
26
+
27
+ /**
28
+ * Replace the snapshot. Failure is swallowed for the same reason the ledger's is: an advisor that
29
+ * cannot write its own bookkeeping must not break the session it is advising, and a missing file
30
+ * reads as "no usage recorded" rather than as zero calls.
31
+ */
32
+ export function writeUsage(snapshot) {
33
+ try {
34
+ writeFileAtomic(usagePath(), `${JSON.stringify(snapshot, null, 4)}\n`);
35
+
36
+ return true;
37
+ } catch {
38
+ return false;
39
+ }
40
+ }
41
+
42
+ /** Every unknown shape reads as absent, never as a partial count. */
43
+ export function normalizeUsage(raw) {
44
+ if (raw?.schema !== 1) {
45
+ return undefined;
46
+ }
47
+
48
+ const systems = {};
49
+ for (const name of SYSTEM_NAMES) {
50
+ const bucket = raw.systems?.[name];
51
+ systems[name] = {
52
+ calls: count(bucket?.calls),
53
+ applied: count(bucket?.applied),
54
+ failed: count(bucket?.failed),
55
+ savedBytes: count(bucket?.savedBytes),
56
+ };
57
+ }
58
+
59
+ return {
60
+ schema: 1,
61
+ session: typeof raw.session === "string" ? raw.session.slice(0, 64) : "",
62
+ startedAt: iso(raw.startedAt),
63
+ updatedAt: iso(raw.updatedAt),
64
+ active: raw.active === true,
65
+ calls: count(raw.calls),
66
+ budgets: Object.fromEntries(["total", ...SYSTEM_NAMES].map((name) => [name, count(raw.budgets?.[name])])),
67
+ systems,
68
+ };
69
+ }
70
+
71
+ function count(value) {
72
+ return Number.isSafeInteger(value) && value >= 0 ? value : 0;
73
+ }
74
+
75
+ function iso(value) {
76
+ return typeof value === "string" && !Number.isNaN(Date.parse(value)) ? value : "";
77
+ }
78
+
79
+ export function readUsage() {
80
+ try {
81
+ const file = usagePath();
82
+ // regularFile refuses links, hard-linked files and anything over 4 KiB. A counts file for
83
+ // eight systems is a few hundred bytes, so the size check is a real one here.
84
+ if (!regularFile(file, "Jev usage")) {
85
+ return undefined;
86
+ }
87
+
88
+ return normalizeUsage(JSON.parse(fs.readFileSync(file, "utf8")));
89
+ } catch {
90
+ return undefined;
91
+ }
92
+ }
@@ -0,0 +1,42 @@
1
+ // Phase 0 of the Jev advisor plan, which needs no model at all.
2
+ //
3
+ // Measured across the recorded eval runs, six of SpecPi's ten offered tools were offered on 31 of
4
+ // 31 attempts and called zero times. Two of them are these: `record_harness_contract` and
5
+ // `finish_harness_improvement` are *authoring* tools, usable only after a human has selected a
6
+ // candidate through /harness-improvement. Whether a selection exists is a fact in local state, so
7
+ // the answer is a boolean — no latency, no cost, no false positives, and nothing for a classifier
8
+ // to route.
9
+ //
10
+ // `report_capability_gap` is deliberately not in this list. It is the *observation* tool and the
11
+ // reason the improvement loop exists; withdrawing it would silently lose the friction reports the
12
+ // loop is built to capture. It stays offered at all times.
13
+
14
+ /** Only usable while an improvement is selected. */
15
+ export const AUTHORING_TOOL_NAMES = Object.freeze(["record_harness_contract", "finish_harness_improvement"]);
16
+
17
+ /**
18
+ * Add or remove the authoring tools without disturbing any other extension's tools. Mirrors the
19
+ * web-access gate: it only ever touches the names it owns, and it does nothing when the active set
20
+ * already matches, so a no-op never costs the cached prompt prefix.
21
+ */
22
+ export function syncAuthoringTools(pi, selected) {
23
+ if (typeof pi?.getActiveTools !== "function" || typeof pi?.setActiveTools !== "function") {
24
+ return false;
25
+ }
26
+
27
+ const owned = new Set(AUTHORING_TOOL_NAMES);
28
+ const active = pi.getActiveTools();
29
+ const present = active.filter((name) => owned.has(name));
30
+ if (selected && present.length === owned.size) {
31
+ return false;
32
+ }
33
+
34
+ if (!selected && present.length === 0) {
35
+ return false;
36
+ }
37
+
38
+ const others = active.filter((name) => !owned.has(name));
39
+ pi.setActiveTools(selected ? [...others, ...AUTHORING_TOOL_NAMES] : others);
40
+
41
+ return true;
42
+ }
@@ -13,6 +13,7 @@ import { getMarkdownTheme, type ExtensionAPI } from "@earendil-works/pi-coding-a
13
13
  import { StringEnum } from "@earendil-works/pi-ai";
14
14
  import { Box, Markdown, Text } from "@earendil-works/pi-tui";
15
15
  import { Type } from "typebox";
16
+ import { syncAuthoringTools } from "./authoring-tools.mjs";
16
17
  import {
17
18
  appendWishlistDecision,
18
19
  archiveWishlist,
@@ -498,6 +499,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
498
499
  let activeRunId = randomUUID();
499
500
  let activeImprovement: ActiveImprovement | undefined;
500
501
  let improvementLifecycleGeneration = 0;
502
+ // The authoring tools ride every request whether or not they can be used. Local state knows
503
+ // when they cannot be, so the answer is a boolean rather than a prediction.
504
+ const syncAuthoring = () => syncAuthoringTools(pi, activeImprovement !== undefined);
501
505
  let improvementMenuGeneration = 0;
502
506
  let finishBusy = false;
503
507
  let contractBusy = false;
@@ -583,15 +587,18 @@ export default function toolWishlist(pi: ExtensionAPI) {
583
587
 
584
588
  pi.on("session_start", (_event, ctx) => {
585
589
  restoreActiveImprovement(ctx);
590
+ syncAuthoring();
586
591
  });
587
592
 
588
593
  pi.on("session_tree", (_event, ctx) => {
589
594
  restoreActiveImprovement(ctx);
595
+ syncAuthoring();
590
596
  });
591
597
 
592
598
  pi.on("session_shutdown", () => {
593
599
  improvementLifecycleGeneration += 1;
594
600
  activeImprovement = undefined;
601
+ syncAuthoring();
595
602
  });
596
603
 
597
604
  const assertImprovementStillCurrent = (expected: ActiveImprovement, generation: number, ctx: any, signal?: any) => {
@@ -1131,6 +1138,7 @@ export default function toolWishlist(pi: ExtensionAPI) {
1131
1138
  });
1132
1139
  if (activeImprovement?.selectionId === selectedImprovement.selectionId) {
1133
1140
  activeImprovement = undefined;
1141
+ syncAuthoring();
1134
1142
  }
1135
1143
 
1136
1144
  return {
@@ -1312,6 +1320,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
1312
1320
  ...policy,
1313
1321
  };
1314
1322
  assertSelectionContextCurrent();
1323
+ // Pi offers tools added during a call from the next assistant message, so restoring
1324
+ // them here lands exactly when the selection they belong to becomes usable.
1325
+ syncAuthoring();
1315
1326
  pi.appendEntry(TASK_CONTRACT_ENTRY, { kind: "cleared" });
1316
1327
  leafId = ctx.sessionManager.getLeafId?.();
1317
1328
  assertSelectionContextCurrent();
@@ -37,11 +37,35 @@ export const BROWSER_TOOL_NAMES = Object.freeze([
37
37
  * the package's own state when it finishes, so a tool-name activation would undo itself.
38
38
  * Delegation needs an activation path inside its own package.
39
39
  */
40
+ /**
41
+ * Restoring a group mid-session costs twice, and only one of those costs was ever stated.
42
+ *
43
+ * `schemaCost` is the standing price: those bytes ride every request until the group is withdrawn
44
+ * again. `activationCost` is the one-off, and it is much larger. Adding tool schemas partway
45
+ * through a session is not an additive change the provider can absorb -- it invalidates the cached
46
+ * prompt prefix, and the next request pays fresh input rates for the whole conversation so far.
47
+ *
48
+ * This was believed and reasoned about here for a long time and never measured. It is measured now.
49
+ * Three attempts on `t3-cascade-ledger` flipped the browser group on at turn 6: in all three,
50
+ * cached tokens collapsed to 3,200 at the next request while the prompt kept climbing, and the
51
+ * re-warm cost 14.6%, 21.6% and 23.9% of the attempt -- 20% on average, against a threshold of 10%
52
+ * fixed before the run. See `evals/runs/cache-probe/` and `scripts/cache-probe.mjs`.
53
+ *
54
+ * The same run says what to do about it: arming the same group from the first request cost 16% more
55
+ * than never arming it at all, against 47% for flipping mid-session. Paying up front is roughly
56
+ * three times cheaper than paying when the need appears.
57
+ */
40
58
  export const CAPABILITIES = Object.freeze({
41
59
  web: {
42
60
  label: "Web access",
43
61
  tools: WEB_TOOL_NAMES,
44
62
  schemaCost: "about 11 KB of tool schema per request",
63
+ // Not measured directly, and deliberately not extrapolated into a number. It is the larger
64
+ // schema, and pi-web-access is a third-party package that still carries promptSnippet and
65
+ // promptGuidelines, so activating it rebuilds the system prompt as well as the tool schema
66
+ // -- a second invalidation path Browser QA no longer has.
67
+ activationCost:
68
+ "and discards the cached prompt prefix once, which is not measured for this group but is at least as expensive as Browser QA's 20% of attempt cost, because its schema is larger and activating it also rebuilds the system prompt",
45
69
  summary: "search the web and fetch page or source content",
46
70
  command: "/webaccess",
47
71
  },
@@ -49,6 +73,8 @@ export const CAPABILITIES = Object.freeze({
49
73
  label: "Browser QA",
50
74
  tools: BROWSER_TOOL_NAMES,
51
75
  schemaCost: "about 8.7 KB of tool schema per request",
76
+ activationCost:
77
+ "and discards the cached prompt prefix once, measured at about 20% of a mid-length attempt's cost",
52
78
  summary: "open pages in an isolated browser to verify rendering, behavior and accessibility",
53
79
  command: "/browser",
54
80
  },