specpi 0.26.0 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/README.md +37 -3
- package/SECURITY_MODEL.md +44 -4
- package/THIRD_PARTY.md +9 -1
- package/extensions/jev-advisor/broker.mjs +277 -0
- package/extensions/jev-advisor/client.mjs +172 -0
- package/extensions/jev-advisor/config.mjs +270 -0
- package/extensions/jev-advisor/consent.mjs +133 -0
- package/extensions/jev-advisor/gate.mjs +263 -0
- package/extensions/jev-advisor/index.ts +999 -0
- package/extensions/jev-advisor/key-source.mjs +252 -0
- package/extensions/jev-advisor/layer.mjs +169 -0
- package/extensions/jev-advisor/ledger.mjs +138 -0
- package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
- package/extensions/jev-advisor/questions/compaction.mjs +153 -0
- package/extensions/jev-advisor/questions/gap.mjs +140 -0
- package/extensions/jev-advisor/questions/guard.mjs +168 -0
- package/extensions/jev-advisor/questions/progress.mjs +195 -0
- package/extensions/jev-advisor/questions/retention.mjs +188 -0
- package/extensions/jev-advisor/questions/sources.mjs +91 -0
- package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
- package/extensions/jev-advisor/risk.mjs +442 -0
- package/extensions/jev-advisor/sanitize.mjs +0 -0
- package/extensions/jev-advisor/usage.mjs +92 -0
- package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
- package/extensions/tool-wishlist/index.ts +11 -0
- package/extensions/workflow-controls/capabilities.mjs +26 -0
- package/extensions/workflow-controls/index.ts +2 -2
- package/package.json +1 -1
- package/scripts/packages.mjs +56 -0
- package/scripts/specpi.mjs +73 -4
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
// Where the Jev layer's key comes from, and the one place that answers it.
|
|
2
|
+
//
|
|
3
|
+
// This file exists because the layer used to answer it wrongly. Jev read `OPENROUTER_API_KEY` from
|
|
4
|
+
// the environment and nothing else, while Pi itself had already resolved an OpenRouter credential
|
|
5
|
+
// for the session through `/login openrouter` and stored it where every other provider stores one.
|
|
6
|
+
// The result was a layer that reported "key: missing" to someone who had a working OpenRouter key
|
|
7
|
+
// sitting in `auth.json`, with no interface anywhere that would have explained the gap. Reading the
|
|
8
|
+
// credential Pi already has is not a new integration; it is stopping an old one from opting out.
|
|
9
|
+
//
|
|
10
|
+
// Pi's documented resolution order (docs/providers.md, "Resolution Order") is:
|
|
11
|
+
//
|
|
12
|
+
// 1. the `--api-key` CLI flag -- Pi's own, scoped to Pi's model calls, not ours
|
|
13
|
+
// 2. `<agent-dir>/auth.json` -- what `/login` writes
|
|
14
|
+
// 3. the provider environment variable
|
|
15
|
+
// 4. custom provider keys in models.json
|
|
16
|
+
//
|
|
17
|
+
// We implement 2 then 3, in that order, so a person who has logged in once is served and a person
|
|
18
|
+
// who exports the variable is still served. Step 1 is Pi's alone and step 4 describes provider
|
|
19
|
+
// catalogue entries the decisions endpoint has no equivalent of.
|
|
20
|
+
//
|
|
21
|
+
// Two rules hold everywhere below.
|
|
22
|
+
//
|
|
23
|
+
// The key is returned by exactly one function, `resolveKey()`, and callers pass it straight to a
|
|
24
|
+
// request header. Nothing else here ever sees the value: `keySource()` returns a label so status
|
|
25
|
+
// output, the Chat panel and the ledger can say where a key came from without any of them being a
|
|
26
|
+
// place a key could leak from. That split is the whole reason this is a module rather than two
|
|
27
|
+
// lines in client.mjs.
|
|
28
|
+
//
|
|
29
|
+
// And a missing or malformed credential is never an error. Every failure path returns undefined,
|
|
30
|
+
// the caller reports `unavailable("no-key")`, and the harness does what it did before the layer
|
|
31
|
+
// existed. A credential store that can throw is a credential store that can take the session down.
|
|
32
|
+
|
|
33
|
+
import fs from "node:fs";
|
|
34
|
+
import path from "node:path";
|
|
35
|
+
import { agentDirectory } from "./config.mjs";
|
|
36
|
+
|
|
37
|
+
/** Pi's provider id for OpenRouter, from the provider table in its own docs. */
|
|
38
|
+
export const OPENROUTER_PROVIDER = "openrouter";
|
|
39
|
+
|
|
40
|
+
// auth.json holds one entry per provider and OAuth entries carry refresh and access tokens, so it
|
|
41
|
+
// is meaningfully larger than the 4 KiB settings bound. This is still small enough that anything
|
|
42
|
+
// above it was not written by Pi, and reading it is what stops a hostile or corrupt file from
|
|
43
|
+
// costing the session a multi-megabyte synchronous read on the tool path.
|
|
44
|
+
const MAX_AUTH_BYTES = 256 * 1024;
|
|
45
|
+
|
|
46
|
+
export function authPath() {
|
|
47
|
+
return path.join(agentDirectory(), "auth.json");
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Pi strips a BOM before parsing and so do we: an auth.json written by a Windows editor parses for
|
|
52
|
+
* Pi and would otherwise fail here, which is the worst kind of difference -- the key works for
|
|
53
|
+
* every model call and appears missing to this layer alone.
|
|
54
|
+
*/
|
|
55
|
+
function stripBom(text) {
|
|
56
|
+
return text.charCodeAt(0) === 0xfeff ? text.slice(1) : text;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* The parsed store, cached against the file's own identity.
|
|
61
|
+
*
|
|
62
|
+
* `ask()` resolves a key per request and retention fires on every large read-only tool result, so an
|
|
63
|
+
* uncached read put a synchronous stat, read and JSON parse of up to 256 KiB on the tool path inside
|
|
64
|
+
* a 1500 ms latency budget -- where it used to be one `process.env` lookup. The cache key is the
|
|
65
|
+
* file's size and modification time, so `/login` writing a new credential mid-session invalidates it
|
|
66
|
+
* on the next call rather than being masked until restart, which a plain memo would have done.
|
|
67
|
+
*
|
|
68
|
+
* A file that will not parse is cached as firmly as one that will. Recording only successes left the
|
|
69
|
+
* worst case uncached: a truncated or hand-edited auth.json threw on every call, so every request
|
|
70
|
+
* paid a fresh stat, a 256 KiB read and a failing parse inside the same latency budget the cache
|
|
71
|
+
* exists to protect -- forever, since nothing about it would change until the file did.
|
|
72
|
+
*/
|
|
73
|
+
let parsedStore = { key: "", data: undefined };
|
|
74
|
+
|
|
75
|
+
function readStore(file, stat) {
|
|
76
|
+
const identity = `${stat.mtimeMs}:${stat.size}:${file}`;
|
|
77
|
+
if (parsedStore.key === identity) {
|
|
78
|
+
return parsedStore.data;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
let data;
|
|
82
|
+
try {
|
|
83
|
+
const parsed = JSON.parse(stripBom(fs.readFileSync(file, "utf8")));
|
|
84
|
+
data = parsed && typeof parsed === "object" && !Array.isArray(parsed) ? parsed : undefined;
|
|
85
|
+
} catch {
|
|
86
|
+
// An unreadable or unparseable store is "no credential", which is the same answer the caller
|
|
87
|
+
// would have reached by catching this; caching it is what stops it being recomputed.
|
|
88
|
+
data = undefined;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
parsedStore = { key: identity, data };
|
|
92
|
+
|
|
93
|
+
return data;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* The raw stored entry for one provider, or undefined. Deliberately not exported: an entry is a
|
|
98
|
+
* credential, and the only thing outside this file that needs one is the request header.
|
|
99
|
+
*/
|
|
100
|
+
function storedCredential(providerId) {
|
|
101
|
+
try {
|
|
102
|
+
const file = authPath();
|
|
103
|
+
// `stat`, not `lstat`: a symlinked auth.json has to resolve, because dotfile managers like
|
|
104
|
+
// chezmoi and stow routinely link it into a managed directory. Refusing links here would
|
|
105
|
+
// recreate the exact divergence this module was written to remove -- Pi resolves the
|
|
106
|
+
// credential and every model call works, while this layer alone reports "key: none found"
|
|
107
|
+
// and gives no way to tell that from a missing key.
|
|
108
|
+
//
|
|
109
|
+
// config.mjs refuses links on its own files for a reason that does not apply here: those
|
|
110
|
+
// are writes, where a link can redirect a trusted write somewhere it was not meant to go.
|
|
111
|
+
// This is a bounded read of a file Pi owns, so an unsupported shape reads as "no credential"
|
|
112
|
+
// rather than throwing. It is not our file to have opinions about.
|
|
113
|
+
const stat = fs.statSync(file, { throwIfNoEntry: false });
|
|
114
|
+
if (!stat || !stat.isFile() || stat.size > MAX_AUTH_BYTES) {
|
|
115
|
+
return undefined;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return readStore(file, stat)?.[providerId];
|
|
119
|
+
} catch {
|
|
120
|
+
return undefined;
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* An api_key credential's key, or undefined for anything else.
|
|
126
|
+
*
|
|
127
|
+
* An `oauth` entry is ignored rather than unwrapped. Pi refreshes OAuth tokens inside a lock in its
|
|
128
|
+
* own credential store, and a second process reading an access token out of the file would be
|
|
129
|
+
* reading a value that may already have been rotated -- and would be doing it without the lock.
|
|
130
|
+
* OpenRouter's own login mints a durable `api_key` anyway, so the case this skips is not the case
|
|
131
|
+
* anyone reaches.
|
|
132
|
+
*/
|
|
133
|
+
function apiKeyOf(credential) {
|
|
134
|
+
if (!credential || typeof credential !== "object" || credential.type !== "api_key") {
|
|
135
|
+
return undefined;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const key = credential.key;
|
|
139
|
+
|
|
140
|
+
return typeof key === "string" && key.trim().length > 0 ? key.trim() : undefined;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function environmentKey(name) {
|
|
144
|
+
const value = process.env[name];
|
|
145
|
+
|
|
146
|
+
return typeof value === "string" && value.trim().length > 0 ? value.trim() : undefined;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Whether the credential store is consulted at all.
|
|
151
|
+
*
|
|
152
|
+
* `JEV_KEY_SOURCE=environment` restricts resolution to the environment variables. That exists for
|
|
153
|
+
* this repository's own scripts, and it is a correctness fix rather than a convenience: the
|
|
154
|
+
* calibration and triage runs load a key from the gitignored `evals/.env` and AGENTS.md promises
|
|
155
|
+
* "a variable already set in the shell always wins". Once `auth.json` was consulted first, those
|
|
156
|
+
* runs would silently bill a developer's personal `/login openrouter` account instead of the eval
|
|
157
|
+
* key, and `--probe` would verify a key the run then did not use.
|
|
158
|
+
*/
|
|
159
|
+
function environmentOnly() {
|
|
160
|
+
return process.env.JEV_KEY_SOURCE === "environment";
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Jev is reached through OpenRouter by default: that is where it is published, it is what
|
|
165
|
+
* the command guard uses, and an OpenRouter key (`sk-or-...`) is rejected by the direct
|
|
166
|
+
* TypeSafe API with a bare 401. `JEV_BACKEND=typesafe` selects the direct API for a TypeSafe key.
|
|
167
|
+
*
|
|
168
|
+
* It lives here rather than in client.mjs because everything below has to bind it. A parameter
|
|
169
|
+
* defaulting to the string "openrouter" is not a binding: re-exporting such a function under a name
|
|
170
|
+
* whose previous version read the backend itself silently rebound every no-arg caller to the wrong
|
|
171
|
+
* route, which is how `keyPresent()` came to report a key that `resolveKey()` would not return.
|
|
172
|
+
*/
|
|
173
|
+
export function backend() {
|
|
174
|
+
return process.env.JEV_BACKEND === "typesafe" ? "typesafe" : "openrouter";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
export function keyEnvName(route = backend()) {
|
|
178
|
+
return route === "typesafe" ? "TYPESAFE_API_KEY" : "OPENROUTER_API_KEY";
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Every place a key for this backend could come from, in the order they are consulted, each with
|
|
183
|
+
* whether it currently holds one. This is what `/jev status`, `specpi doctor` and the Chat panel
|
|
184
|
+
* render, and it carries labels only -- never a key, not even a truncated one.
|
|
185
|
+
*
|
|
186
|
+
* The list is returned whole rather than filtered to the winner because "which of these do I need
|
|
187
|
+
* to fix" is the question someone with no key is actually asking, and a bare "missing" has never
|
|
188
|
+
* answered it.
|
|
189
|
+
*/
|
|
190
|
+
export function keySources(route = backend()) {
|
|
191
|
+
const variable = keyEnvName(route);
|
|
192
|
+
const sources = [];
|
|
193
|
+
// Only the OpenRouter route has a provider entry to read: `auth.json` is keyed by Pi provider
|
|
194
|
+
// id, and the direct TypeSafe API is not one of Pi's providers.
|
|
195
|
+
if (route !== "typesafe" && !environmentOnly()) {
|
|
196
|
+
sources.push({
|
|
197
|
+
name: "auth.json",
|
|
198
|
+
label: `Pi credential store (${OPENROUTER_PROVIDER})`,
|
|
199
|
+
detail: "/login openrouter",
|
|
200
|
+
present: apiKeyOf(storedCredential(OPENROUTER_PROVIDER)) !== undefined,
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
sources.push({
|
|
205
|
+
name: variable,
|
|
206
|
+
label: `${variable} in the environment`,
|
|
207
|
+
detail: `export ${variable}=...`,
|
|
208
|
+
present: environmentKey(variable) !== undefined,
|
|
209
|
+
});
|
|
210
|
+
|
|
211
|
+
// Accepted on the OpenRouter route so an env file predating the OpenRouter default keeps
|
|
212
|
+
// working. Listed last because it is a compatibility path, and listed at all because a person
|
|
213
|
+
// whose key is only here should be able to see that that is why it still works.
|
|
214
|
+
if (route !== "typesafe") {
|
|
215
|
+
sources.push({
|
|
216
|
+
name: "TYPESAFE_API_KEY",
|
|
217
|
+
label: "TYPESAFE_API_KEY in the environment (legacy)",
|
|
218
|
+
detail: "export TYPESAFE_API_KEY=...",
|
|
219
|
+
present: environmentKey("TYPESAFE_API_KEY") !== undefined,
|
|
220
|
+
});
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
return sources;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/** The name of the source a key would be taken from, or undefined when there is none. */
|
|
227
|
+
export function keySource(route = backend()) {
|
|
228
|
+
return keySources(route).find((source) => source.present)?.name;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* The key itself. The only function here that returns one, and the only caller is the request
|
|
233
|
+
* header in client.mjs.
|
|
234
|
+
*/
|
|
235
|
+
export function resolveKey(route = backend()) {
|
|
236
|
+
if (route !== "typesafe" && !environmentOnly()) {
|
|
237
|
+
const stored = apiKeyOf(storedCredential(OPENROUTER_PROVIDER));
|
|
238
|
+
if (stored) {
|
|
239
|
+
return stored;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
return environmentKey(keyEnvName(route)) ?? (route !== "typesafe" ? environmentKey("TYPESAFE_API_KEY") : undefined);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* Whether any source holds a key. Callers only ever ask this; the client reads the value itself at
|
|
248
|
+
* call time, so a key never has to exist inside a structure that something might log or serialize.
|
|
249
|
+
*/
|
|
250
|
+
export function keyPresent(route = backend()) {
|
|
251
|
+
return keySources(route).some((source) => source.present);
|
|
252
|
+
}
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
// What `/jev on` and `/jev off` actually do, as functions a test can call.
|
|
2
|
+
//
|
|
3
|
+
// This logic used to live in closures inside index.ts, which no test imports -- every test in the
|
|
4
|
+
// suite reaches for a `.mjs` module, so the whole feature was covered only by prose. Two review
|
|
5
|
+
// rounds found the consequences: a guard preference destroyed on a path that had decided nothing, a
|
|
6
|
+
// fail-closed gate armed by a headless session that was told nothing had changed, a message
|
|
7
|
+
// reporting the guard "left off" while it was on and blocking, and a rollback that never happened
|
|
8
|
+
// because the flag was set before the write. None of those were visible to a green suite, because
|
|
9
|
+
// the suite could not see them at all.
|
|
10
|
+
//
|
|
11
|
+
// So the shape here is deliberate: no module state, no reads of `process` beyond an injected `env`,
|
|
12
|
+
// and every effect either returned as data or performed through an injected dependency. The caller
|
|
13
|
+
// owns the session; this decides what should happen to it and says so.
|
|
14
|
+
|
|
15
|
+
import { SYSTEM_NAMES } from "./config.mjs";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Decide what a layer switch means, and report it honestly.
|
|
19
|
+
*
|
|
20
|
+
* `state` is `{ settings }` and is never mutated -- the next state comes back in the result. `deps`
|
|
21
|
+
* supplies the outside world, which is now only `keySources`: the command guard used to need a
|
|
22
|
+
* package's global configuration file arbitrated here, and as the eighth system it needs nothing.
|
|
23
|
+
*
|
|
24
|
+
* Scope belongs to `layerScopeLine`, not here, so this takes `{ on }` and nothing else. It used to
|
|
25
|
+
* be handed `sessionOnly` and `interactive` as well and read neither, which reads as a decision
|
|
26
|
+
* being made from them.
|
|
27
|
+
*/
|
|
28
|
+
export function applyLayer({ on }, state, deps) {
|
|
29
|
+
if (!on) {
|
|
30
|
+
return { settings: { ...state.settings, master: false }, lines: [offLine()] };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// Resolved inside the on branch only. `offLine` never names a key, so doing this first meant
|
|
34
|
+
// every `/jev off` paid for a stat, read and parse of Pi's credential store to discard it --
|
|
35
|
+
// the one file this layer reads under a narrow, stated exception.
|
|
36
|
+
const active = deps.keySources().find((item) => item.present)?.name;
|
|
37
|
+
const settings = enableSystems(state.settings);
|
|
38
|
+
|
|
39
|
+
return { settings, lines: onLines(state.settings, settings, active) };
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* What arming the command guard means, in the words every path that arms it has to use.
|
|
44
|
+
*
|
|
45
|
+
* Seven of the eight systems only ever add advice; this one can refuse a tool call. Saying so is a
|
|
46
|
+
* rule rather than a nicety, and it lives here because it was a rule `/jev on` honoured alone while
|
|
47
|
+
* `/jev enable guard` and `/jev startup on` armed the same system in silence.
|
|
48
|
+
*/
|
|
49
|
+
export function guardWarning() {
|
|
50
|
+
return (
|
|
51
|
+
"guard is the only system that can refuse a tool call: it blocks a call it reads as " +
|
|
52
|
+
"destructive and not what was asked for, asks you about the uncertain ones, and hands " +
|
|
53
|
+
"everything else to the permission system unchanged. Turn it off with /jev disable guard."
|
|
54
|
+
);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Enabling the layer enables its systems, because a layer with none on runs and does nothing -- the
|
|
59
|
+
* state people kept arriving at, with the notification cheerfully reporting "0 of 8".
|
|
60
|
+
*
|
|
61
|
+
* Only when none are on. Someone deliberately running retention alone has expressed a preference,
|
|
62
|
+
* and `/jev off` then `/jev on` must not hand back the seven they turned off.
|
|
63
|
+
*
|
|
64
|
+
* The command guard is one of the eight, and it is the only system that can refuse a tool call. That
|
|
65
|
+
* is a deliberate answer to a question this file and `config.mjs` resolve differently on purpose:
|
|
66
|
+
* `/jev on` is a person acting now, so it arms everything and `onLines` says in as many words that
|
|
67
|
+
* one of them can refuse a command; `migrateToThree` runs without anyone present, so it arms
|
|
68
|
+
* nothing new. Silence is the difference -- an unattended migration must not change what a session
|
|
69
|
+
* is allowed to run, and an explicit command that reports what it did may.
|
|
70
|
+
*/
|
|
71
|
+
export function enableSystems(settings) {
|
|
72
|
+
const chosen = SYSTEM_NAMES.filter((name) => settings.systems[name]);
|
|
73
|
+
|
|
74
|
+
return {
|
|
75
|
+
...settings,
|
|
76
|
+
master: true,
|
|
77
|
+
systems: chosen.length > 0 ? settings.systems : Object.fromEntries(SYSTEM_NAMES.map((name) => [name, true])),
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function onLines(before, after, activeSource) {
|
|
82
|
+
const chosen = SYSTEM_NAMES.filter((name) => before.systems[name]);
|
|
83
|
+
const active = SYSTEM_NAMES.filter((name) => after.systems[name]);
|
|
84
|
+
const lines = [`Jev layer on with ${active.length} of ${SYSTEM_NAMES.length} systems: ${active.join(", ")}.`];
|
|
85
|
+
if (chosen.length === 0) {
|
|
86
|
+
lines.push("No system was enabled, so all of them were. Turn any back off with /jev disable <system>.");
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Said out loud every time, because seven of the eight only ever add advice and this one can
|
|
90
|
+
// take a command away. Nobody should discover that from a blocked call.
|
|
91
|
+
if (after.systems.guard) {
|
|
92
|
+
lines.push(guardWarning());
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
lines.push(keyLine(activeSource));
|
|
96
|
+
|
|
97
|
+
return lines;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
function offLine() {
|
|
101
|
+
return "Jev layer off. No state leaves this machine, and every tool call goes to the permission system.";
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Where the key is coming from, by name and never by value. Takes the already-resolved source name
|
|
106
|
+
* rather than looking it up, so one command cannot read the credential store twice.
|
|
107
|
+
*/
|
|
108
|
+
export function keyLine(source) {
|
|
109
|
+
if (source) {
|
|
110
|
+
return `Key: found in ${source === "auth.json" ? "Pi's credential store (auth.json)" : source}.`;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
return (
|
|
114
|
+
"Key: none found, so every system will report no advice and the harness runs exactly as it did before. " +
|
|
115
|
+
"Run /login openrouter to store one, or set OPENROUTER_API_KEY."
|
|
116
|
+
);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* The settings to write so that turning the layer on is remembered.
|
|
121
|
+
*
|
|
122
|
+
* `master` and `startup` always move together. Storing them apart is what made the Chat panel's
|
|
123
|
+
* "enabled" checkbox do nothing on its own: `session_start` zeroes a stored master whenever
|
|
124
|
+
* `startup` is false, so `master: true, startup: false` describes a layer that is on and never runs.
|
|
125
|
+
*
|
|
126
|
+
* The stored file is the base, so budgets and the nudge mode written by Chat while this session was
|
|
127
|
+
* running survive -- including any system the panel enabled since this session started, which is why
|
|
128
|
+
* `systems` is merged rather than written over.
|
|
129
|
+
*/
|
|
130
|
+
export function layerToPersist({ settings }, stored) {
|
|
131
|
+
return {
|
|
132
|
+
...stored,
|
|
133
|
+
master: settings.master,
|
|
134
|
+
startup: settings.master,
|
|
135
|
+
// Merged onto the stored map, not written over it. This session's copy may predate systems
|
|
136
|
+
// enabled on disk since it started -- by the Chat panel, or by another session -- and
|
|
137
|
+
// writing it whole turned those back off with nothing reporting it.
|
|
138
|
+
systems: { ...stored.systems, ...settings.systems },
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** What the change applies to: this session, or every session from now on. */
|
|
143
|
+
export function layerScopeLine({ sessionOnly, interactive, persisted, stored, settingsFile }) {
|
|
144
|
+
if (sessionOnly) {
|
|
145
|
+
return `This session only, as asked. New sessions still start ${stored.startup && stored.master ? "on" : "off"}.`;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
if (!interactive) {
|
|
149
|
+
return "This session only: writing the startup preference needs an interactive command.";
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
if (!persisted) {
|
|
153
|
+
return `This session only: ${settingsFile} could not be written.`;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
return `Remembered -- new Pi sessions start this way too. Preference: ${settingsFile}`;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* `/jev startup on|off`, which was the last command still writing one half of the pair -- the exact
|
|
161
|
+
* trap this module exists to remove, left in the command named after it, while reporting that new
|
|
162
|
+
* sessions would start on when they would not.
|
|
163
|
+
*/
|
|
164
|
+
export function startupToPersist(wanted, stored) {
|
|
165
|
+
// Delegates to `enableSystems` rather than restating the rule. Three copies of "what enabling
|
|
166
|
+
// the layer means" -- here, there, and `couple` in the Chat panel -- is precisely how the
|
|
167
|
+
// advisor and the panel drift apart, which is the class of bug this module was extracted over.
|
|
168
|
+
return wanted ? { ...enableSystems(stored), startup: true } : { ...stored, master: false, startup: false };
|
|
169
|
+
}
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
// "We only send digests" is a promise until someone can check it. This ledger is the check: one
|
|
2
|
+
// line per call recording what was sent, how big it was and what came back, with a SHA-256 of the
|
|
3
|
+
// exact payload and never the payload itself.
|
|
4
|
+
//
|
|
5
|
+
// It doubles as the calibration corpus. Thresholds in this layer are meant to be read off a curve
|
|
6
|
+
// rather than guessed, and the curve has to come from somewhere.
|
|
7
|
+
|
|
8
|
+
import fs from "node:fs";
|
|
9
|
+
import path from "node:path";
|
|
10
|
+
import { createHash } from "node:crypto";
|
|
11
|
+
import { jevDirectory, regularFile } from "./config.mjs";
|
|
12
|
+
|
|
13
|
+
const MAX_LINES = 5000;
|
|
14
|
+
const MAX_LEDGER_BYTES = 8 * 1024 * 1024;
|
|
15
|
+
|
|
16
|
+
export function ledgerPath() {
|
|
17
|
+
return path.join(jevDirectory(), "transmissions.jsonl");
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function payloadDigest(payload) {
|
|
21
|
+
return createHash("sha256").update(JSON.stringify(payload)).digest("hex");
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Append one record. Ledger failure never fails the call that produced it: an advisor that cannot
|
|
26
|
+
* write its audit line still must not break the session, so the error is swallowed and the caller
|
|
27
|
+
* proceeds. The absence of a line is itself visible in `/jev ledger`.
|
|
28
|
+
*/
|
|
29
|
+
export function record(entry) {
|
|
30
|
+
try {
|
|
31
|
+
const file = ledgerPath();
|
|
32
|
+
fs.mkdirSync(path.dirname(file), { recursive: true, mode: 0o700 });
|
|
33
|
+
if (fs.existsSync(file)) {
|
|
34
|
+
const stat = fs.lstatSync(file, { throwIfNoEntry: false });
|
|
35
|
+
if (!stat || !stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1) {
|
|
36
|
+
return false;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
if (stat.size > MAX_LEDGER_BYTES) {
|
|
40
|
+
rotate(file);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
fs.appendFileSync(file, `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\n`, {
|
|
45
|
+
mode: 0o600,
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
return true;
|
|
49
|
+
} catch {
|
|
50
|
+
return false;
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Keep the newest half. Unbounded growth is a worse failure than losing old audit lines. */
|
|
55
|
+
function rotate(file) {
|
|
56
|
+
const lines = fs.readFileSync(file, "utf8").split("\n").filter(Boolean);
|
|
57
|
+
const kept = lines.slice(-Math.floor(MAX_LINES / 2));
|
|
58
|
+
fs.writeFileSync(file, kept.length > 0 ? `${kept.join("\n")}\n` : "", { mode: 0o600 });
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Roll the ledger up into the few numbers a report wants. Kept here rather than in the eval suite
|
|
63
|
+
* because the shape of a line is this module's business, and a reader that has to know it is a
|
|
64
|
+
* second copy of the schema waiting to drift.
|
|
65
|
+
*/
|
|
66
|
+
export function summarize(entries) {
|
|
67
|
+
const lines = Array.isArray(entries) ? entries : [];
|
|
68
|
+
const bySystem = {};
|
|
69
|
+
for (const entry of lines) {
|
|
70
|
+
const name = typeof entry?.system === "string" ? entry.system : "unknown";
|
|
71
|
+
const bucket = (bySystem[name] ??= {
|
|
72
|
+
calls: 0,
|
|
73
|
+
failed: 0,
|
|
74
|
+
applied: 0,
|
|
75
|
+
savedBytes: 0,
|
|
76
|
+
stateBytes: 0,
|
|
77
|
+
outcomes: {},
|
|
78
|
+
});
|
|
79
|
+
bucket.calls += 1;
|
|
80
|
+
bucket.failed += entry?.ok === true ? 0 : 1;
|
|
81
|
+
bucket.applied += entry?.applied === true ? 1 : 0;
|
|
82
|
+
bucket.savedBytes += Number.isFinite(entry?.savedBytes) ? entry.savedBytes : 0;
|
|
83
|
+
bucket.stateBytes += Number.isFinite(entry?.stateBytes) ? entry.stateBytes : 0;
|
|
84
|
+
// "Asked 3 times, applied 0" is a number. "Asked 3 times, applied 0, all three because the
|
|
85
|
+
// result might hold the answer" is a finding.
|
|
86
|
+
const outcome =
|
|
87
|
+
typeof entry?.outcome === "string" ? entry.outcome : entry?.ok === true ? "unrecorded" : "failed";
|
|
88
|
+
bucket.outcomes[outcome] = (bucket.outcomes[outcome] ?? 0) + 1;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const totals = Object.values(bySystem);
|
|
92
|
+
|
|
93
|
+
return {
|
|
94
|
+
calls: lines.length,
|
|
95
|
+
failed: totals.reduce((sum, item) => sum + item.failed, 0),
|
|
96
|
+
applied: totals.reduce((sum, item) => sum + item.applied, 0),
|
|
97
|
+
savedBytes: totals.reduce((sum, item) => sum + item.savedBytes, 0),
|
|
98
|
+
stateBytes: totals.reduce((sum, item) => sum + item.stateBytes, 0),
|
|
99
|
+
// Named for what retention does, because it is the only system that shortens anything and
|
|
100
|
+
// the number is meaningless averaged with systems that cannot.
|
|
101
|
+
elisions: bySystem.retention?.applied ?? 0,
|
|
102
|
+
bytesDropped: bySystem.retention?.savedBytes ?? 0,
|
|
103
|
+
bySystem,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Every line, for a caller that wants to summarize rather than display. `read` is for display. */
|
|
108
|
+
export function readAll() {
|
|
109
|
+
return read(Number.MAX_SAFE_INTEGER);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export function read(limit = 20) {
|
|
113
|
+
try {
|
|
114
|
+
const file = ledgerPath();
|
|
115
|
+
if (!regularFile(file, "Jev ledger")) {
|
|
116
|
+
return [];
|
|
117
|
+
}
|
|
118
|
+
} catch {
|
|
119
|
+
// An oversize ledger is still readable; only a link or irregular file is refused.
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
try {
|
|
123
|
+
const lines = fs.readFileSync(ledgerPath(), "utf8").split("\n").filter(Boolean);
|
|
124
|
+
|
|
125
|
+
return lines
|
|
126
|
+
.slice(-Math.max(1, limit))
|
|
127
|
+
.map((line) => {
|
|
128
|
+
try {
|
|
129
|
+
return JSON.parse(line);
|
|
130
|
+
} catch {
|
|
131
|
+
return undefined;
|
|
132
|
+
}
|
|
133
|
+
})
|
|
134
|
+
.filter(Boolean);
|
|
135
|
+
} catch {
|
|
136
|
+
return [];
|
|
137
|
+
}
|
|
138
|
+
}
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
// System 6: decide, once, before the first provider request, whether this session is going to need
|
|
2
|
+
// a withdrawn tool group -- and if so, offer it then rather than in the middle.
|
|
3
|
+
//
|
|
4
|
+
// This is permitted by the standing rule because it is a once-per-session decision made before the
|
|
5
|
+
// first request, which is one of the three cache-safe shapes. It is also distinct from the tool
|
|
6
|
+
// router this project rejected: that rejection was about six tools whose usefulness local state can
|
|
7
|
+
// already determine from wishlist and goal state. "Will this task need a browser" is not in local
|
|
8
|
+
// state, and no amount of inspecting the repository answers it.
|
|
9
|
+
//
|
|
10
|
+
// It exists because Phase 7 measured what the rejection was always assumed to be worth. Flipping
|
|
11
|
+
// Browser QA on at turn 6 of `t3-cascade-ledger` collapsed cached tokens to 3,200 at the next
|
|
12
|
+
// request in three attempts out of three, while the prompt kept climbing, and the re-warm cost 20%
|
|
13
|
+
// of the attempt on average. Arming the same group from the first request cost 16% more than never
|
|
14
|
+
// arming it, against 47% for flipping mid-session. Paying up front is about three times cheaper
|
|
15
|
+
// than paying when the need appears -- so the whole value of this system is moving the decision
|
|
16
|
+
// earlier, and that is all it does.
|
|
17
|
+
//
|
|
18
|
+
// It holds no authority. It activates nothing on its own: it pre-fills the same confirmation the
|
|
19
|
+
// human would have seen from `request_capability`, at turn 0 instead of turn 6. A decline is
|
|
20
|
+
// remembered for the session, and with no interactive human it proposes nothing at all.
|
|
21
|
+
|
|
22
|
+
import { choice, noul } from "../client.mjs";
|
|
23
|
+
import { choiceValue, nounTrue } from "../gate.mjs";
|
|
24
|
+
import { compact } from "../sanitize.mjs";
|
|
25
|
+
|
|
26
|
+
export const TASK_KINDS = Object.freeze({
|
|
27
|
+
code: "Reading or changing source code in this repository",
|
|
28
|
+
web: "Looking something up online, or reading external pages",
|
|
29
|
+
ui: "Checking how a page renders or behaves in a browser",
|
|
30
|
+
ops: "Builds, releases, configuration or tooling",
|
|
31
|
+
docs: "Writing or editing prose",
|
|
32
|
+
other: "Anything else",
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Cheap, local, and checked before anything is sent. The rule is ask local state first, and while
|
|
37
|
+
* local state cannot answer "will this need a browser", it answers "is that question worth asking":
|
|
38
|
+
* a task with no web signal in its wording and a repository with no web assets is not one.
|
|
39
|
+
*/
|
|
40
|
+
const PROMPT_SIGNALS =
|
|
41
|
+
/\b(https?:\/\/|www\.|url|link|page|site|website|web|browser|render|screenshot|accessibility|a11y|css|dom|localhost|search online|look .{0,12}up)\b/iu;
|
|
42
|
+
|
|
43
|
+
/** Names that mean this repository has something a browser could open. */
|
|
44
|
+
const WEB_ASSETS = /^(index\.html|.*\.html?|vite\.config\..*|next\.config\..*|svelte\.config\..*|astro\.config\..*)$/iu;
|
|
45
|
+
|
|
46
|
+
export function promptSignal(prompt) {
|
|
47
|
+
return PROMPT_SIGNALS.test(String(prompt ?? ""));
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** One bounded readdir of the working directory. Never recursive: this is a hint, not a survey. */
|
|
51
|
+
export function repositorySignal(entries) {
|
|
52
|
+
return (entries ?? []).slice(0, 200).some((name) => WEB_ASSETS.test(String(name)));
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function localSignals({ prompt, entries }) {
|
|
56
|
+
const reasons = [];
|
|
57
|
+
if (promptSignal(prompt)) {
|
|
58
|
+
reasons.push("prompt-mentions-web");
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
if (repositorySignal(entries)) {
|
|
62
|
+
reasons.push("repository-has-web-assets");
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
return { ask: reasons.length > 0, reasons };
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* The prompt is the one place in this layer where the user's own words are sent rather than a
|
|
70
|
+
* digest of them, and it is bounded hard for that reason. It is the only thing that can answer the
|
|
71
|
+
* question, and it is one message rather than a transcript.
|
|
72
|
+
*/
|
|
73
|
+
export function buildInput({ prompt, reasons, available, cwdEntries = [] }) {
|
|
74
|
+
return {
|
|
75
|
+
request: compact(prompt ?? "", 400),
|
|
76
|
+
reasons,
|
|
77
|
+
withdrawnGroups: available,
|
|
78
|
+
workspaceFiles: cwdEntries.slice(0, 12).map((name) => compact(name, 40)),
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
export function questions({ available = [] } = {}) {
|
|
83
|
+
const asked = { task_kind: choice("What kind of work is this request asking for?", TASK_KINDS) };
|
|
84
|
+
if (available.includes("web")) {
|
|
85
|
+
asked.needs_web = noul("Completing this request will require searching the web or fetching an external page");
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
if (available.includes("browser")) {
|
|
89
|
+
asked.needs_browser = noul(
|
|
90
|
+
"Completing this request will require opening a page in a browser to check how it renders or behaves",
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// Asked whenever the session is capable of it, because the answer is a suggestion to the human
|
|
95
|
+
// rather than an activation: delegation binds a model and a host and needs its own command.
|
|
96
|
+
asked.needs_delegation = noul(
|
|
97
|
+
"This request would be better answered by reading a large set of files than by reasoning about a few",
|
|
98
|
+
);
|
|
99
|
+
|
|
100
|
+
return asked;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Deliberately asymmetric. A false positive costs the group's schema on every request for the rest
|
|
105
|
+
* of the session and a confirmation the human did not need; a false negative costs nothing at all,
|
|
106
|
+
* because it leaves today's behaviour exactly as it is and `request_capability` is still there. So
|
|
107
|
+
* this proposes only on a high bar, and the `capability` thresholds are the strictest in the layer.
|
|
108
|
+
*/
|
|
109
|
+
export function decide(answers, available = []) {
|
|
110
|
+
const propose = [];
|
|
111
|
+
if (available.includes("web") && nounTrue(answers?.needs_web, "capability")) {
|
|
112
|
+
propose.push("web");
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
if (available.includes("browser") && nounTrue(answers?.needs_browser, "capability")) {
|
|
116
|
+
propose.push("browser");
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
return {
|
|
120
|
+
propose,
|
|
121
|
+
taskKind: choiceValue(answers?.task_kind, "capability"),
|
|
122
|
+
suggestDelegation: nounTrue(answers?.needs_delegation, "capability"),
|
|
123
|
+
};
|
|
124
|
+
}
|