specpi 0.26.0 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +80 -0
- package/README.md +37 -3
- package/SECURITY_MODEL.md +44 -4
- package/THIRD_PARTY.md +9 -1
- package/extensions/jev-advisor/broker.mjs +277 -0
- package/extensions/jev-advisor/client.mjs +172 -0
- package/extensions/jev-advisor/config.mjs +270 -0
- package/extensions/jev-advisor/consent.mjs +133 -0
- package/extensions/jev-advisor/gate.mjs +263 -0
- package/extensions/jev-advisor/index.ts +999 -0
- package/extensions/jev-advisor/key-source.mjs +252 -0
- package/extensions/jev-advisor/layer.mjs +169 -0
- package/extensions/jev-advisor/ledger.mjs +138 -0
- package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
- package/extensions/jev-advisor/questions/compaction.mjs +153 -0
- package/extensions/jev-advisor/questions/gap.mjs +140 -0
- package/extensions/jev-advisor/questions/guard.mjs +168 -0
- package/extensions/jev-advisor/questions/progress.mjs +195 -0
- package/extensions/jev-advisor/questions/retention.mjs +188 -0
- package/extensions/jev-advisor/questions/sources.mjs +91 -0
- package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
- package/extensions/jev-advisor/risk.mjs +442 -0
- package/extensions/jev-advisor/sanitize.mjs +0 -0
- package/extensions/jev-advisor/usage.mjs +92 -0
- package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
- package/extensions/tool-wishlist/index.ts +11 -0
- package/extensions/workflow-controls/capabilities.mjs +26 -0
- package/extensions/workflow-controls/index.ts +2 -2
- package/package.json +1 -1
- package/scripts/packages.mjs +56 -0
- package/scripts/specpi.mjs +73 -4
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
// Jev is the first thing in SpecPi that talks to a third party, so its switch is the first thing
|
|
2
|
+
// every other module in this directory consults. Master off means no key value read, no consent
|
|
3
|
+
// read, no network call and no prompt injection: the harness behaves exactly as it did before the
|
|
4
|
+
// extension existed. `/jev status` still reports whether a key exists while the layer is off, by
|
|
5
|
+
// name and never by value, because "how do I configure this" is a question asked before enabling
|
|
6
|
+
// anything -- see key-source.mjs.
|
|
7
|
+
//
|
|
8
|
+
// The file is SpecPi's own, hardened the same way as web-access and capability-policy: atomic
|
|
9
|
+
// write, mode 0600, symlinks refused, and a missing or unreadable file read as off.
|
|
10
|
+
|
|
11
|
+
import fs from "node:fs";
|
|
12
|
+
import os from "node:os";
|
|
13
|
+
import path from "node:path";
|
|
14
|
+
import { randomUUID } from "node:crypto";
|
|
15
|
+
|
|
16
|
+
/** Systems that may run inside a session. Offline scripts are not gated here. */
|
|
17
|
+
export const SYSTEM_NAMES = Object.freeze([
|
|
18
|
+
"retention",
|
|
19
|
+
"compaction",
|
|
20
|
+
"gap",
|
|
21
|
+
"sources",
|
|
22
|
+
"progress",
|
|
23
|
+
"untrusted",
|
|
24
|
+
"capability",
|
|
25
|
+
// The command guard, native since schema 3. It used to be a separate pinned package with its own
|
|
26
|
+
// global configuration file, which is why it used to carry its own switch here: a second switch,
|
|
27
|
+
// outside the master, with its own startup preference and its own persistence rules. Those rules
|
|
28
|
+
// disagreed with the layer's often enough to be their own source of defects -- a preference
|
|
29
|
+
// erased by a command that had decided nothing about the guard, a state written by one command
|
|
30
|
+
// and reverted by the next session. As a system it is gated, budgeted, reported and toggled by
|
|
31
|
+
// exactly the same code as the other seven.
|
|
32
|
+
"guard",
|
|
33
|
+
]);
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* What a confident stuck verdict is allowed to do. `notify` tells the person and cannot be wrong in
|
|
37
|
+
* a way that costs anything; `message` appends a fixed line the model reads before its next
|
|
38
|
+
* request, which changes behaviour.
|
|
39
|
+
*
|
|
40
|
+
* It ships on `notify`. Not because the gate cannot tell the cases apart -- it demonstrably can: on
|
|
41
|
+
* the recorded fixtures a session repeating one failing call scores 0.89 for stuck with the mode at
|
|
42
|
+
* 0.99 confidence, and a session working steadily scores 0.30 and reports "unknown" below the gate.
|
|
43
|
+
* The missing number is the false-positive rate on real sessions, and the two things that bear on it
|
|
44
|
+
* point the other way: running the same taxonomy over the 24 recorded failures left 14 of them
|
|
45
|
+
* ungated, and a wrong nudge costs a turn, which is the exact quantity this system exists to save.
|
|
46
|
+
*
|
|
47
|
+
* So the condition for changing this default is a measurement, not an opinion, and the eval suite
|
|
48
|
+
* is where it comes from: `--harness=specpi-jev` sets `message` and discloses it, because a
|
|
49
|
+
* notification in a headless run reaches nobody and would measure the cost of the system with none
|
|
50
|
+
* of its effect.
|
|
51
|
+
*/
|
|
52
|
+
export const NUDGE_MODES = Object.freeze(["notify", "message"]);
|
|
53
|
+
|
|
54
|
+
const MAX_SETTINGS_BYTES = 4096;
|
|
55
|
+
const MAX_CALL_BUDGET = 256;
|
|
56
|
+
const MAX_TOTAL_BUDGET = 512;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* One shared budget could not survive a turn-level system. A system that fires once per turn would
|
|
60
|
+
* reach a shared ceiling of 8 inside the first few turns and starve retention for the rest of the
|
|
61
|
+
* session, and which one won would be decided by event ordering rather than by anyone's policy.
|
|
62
|
+
*
|
|
63
|
+
* So the ceiling is two-level: each system gets its own, and the total is a real constraint because
|
|
64
|
+
* it is deliberately less than their sum. Running out of one system's budget stops that system and
|
|
65
|
+
* nothing else.
|
|
66
|
+
*
|
|
67
|
+
* The per-system numbers follow how often each one can fire: retention on every large read-only
|
|
68
|
+
* result, compaction once or twice in a long session, gap per report, sources per delegation batch.
|
|
69
|
+
*/
|
|
70
|
+
export const DEFAULT_BUDGETS = Object.freeze({
|
|
71
|
+
// A backstop, not a working limit, and the number says which. Measured, a full tier-3 task -- a
|
|
72
|
+
// 120-step repair chain over about 25 model requests -- spends 4 to 7 calls, and the busiest
|
|
73
|
+
// attempt ever recorded spent 12. A session would have to run for days before 512 bound
|
|
74
|
+
// anything a person was actually doing, which is the point: the ceiling should only ever be hit
|
|
75
|
+
// by a loop, and hitting it should therefore be information rather than an inconvenience.
|
|
76
|
+
//
|
|
77
|
+
// The earlier 120 was sized against eval attempts, which is the wrong reference. An attempt
|
|
78
|
+
// runs for two minutes; an interactive session runs for a day, and a turn-level system at one
|
|
79
|
+
// call every four turns reaches 120 somewhere in the afternoon and then goes quiet without
|
|
80
|
+
// having found anything wrong. A ceiling that a normal long session reaches is not protecting
|
|
81
|
+
// anyone, it is just failing later than it looks.
|
|
82
|
+
//
|
|
83
|
+
// Cost is not what these are for. A call is about $0.00003, so the whole total is about a cent
|
|
84
|
+
// and a half. They bound two things that do not get cheaper with scale: how much digest leaves
|
|
85
|
+
// the machine for a third party, at up to 1 KB a call, and how much awaited latency a runaway
|
|
86
|
+
// loop can add before something stops it. Half a megabyte of digest and an announced stop is
|
|
87
|
+
// the shape of the trade.
|
|
88
|
+
total: 512,
|
|
89
|
+
retention: 208,
|
|
90
|
+
compaction: 12,
|
|
91
|
+
gap: 48,
|
|
92
|
+
sources: 32,
|
|
93
|
+
// Turn-level, but gated behind local signals and a four-turn cooldown, so it only spends on
|
|
94
|
+
// sessions that already look wrong. The ceiling is what stops a genuinely thrashing session
|
|
95
|
+
// from spending the total on being told it is thrashing.
|
|
96
|
+
progress: 176,
|
|
97
|
+
// Usually free: when retention is on, system 7's question rides the call retention was already
|
|
98
|
+
// making against the same state. This ceiling only binds when retention is off, or when the
|
|
99
|
+
// fetched result is too small for retention to be interested in it.
|
|
100
|
+
untrusted: 104,
|
|
101
|
+
// Once per session by construction, and only when local signals already suggest it. Two rather
|
|
102
|
+
// than one so a retried first turn is not silently un-served.
|
|
103
|
+
capability: 2,
|
|
104
|
+
// Per gated tool call that local rules could not settle, so its frequency is retention's rather
|
|
105
|
+
// than compaction's -- and like retention, most calls never reach it: read-only commands and
|
|
106
|
+
// ordinary project writes are answered locally for nothing. Running out means the guard defers
|
|
107
|
+
// to the permission system for the rest of the session, which is what it does for every other
|
|
108
|
+
// kind of unavailability.
|
|
109
|
+
guard: 208,
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
/** 0 is a real budget meaning no calls. Switching a system off is what `systems[name] = false` is for. */
|
|
113
|
+
export const MAX_BUDGETS = Object.freeze({ system: MAX_CALL_BUDGET, total: MAX_TOTAL_BUDGET });
|
|
114
|
+
|
|
115
|
+
export function agentDirectory() {
|
|
116
|
+
const configured = process.env.PI_CODING_AGENT_DIR;
|
|
117
|
+
|
|
118
|
+
return path.resolve(configured && configured.length > 0 ? configured : path.join(os.homedir(), ".pi", "agent"));
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export function jevDirectory() {
|
|
122
|
+
return path.join(agentDirectory(), "specpi", "jev");
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function settingsFile() {
|
|
126
|
+
return path.join(jevDirectory(), "settings.json");
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/** Refuses links and irregular files so the preference cannot redirect a write. */
|
|
130
|
+
export function regularFile(file, label) {
|
|
131
|
+
const stat = fs.lstatSync(file, { throwIfNoEntry: false });
|
|
132
|
+
if (!stat) {
|
|
133
|
+
return false;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1 || stat.size > MAX_SETTINGS_BYTES) {
|
|
137
|
+
throw new Error(`Unsupported ${label} file`);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return true;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Every unknown shape collapses to the same all-off default rather than a partial enable. */
|
|
144
|
+
export function defaultSettings() {
|
|
145
|
+
return {
|
|
146
|
+
schema: 3,
|
|
147
|
+
master: false,
|
|
148
|
+
startup: false,
|
|
149
|
+
systems: Object.fromEntries(SYSTEM_NAMES.map((name) => [name, false])),
|
|
150
|
+
budgets: { ...DEFAULT_BUDGETS },
|
|
151
|
+
progressNudge: "notify",
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
function clamp(value, fallback, ceiling) {
|
|
156
|
+
return Number.isInteger(value) ? Math.min(Math.max(value, 0), ceiling) : fallback;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function normalizeBudgets(raw) {
|
|
160
|
+
const budgets = { total: clamp(raw?.total, DEFAULT_BUDGETS.total, MAX_TOTAL_BUDGET) };
|
|
161
|
+
for (const name of SYSTEM_NAMES) {
|
|
162
|
+
budgets[name] = clamp(raw?.[name], DEFAULT_BUDGETS[name] ?? 0, MAX_CALL_BUDGET);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
return budgets;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Schema 1 carried one `callBudgetPerSession`. Reading it as an unknown shape would switch the
|
|
170
|
+
* layer off for anyone who had turned it on, which is a reset the user never asked for -- "unknown
|
|
171
|
+
* shapes collapse to all-off" is a rule for corrupt input, not for our own previous version. The
|
|
172
|
+
* one number becomes the total, and each system gets the smaller of its default and that total, so
|
|
173
|
+
* a user who set a deliberately tight ceiling keeps it.
|
|
174
|
+
*/
|
|
175
|
+
function migrate(raw) {
|
|
176
|
+
// Schema 1 read 0 as "no ceiling"; schema 2 reads it as "no calls", because a per-system 0 that
|
|
177
|
+
// silently meant unlimited is the wrong way for a budget to fail. Carrying the old meaning
|
|
178
|
+
// forward here is what stops the bump from inverting a user's intent.
|
|
179
|
+
const stored = raw?.callBudgetPerSession === 0 ? MAX_TOTAL_BUDGET : raw?.callBudgetPerSession;
|
|
180
|
+
const total = clamp(stored, DEFAULT_BUDGETS.total, MAX_TOTAL_BUDGET);
|
|
181
|
+
const budgets = { total };
|
|
182
|
+
for (const name of SYSTEM_NAMES) {
|
|
183
|
+
budgets[name] = Math.min(DEFAULT_BUDGETS[name] ?? 0, total);
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
return { ...raw, schema: 2, budgets, callBudgetPerSession: undefined };
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Schema 2 carried the command guard as a separate `guard: { enabled, startup }` pair, because it
|
|
191
|
+
* was a separate package with its own global configuration file. Schema 3 makes it the eighth
|
|
192
|
+
* system, so the stored preference becomes `systems.guard`.
|
|
193
|
+
*
|
|
194
|
+
* `guard.startup` is what migrates, not `guard.enabled`: the former is what the user chose for new
|
|
195
|
+
* sessions, and the latter was a session flag that happened to be written to disk. A file where the
|
|
196
|
+
* guard was wanted at startup but the layer itself was not produces a system that is on inside a
|
|
197
|
+
* layer that is off, which is inactive -- the guard used to sit outside the master switch and now
|
|
198
|
+
* does not. That direction is deliberate: a gate quietly becoming inactive is recoverable in one
|
|
199
|
+
* command, and a gate quietly becoming active is how a session stops being able to run anything.
|
|
200
|
+
*/
|
|
201
|
+
function migrateToThree(raw) {
|
|
202
|
+
const systems = { ...(raw?.systems ?? {}), guard: raw?.guard?.startup === true };
|
|
203
|
+
|
|
204
|
+
const { guard: _guard, ...rest } = raw ?? {};
|
|
205
|
+
|
|
206
|
+
return { ...rest, schema: 3, systems };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
function normalize(raw) {
|
|
210
|
+
const one = raw?.schema === 1 ? migrate(raw) : raw;
|
|
211
|
+
const source = one?.schema === 2 ? migrateToThree(one) : one;
|
|
212
|
+
if (source?.schema !== 3) {
|
|
213
|
+
return defaultSettings();
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
const systems = Object.fromEntries(SYSTEM_NAMES.map((name) => [name, source.systems?.[name] === true]));
|
|
217
|
+
|
|
218
|
+
return {
|
|
219
|
+
schema: 3,
|
|
220
|
+
master: source.master === true,
|
|
221
|
+
startup: source.startup === true,
|
|
222
|
+
systems,
|
|
223
|
+
budgets: normalizeBudgets(source.budgets),
|
|
224
|
+
progressNudge: NUDGE_MODES.includes(source.progressNudge) ? source.progressNudge : "notify",
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
export function loadSettings() {
|
|
229
|
+
try {
|
|
230
|
+
const file = settingsFile();
|
|
231
|
+
if (!regularFile(file, "Jev settings")) {
|
|
232
|
+
return defaultSettings();
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
return normalize(JSON.parse(fs.readFileSync(file, "utf8")));
|
|
236
|
+
} catch {
|
|
237
|
+
return defaultSettings();
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/** Shared atomic write for every file this extension owns. */
|
|
242
|
+
export function writeFileAtomic(file, contents) {
|
|
243
|
+
const directory = path.dirname(file);
|
|
244
|
+
fs.mkdirSync(directory, { recursive: true, mode: 0o700 });
|
|
245
|
+
const temporary = path.join(directory, `.${path.basename(file)}.${randomUUID()}.tmp`);
|
|
246
|
+
try {
|
|
247
|
+
fs.writeFileSync(temporary, contents, { mode: 0o600, flag: "wx" });
|
|
248
|
+
fs.renameSync(temporary, file);
|
|
249
|
+
} finally {
|
|
250
|
+
fs.rmSync(temporary, { force: true });
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
export function saveSettings(settings) {
|
|
255
|
+
// A caller handing back a schema-1 shape is migrated rather than reset, so a round trip through
|
|
256
|
+
// an old reader cannot quietly disable the layer.
|
|
257
|
+
const next = normalize(settings?.schema === 1 || settings?.schema === 2 ? settings : { ...settings, schema: 3 });
|
|
258
|
+
const file = settingsFile();
|
|
259
|
+
if (fs.existsSync(file)) {
|
|
260
|
+
regularFile(file, "Jev settings");
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
writeFileAtomic(file, `${JSON.stringify(next, null, 4)}\n`);
|
|
264
|
+
|
|
265
|
+
return next;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
export function settingsPath() {
|
|
269
|
+
return settingsFile();
|
|
270
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
// A config flag records an intention. It does not record that a human was told what the intention
|
|
2
|
+
// costs. This file holds the separate, explicit grant: the first time any system would put session
|
|
3
|
+
// state on the wire, a dialog names the endpoint, the shape of the data and the byte budget, and
|
|
4
|
+
// nothing is sent until someone says yes.
|
|
5
|
+
//
|
|
6
|
+
// Mirrors capability-policy: SpecPi's own file, atomic, mode 0600, symlinks refused, missing or
|
|
7
|
+
// unreadable read as "ask". No interactive UI means no send, ever.
|
|
8
|
+
|
|
9
|
+
import fs from "node:fs";
|
|
10
|
+
import path from "node:path";
|
|
11
|
+
import { jevDirectory, regularFile, writeFileAtomic } from "./config.mjs";
|
|
12
|
+
import { baseUrl } from "./client.mjs";
|
|
13
|
+
import { MAX_STATE_BYTES } from "./sanitize.mjs";
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* The host the bytes will actually reach, not the service they are nominally for. This was a fixed
|
|
17
|
+
* "api.typesafe.ai" until the default backend became OpenRouter, at which point the dialog named a
|
|
18
|
+
* host the data no longer went to, which is the one thing a consent dialog may never do. It is
|
|
19
|
+
* derived now, so each of the three selectable destinations (OpenRouter, the direct TypeSafe API,
|
|
20
|
+
* and a TYPESAFE_BASE_URL override) names itself, and so the rule below is true rather than
|
|
21
|
+
* aspirational: a grant is keyed on this string, so repointing the client really does ask again.
|
|
22
|
+
*/
|
|
23
|
+
export function endpointLabel() {
|
|
24
|
+
const base = baseUrl();
|
|
25
|
+
try {
|
|
26
|
+
return new URL(base).host;
|
|
27
|
+
} catch {
|
|
28
|
+
// Unparseable means the fetch will fail anyway. Returning the raw value keeps the dialog
|
|
29
|
+
// honest and cannot collide with a host a real grant was given for.
|
|
30
|
+
return base;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function consentFile() {
|
|
35
|
+
return path.join(jevDirectory(), "consent.json");
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export function loadConsent() {
|
|
39
|
+
try {
|
|
40
|
+
const file = consentFile();
|
|
41
|
+
if (!regularFile(file, "Jev consent")) {
|
|
42
|
+
return undefined;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const stored = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
46
|
+
if (stored?.schema !== 1 || stored.granted !== true || typeof stored.endpoint !== "string") {
|
|
47
|
+
return undefined;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// A grant is for the endpoint it was given for. Repointing the client asks again.
|
|
51
|
+
return stored.endpoint === endpointLabel() ? stored : undefined;
|
|
52
|
+
} catch {
|
|
53
|
+
return undefined;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export function granted() {
|
|
58
|
+
return loadConsent() !== undefined;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function saveConsent() {
|
|
62
|
+
const file = consentFile();
|
|
63
|
+
if (fs.existsSync(file)) {
|
|
64
|
+
regularFile(file, "Jev consent");
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const stored = {
|
|
68
|
+
schema: 1,
|
|
69
|
+
granted: true,
|
|
70
|
+
endpoint: endpointLabel(),
|
|
71
|
+
maxStateBytes: MAX_STATE_BYTES,
|
|
72
|
+
grantedAt: new Date().toISOString(),
|
|
73
|
+
};
|
|
74
|
+
writeFileAtomic(file, `${JSON.stringify(stored, null, 4)}\n`);
|
|
75
|
+
|
|
76
|
+
return stored;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export function revokeConsent() {
|
|
80
|
+
try {
|
|
81
|
+
fs.rmSync(consentFile(), { force: true });
|
|
82
|
+
} catch {
|
|
83
|
+
// A consent file that cannot be removed still reads as granted; the master switch is the
|
|
84
|
+
// reliable stop, and /jev status reports both.
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export function consentPath() {
|
|
89
|
+
return consentFile();
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export const CONSENT_TITLE = "Allow SpecPi to send task summaries to Jev?";
|
|
93
|
+
|
|
94
|
+
export function consentBody(systemLabel) {
|
|
95
|
+
return [
|
|
96
|
+
`${systemLabel} wants to ask TypeSafe's Jev classifier a question about this session.`,
|
|
97
|
+
"",
|
|
98
|
+
`What is sent: a summary object of at most ${MAX_STATE_BYTES} bytes to ${endpointLabel()}, over HTTPS.`,
|
|
99
|
+
"It carries tool names, byte counts, relative paths and short descriptions.",
|
|
100
|
+
"It never carries file contents, command output, credentials, URLs or session history.",
|
|
101
|
+
"",
|
|
102
|
+
"Every call is recorded locally in transmissions.jsonl with a hash of exactly what was sent,",
|
|
103
|
+
"which you can read with /jev ledger. Turn this off at any time with /jev off.",
|
|
104
|
+
].join("\n");
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Resolve consent for a system, prompting once. Returns false without prompting when there is no
|
|
109
|
+
* interactive human: an advisor must never be the reason a headless run sends data.
|
|
110
|
+
*/
|
|
111
|
+
export async function ensureConsent(ctx, systemLabel) {
|
|
112
|
+
if (granted()) {
|
|
113
|
+
return true;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
if (!ctx?.hasUI || typeof ctx.ui?.confirm !== "function") {
|
|
117
|
+
return false;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const accepted = await ctx.ui.confirm(CONSENT_TITLE, consentBody(systemLabel));
|
|
121
|
+
if (!accepted) {
|
|
122
|
+
return false;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
try {
|
|
126
|
+
saveConsent();
|
|
127
|
+
} catch {
|
|
128
|
+
// An unwritable grant means asking again next time, which is the safe direction.
|
|
129
|
+
return true;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
return true;
|
|
133
|
+
}
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
// A probability is not a decision. This file is the only place that turns one into the other, so
|
|
2
|
+
// every system gates the same way and a threshold change is a one-line diff with one test to move.
|
|
3
|
+
//
|
|
4
|
+
// The three primitives do not gate alike. A Noul returns a bare probability with no confidence
|
|
5
|
+
// field, so a confidence test cannot be applied to it. A Choice can be confident and still be a
|
|
6
|
+
// coin flip between its top two options, so the margin matters as much as the confidence. A Score
|
|
7
|
+
// is only actionable when it sits clear of a level boundary rather than straddling one.
|
|
8
|
+
//
|
|
9
|
+
// MEASURED, not guessed. Every number here is now read off `evals/runs/jev-calibration.json`, which
|
|
10
|
+
// `node scripts/jev-calibrate.mjs` writes from 259 recorded eval attempts plus a fixed reachability
|
|
11
|
+
// pass of six synthetic cases run five times each. `tests/jev-calibration.test.mjs` pins each number
|
|
12
|
+
// to that file, so moving one takes new evidence rather than a new opinion.
|
|
13
|
+
//
|
|
14
|
+
// Two questions were asked of the evidence, because a threshold can fail in two different ways.
|
|
15
|
+
//
|
|
16
|
+
// 1. DOES THE CONFIDENCE FIELD SEPARATE RIGHT FROM WRONG? Measured against labels this repository
|
|
17
|
+
// already owns: predicting an attempt's pass (Noul), its task category (Choice) and its tier
|
|
18
|
+
// (Score) from behavioural metadata alone, with the labels withheld from the state.
|
|
19
|
+
//
|
|
20
|
+
// Score: yes, weakly. At scoreConfidence 0.60 and boundary 0.30 the answer is exactly right
|
|
21
|
+
// about two thirds of the time and within one level about 96%, on roughly a fifth of
|
|
22
|
+
// answers, against a 43.2% majority class. The exact figures are in CALIBRATION below,
|
|
23
|
+
// which a test compares against the artifact; they moved from 68.4% to 64.7% between two
|
|
24
|
+
// runs over the same 259 attempts, so any single decimal here is a sample, not a
|
|
25
|
+
// constant, and the honest summary is "a lift of about 1.5".
|
|
26
|
+
// Noul: no. Precision tracks the 90.7% base rate at every threshold (lift 1.01-1.02), and no
|
|
27
|
+
// answer to that question ever exceeded 0.80.
|
|
28
|
+
// Choice: no. About 30% accuracy against a 29.7% majority class, and accuracy falls as
|
|
29
|
+
// confidence rises. The margin changes nothing, because the top-two gap is almost always
|
|
30
|
+
// wide.
|
|
31
|
+
//
|
|
32
|
+
// None of the systems' pre-registered precision targets (0.75 to 0.95) is met anywhere on any of
|
|
33
|
+
// those curves, and the artifact records UNMET rather than a number chosen to fill the gap. The
|
|
34
|
+
// Score point below is the best available operating point, not a met target. That is a real
|
|
35
|
+
// limit on what this layer can claim, and it is published rather than smoothed over.
|
|
36
|
+
//
|
|
37
|
+
// It is also a fair reading that the proxy questions are much harder than the production ones:
|
|
38
|
+
// the eval state is a row of counters, while a production state carries the material being
|
|
39
|
+
// judged. Question 2 is what tests that, and the answer is yes.
|
|
40
|
+
//
|
|
41
|
+
// 2. CAN THE GATE EVER FIRE? This is the one that found a shipped defect. Against the fixture cases
|
|
42
|
+
// the real question sets separate cleanly -- a planted secret scores 0.96 and a clean report
|
|
43
|
+
// 0.04; a page carrying an injected instruction scores 0.97 and an ordinary one 0.04; the one
|
|
44
|
+
// relevant file among noise scores 1.99 while the other two score 0.01 -- but the Score gate
|
|
45
|
+
// shipped at confidence 0.80 with boundary 0.35, and on a maximally obvious "spent" result (a
|
|
46
|
+
// listing of vendor icons during a changelog edit) Jev answered 0.10-0.18 with confidence
|
|
47
|
+
// 0.73-0.85 across ten runs, most of them under 0.80. Boundary 0.35 demands the value sit within
|
|
48
|
+
// 0.15 of a level, which 0.16 and 0.17 miss. So retention could gate through to "keep this
|
|
49
|
+
// result" and essentially never to "this result is spent": the only branch that does anything
|
|
50
|
+
// was unreachable, and running the layer could never have revealed it, because a system that
|
|
51
|
+
// never fires looks exactly like a system whose advice was always to do nothing.
|
|
52
|
+
//
|
|
53
|
+
// compaction's `unresolved_thread` had the same problem at high 0.85: the clearest open
|
|
54
|
+
// investigation the fixture can express scores 0.63-0.65. It is lowered to 0.60, which is
|
|
55
|
+
// defensible only because of what that branch does -- add one sentence to a summariser prompt
|
|
56
|
+
// that is being rebuilt from scratch anyway. It is the cheapest action in the layer, so it can
|
|
57
|
+
// afford the loosest gate. gap keeps 0.85 because its Noul reaches 0.96 on the case that matters
|
|
58
|
+
// and because a firing there blocks a write.
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* The figures the comment above cites, in a form a test can check against the artifact. A citation
|
|
62
|
+
* that drifts from its source is worse than no citation, because it reads like evidence. If these
|
|
63
|
+
* stop matching `evals/runs/jev-calibration.json`, `tests/jev-calibration.test.mjs` fails and
|
|
64
|
+
* whoever re-ran the calibration has to update the prose too.
|
|
65
|
+
*/
|
|
66
|
+
export const CALIBRATION = Object.freeze({
|
|
67
|
+
artifact: "evals/runs/jev-calibration.json",
|
|
68
|
+
attempts: 259,
|
|
69
|
+
passBaseRate: 0.907,
|
|
70
|
+
tierBaseRate: 0.432,
|
|
71
|
+
categoryBaseRate: 0.297,
|
|
72
|
+
// At the shipped scoreConfidence 0.60 / boundary 0.30. Tolerance is deliberate: see above.
|
|
73
|
+
scoreExact: 0.647,
|
|
74
|
+
scoreWithinOne: 0.961,
|
|
75
|
+
scoreCoverage: 0.197,
|
|
76
|
+
tolerance: 0.05,
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
export const THRESHOLDS = Object.freeze({
|
|
80
|
+
// One Score operating point, applied to every system, because one proxy question produced one
|
|
81
|
+
// curve. Four different per-system numbers would be four claims from a single measurement. The
|
|
82
|
+
// per-system asymmetry lives where it belongs instead: retention needs two independent answers
|
|
83
|
+
// to agree before it shortens anything, and sources only ever reorders.
|
|
84
|
+
retention: Object.freeze({
|
|
85
|
+
scoreConfidence: 0.6,
|
|
86
|
+
boundary: 0.3,
|
|
87
|
+
choiceConfidence: 0.75,
|
|
88
|
+
margin: 0.3,
|
|
89
|
+
// Unused by retention, which reads only the low side; kept so every system has a full set.
|
|
90
|
+
high: 0.9,
|
|
91
|
+
// Measured 0.06-0.07 on the spent case, so this clears with room.
|
|
92
|
+
low: 0.1,
|
|
93
|
+
}),
|
|
94
|
+
compaction: Object.freeze({
|
|
95
|
+
scoreConfidence: 0.6,
|
|
96
|
+
boundary: 0.3,
|
|
97
|
+
choiceConfidence: 0.75,
|
|
98
|
+
margin: 0.25,
|
|
99
|
+
// Lowered from 0.85: the clearest open investigation scores 0.63-0.65, and the action is
|
|
100
|
+
// one sentence added to a prompt that is being rebuilt regardless.
|
|
101
|
+
high: 0.6,
|
|
102
|
+
low: 0.15,
|
|
103
|
+
}),
|
|
104
|
+
gap: Object.freeze({
|
|
105
|
+
scoreConfidence: 0.6,
|
|
106
|
+
boundary: 0.3,
|
|
107
|
+
choiceConfidence: 0.75,
|
|
108
|
+
margin: 0.2,
|
|
109
|
+
// Measured 0.96 on a report carrying a machine-specific path and 0.04 on a clean one.
|
|
110
|
+
high: 0.85,
|
|
111
|
+
low: 0.15,
|
|
112
|
+
}),
|
|
113
|
+
progress: Object.freeze({
|
|
114
|
+
scoreConfidence: 0.6,
|
|
115
|
+
boundary: 0.3,
|
|
116
|
+
// The strictest Choice gate in the layer, because it is the only system whose action can
|
|
117
|
+
// change what the model does next. Running the same taxonomy over the 24 recorded failures
|
|
118
|
+
// left 14 of them below this bar, which is the intended behaviour: silence is the correct
|
|
119
|
+
// answer to a session whose trouble is not yet legible.
|
|
120
|
+
choiceConfidence: 0.8,
|
|
121
|
+
margin: 0.25,
|
|
122
|
+
high: 0.85,
|
|
123
|
+
low: 0.15,
|
|
124
|
+
}),
|
|
125
|
+
capability: Object.freeze({
|
|
126
|
+
scoreConfidence: 0.6,
|
|
127
|
+
boundary: 0.3,
|
|
128
|
+
choiceConfidence: 0.75,
|
|
129
|
+
margin: 0.2,
|
|
130
|
+
// Asymmetric on purpose. A false positive costs the group's schema on every request for
|
|
131
|
+
// the rest of the session -- a turn-1 arming measured 16% more than never arming -- plus a
|
|
132
|
+
// confirmation the human did not need. A false negative costs nothing: it leaves today's
|
|
133
|
+
// behaviour exactly as it is, and `request_capability` is still there for the moment the
|
|
134
|
+
// need becomes real.
|
|
135
|
+
//
|
|
136
|
+
// This shipped at 0.90 for exactly as long as it took to measure it, which is the same
|
|
137
|
+
// defect described above, in code written the same day, caught by the same check. On a
|
|
138
|
+
// request that unambiguously needs a browser -- open the pricing page at 375px and fix what
|
|
139
|
+
// overflows -- `needs_browser` answers 0.86-0.87, five times out of five, while the same
|
|
140
|
+
// question on a rename answers 0.09-0.10. 0.90 sits inside the yes cluster and rejects all
|
|
141
|
+
// of it; 0.85 sits below it with a margin of 0.76 to the nearest no. No Noul anywhere in
|
|
142
|
+
// the fixture set has ever exceeded 0.97, so "higher is safer" stops being true well before
|
|
143
|
+
// it stops being tempting.
|
|
144
|
+
high: 0.85,
|
|
145
|
+
low: 0.1,
|
|
146
|
+
}),
|
|
147
|
+
untrusted: Object.freeze({
|
|
148
|
+
scoreConfidence: 0.6,
|
|
149
|
+
boundary: 0.3,
|
|
150
|
+
choiceConfidence: 0.75,
|
|
151
|
+
margin: 0.2,
|
|
152
|
+
// A banner is cheap and a missed injection is not, so this is the one place where the
|
|
153
|
+
// asymmetry runs the other way from gap's. It is still 0.85 rather than lower, because a
|
|
154
|
+
// banner on ordinary prose is exactly the false positive that teaches a model to stop
|
|
155
|
+
// reading the channel -- the objection this system had to answer before it could exist.
|
|
156
|
+
high: 0.85,
|
|
157
|
+
low: 0.15,
|
|
158
|
+
}),
|
|
159
|
+
guard: Object.freeze({
|
|
160
|
+
// The same measured Score operating point as every other system, written out rather than
|
|
161
|
+
// inherited by falling through `thresholdsFor`'s default. The guard asked under the name
|
|
162
|
+
// "gap" for exactly as long as it took to notice that adding a `guard` entry would then have
|
|
163
|
+
// changed nothing -- a silent no-op on the one system whose action takes a tool call away.
|
|
164
|
+
scoreConfidence: 0.6,
|
|
165
|
+
boundary: 0.3,
|
|
166
|
+
choiceConfidence: 0.75,
|
|
167
|
+
margin: 0.2,
|
|
168
|
+
// Unused: this system reads only the Score side. Kept so every system has a full set, and
|
|
169
|
+
// so the calibration tests cover this entry like any other.
|
|
170
|
+
high: 0.85,
|
|
171
|
+
low: 0.15,
|
|
172
|
+
}),
|
|
173
|
+
sources: Object.freeze({
|
|
174
|
+
scoreConfidence: 0.6,
|
|
175
|
+
boundary: 0.3,
|
|
176
|
+
choiceConfidence: 0.7,
|
|
177
|
+
margin: 0.15,
|
|
178
|
+
// Not reached by any fixture case: `worth_delegating` scored 0.20-0.21 on a question that
|
|
179
|
+
// reads as self-contained to a human. Left at 0.85 rather than tuned down, because the only
|
|
180
|
+
// thing that reads it warns and never blocks, and a threshold moved to make a fixture pass
|
|
181
|
+
// is a threshold set by the fixture.
|
|
182
|
+
high: 0.85,
|
|
183
|
+
low: 0.15,
|
|
184
|
+
}),
|
|
185
|
+
});
|
|
186
|
+
|
|
187
|
+
export function thresholdsFor(system) {
|
|
188
|
+
return THRESHOLDS[system] ?? THRESHOLDS.gap;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** True when the Noul is confidently yes. */
|
|
192
|
+
export function nounTrue(answer, system) {
|
|
193
|
+
const limits = thresholdsFor(system);
|
|
194
|
+
|
|
195
|
+
return answer?.kind === "noul" && answer.value >= limits.high;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/** True when the Noul is confidently no. Not the negation of nounTrue: the middle band is silence. */
|
|
199
|
+
export function nounFalse(answer, system) {
|
|
200
|
+
const limits = thresholdsFor(system);
|
|
201
|
+
|
|
202
|
+
return answer?.kind === "noul" && answer.value <= limits.low;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
function topTwo(probabilities) {
|
|
206
|
+
const values = Object.values(probabilities ?? {})
|
|
207
|
+
.filter((value) => typeof value === "number")
|
|
208
|
+
.sort((a, b) => b - a);
|
|
209
|
+
|
|
210
|
+
return { first: values[0] ?? 0, second: values[1] ?? 0 };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* A Choice is actionable when it is both confident and clearly separated from its runner-up.
|
|
215
|
+
* Without a distribution the margin cannot be checked, so the answer is treated as ungated.
|
|
216
|
+
*
|
|
217
|
+
* The reachability pass confirms all 40 Choice answers from this backend carried a distribution, so
|
|
218
|
+
* the margin is a live test rather than a branch that silently never runs.
|
|
219
|
+
*/
|
|
220
|
+
export function choiceValue(answer, system) {
|
|
221
|
+
if (answer?.kind !== "choice" || typeof answer.confidence !== "number") {
|
|
222
|
+
return undefined;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const limits = thresholdsFor(system);
|
|
226
|
+
if (answer.confidence < limits.choiceConfidence) {
|
|
227
|
+
return undefined;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
const { first, second } = topTwo(answer.probabilities);
|
|
231
|
+
if (answer.probabilities && first - second < limits.margin) {
|
|
232
|
+
return undefined;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
return answer.value;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* A Score is actionable when it is confident and sits clear of the nearest level boundary. The
|
|
240
|
+
* returned level is the rounded band; callers compare against their own rubric.
|
|
241
|
+
*/
|
|
242
|
+
export function scoreLevel(answer, system) {
|
|
243
|
+
if (answer?.kind !== "score" || typeof answer.confidence !== "number") {
|
|
244
|
+
return undefined;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
const limits = thresholdsFor(system);
|
|
248
|
+
if (answer.confidence < limits.scoreConfidence) {
|
|
249
|
+
return undefined;
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
const level = Math.round(answer.value);
|
|
253
|
+
if (Math.abs(answer.value - level) > 0.5 - limits.boundary) {
|
|
254
|
+
return undefined;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
return level;
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/** Raw probability for logging and calibration, with no gate applied. */
|
|
261
|
+
export function rawValue(answer) {
|
|
262
|
+
return answer?.value;
|
|
263
|
+
}
|