specpi 0.26.0 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/README.md +37 -3
  3. package/SECURITY_MODEL.md +44 -4
  4. package/THIRD_PARTY.md +9 -1
  5. package/extensions/jev-advisor/broker.mjs +277 -0
  6. package/extensions/jev-advisor/client.mjs +172 -0
  7. package/extensions/jev-advisor/config.mjs +270 -0
  8. package/extensions/jev-advisor/consent.mjs +133 -0
  9. package/extensions/jev-advisor/gate.mjs +263 -0
  10. package/extensions/jev-advisor/index.ts +999 -0
  11. package/extensions/jev-advisor/key-source.mjs +252 -0
  12. package/extensions/jev-advisor/layer.mjs +169 -0
  13. package/extensions/jev-advisor/ledger.mjs +138 -0
  14. package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
  15. package/extensions/jev-advisor/questions/compaction.mjs +153 -0
  16. package/extensions/jev-advisor/questions/gap.mjs +140 -0
  17. package/extensions/jev-advisor/questions/guard.mjs +168 -0
  18. package/extensions/jev-advisor/questions/progress.mjs +195 -0
  19. package/extensions/jev-advisor/questions/retention.mjs +188 -0
  20. package/extensions/jev-advisor/questions/sources.mjs +91 -0
  21. package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
  22. package/extensions/jev-advisor/risk.mjs +442 -0
  23. package/extensions/jev-advisor/sanitize.mjs +0 -0
  24. package/extensions/jev-advisor/usage.mjs +92 -0
  25. package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
  26. package/extensions/tool-wishlist/index.ts +11 -0
  27. package/extensions/workflow-controls/capabilities.mjs +26 -0
  28. package/extensions/workflow-controls/index.ts +2 -2
  29. package/package.json +1 -1
  30. package/scripts/packages.mjs +56 -0
  31. package/scripts/specpi.mjs +73 -4
@@ -0,0 +1,270 @@
1
+ // Jev is the first thing in SpecPi that talks to a third party, so its switch is the first thing
2
+ // every other module in this directory consults. Master off means no key value read, no consent
3
+ // read, no network call and no prompt injection: the harness behaves exactly as it did before the
4
+ // extension existed. `/jev status` still reports whether a key exists while the layer is off, by
5
+ // name and never by value, because "how do I configure this" is a question asked before enabling
6
+ // anything -- see key-source.mjs.
7
+ //
8
+ // The file is SpecPi's own, hardened the same way as web-access and capability-policy: atomic
9
+ // write, mode 0600, symlinks refused, and a missing or unreadable file read as off.
10
+
11
+ import fs from "node:fs";
12
+ import os from "node:os";
13
+ import path from "node:path";
14
+ import { randomUUID } from "node:crypto";
15
+
16
+ /** Systems that may run inside a session. Offline scripts are not gated here. */
17
+ export const SYSTEM_NAMES = Object.freeze([
18
+ "retention",
19
+ "compaction",
20
+ "gap",
21
+ "sources",
22
+ "progress",
23
+ "untrusted",
24
+ "capability",
25
+ // The command guard, native since schema 3. It used to be a separate pinned package with its own
26
+ // global configuration file, which is why it used to carry its own switch here: a second switch,
27
+ // outside the master, with its own startup preference and its own persistence rules. Those rules
28
+ // disagreed with the layer's often enough to be their own source of defects -- a preference
29
+ // erased by a command that had decided nothing about the guard, a state written by one command
30
+ // and reverted by the next session. As a system it is gated, budgeted, reported and toggled by
31
+ // exactly the same code as the other seven.
32
+ "guard",
33
+ ]);
34
+
35
+ /**
36
+ * What a confident stuck verdict is allowed to do. `notify` tells the person and cannot be wrong in
37
+ * a way that costs anything; `message` appends a fixed line the model reads before its next
38
+ * request, which changes behaviour.
39
+ *
40
+ * It ships on `notify`. Not because the gate cannot tell the cases apart -- it demonstrably can: on
41
+ * the recorded fixtures a session repeating one failing call scores 0.89 for stuck with the mode at
42
+ * 0.99 confidence, and a session working steadily scores 0.30 and reports "unknown" below the gate.
43
+ * The missing number is the false-positive rate on real sessions, and the two things that bear on it
44
+ * point the other way: running the same taxonomy over the 24 recorded failures left 14 of them
45
+ * ungated, and a wrong nudge costs a turn, which is the exact quantity this system exists to save.
46
+ *
47
+ * So the condition for changing this default is a measurement, not an opinion, and the eval suite
48
+ * is where it comes from: `--harness=specpi-jev` sets `message` and discloses it, because a
49
+ * notification in a headless run reaches nobody and would measure the cost of the system with none
50
+ * of its effect.
51
+ */
52
+ export const NUDGE_MODES = Object.freeze(["notify", "message"]);
53
+
54
+ const MAX_SETTINGS_BYTES = 4096;
55
+ const MAX_CALL_BUDGET = 256;
56
+ const MAX_TOTAL_BUDGET = 512;
57
+
58
+ /**
59
+ * One shared budget could not survive a turn-level system. A system that fires once per turn would
60
+ * reach a shared ceiling of 8 inside the first few turns and starve retention for the rest of the
61
+ * session, and which one won would be decided by event ordering rather than by anyone's policy.
62
+ *
63
+ * So the ceiling is two-level: each system gets its own, and the total is a real constraint because
64
+ * it is deliberately less than their sum. Running out of one system's budget stops that system and
65
+ * nothing else.
66
+ *
67
+ * The per-system numbers follow how often each one can fire: retention on every large read-only
68
+ * result, compaction once or twice in a long session, gap per report, sources per delegation batch.
69
+ */
70
+ export const DEFAULT_BUDGETS = Object.freeze({
71
+ // A backstop, not a working limit, and the number says which. Measured, a full tier-3 task -- a
72
+ // 120-step repair chain over about 25 model requests -- spends 4 to 7 calls, and the busiest
73
+ // attempt ever recorded spent 12. A session would have to run for days before 512 bound
74
+ // anything a person was actually doing, which is the point: the ceiling should only ever be hit
75
+ // by a loop, and hitting it should therefore be information rather than an inconvenience.
76
+ //
77
+ // The earlier 120 was sized against eval attempts, which is the wrong reference. An attempt
78
+ // runs for two minutes; an interactive session runs for a day, and a turn-level system at one
79
+ // call every four turns reaches 120 somewhere in the afternoon and then goes quiet without
80
+ // having found anything wrong. A ceiling that a normal long session reaches is not protecting
81
+ // anyone, it is just failing later than it looks.
82
+ //
83
+ // Cost is not what these are for. A call is about $0.00003, so the whole total is about a cent
84
+ // and a half. They bound two things that do not get cheaper with scale: how much digest leaves
85
+ // the machine for a third party, at up to 1 KB a call, and how much awaited latency a runaway
86
+ // loop can add before something stops it. Half a megabyte of digest and an announced stop is
87
+ // the shape of the trade.
88
+ total: 512,
89
+ retention: 208,
90
+ compaction: 12,
91
+ gap: 48,
92
+ sources: 32,
93
+ // Turn-level, but gated behind local signals and a four-turn cooldown, so it only spends on
94
+ // sessions that already look wrong. The ceiling is what stops a genuinely thrashing session
95
+ // from spending the total on being told it is thrashing.
96
+ progress: 176,
97
+ // Usually free: when retention is on, system 7's question rides the call retention was already
98
+ // making against the same state. This ceiling only binds when retention is off, or when the
99
+ // fetched result is too small for retention to be interested in it.
100
+ untrusted: 104,
101
+ // Once per session by construction, and only when local signals already suggest it. Two rather
102
+ // than one so a retried first turn is not silently un-served.
103
+ capability: 2,
104
+ // Per gated tool call that local rules could not settle, so its frequency is retention's rather
105
+ // than compaction's -- and like retention, most calls never reach it: read-only commands and
106
+ // ordinary project writes are answered locally for nothing. Running out means the guard defers
107
+ // to the permission system for the rest of the session, which is what it does for every other
108
+ // kind of unavailability.
109
+ guard: 208,
110
+ });
111
+
112
+ /** 0 is a real budget meaning no calls. Switching a system off is what `systems[name] = false` is for. */
113
+ export const MAX_BUDGETS = Object.freeze({ system: MAX_CALL_BUDGET, total: MAX_TOTAL_BUDGET });
114
+
115
+ export function agentDirectory() {
116
+ const configured = process.env.PI_CODING_AGENT_DIR;
117
+
118
+ return path.resolve(configured && configured.length > 0 ? configured : path.join(os.homedir(), ".pi", "agent"));
119
+ }
120
+
121
+ export function jevDirectory() {
122
+ return path.join(agentDirectory(), "specpi", "jev");
123
+ }
124
+
125
+ function settingsFile() {
126
+ return path.join(jevDirectory(), "settings.json");
127
+ }
128
+
129
+ /** Refuses links and irregular files so the preference cannot redirect a write. */
130
+ export function regularFile(file, label) {
131
+ const stat = fs.lstatSync(file, { throwIfNoEntry: false });
132
+ if (!stat) {
133
+ return false;
134
+ }
135
+
136
+ if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1 || stat.size > MAX_SETTINGS_BYTES) {
137
+ throw new Error(`Unsupported ${label} file`);
138
+ }
139
+
140
+ return true;
141
+ }
142
+
143
+ /** Every unknown shape collapses to the same all-off default rather than a partial enable. */
144
+ export function defaultSettings() {
145
+ return {
146
+ schema: 3,
147
+ master: false,
148
+ startup: false,
149
+ systems: Object.fromEntries(SYSTEM_NAMES.map((name) => [name, false])),
150
+ budgets: { ...DEFAULT_BUDGETS },
151
+ progressNudge: "notify",
152
+ };
153
+ }
154
+
155
+ function clamp(value, fallback, ceiling) {
156
+ return Number.isInteger(value) ? Math.min(Math.max(value, 0), ceiling) : fallback;
157
+ }
158
+
159
+ function normalizeBudgets(raw) {
160
+ const budgets = { total: clamp(raw?.total, DEFAULT_BUDGETS.total, MAX_TOTAL_BUDGET) };
161
+ for (const name of SYSTEM_NAMES) {
162
+ budgets[name] = clamp(raw?.[name], DEFAULT_BUDGETS[name] ?? 0, MAX_CALL_BUDGET);
163
+ }
164
+
165
+ return budgets;
166
+ }
167
+
168
+ /**
169
+ * Schema 1 carried one `callBudgetPerSession`. Reading it as an unknown shape would switch the
170
+ * layer off for anyone who had turned it on, which is a reset the user never asked for -- "unknown
171
+ * shapes collapse to all-off" is a rule for corrupt input, not for our own previous version. The
172
+ * one number becomes the total, and each system gets the smaller of its default and that total, so
173
+ * a user who set a deliberately tight ceiling keeps it.
174
+ */
175
+ function migrate(raw) {
176
+ // Schema 1 read 0 as "no ceiling"; schema 2 reads it as "no calls", because a per-system 0 that
177
+ // silently meant unlimited is the wrong way for a budget to fail. Carrying the old meaning
178
+ // forward here is what stops the bump from inverting a user's intent.
179
+ const stored = raw?.callBudgetPerSession === 0 ? MAX_TOTAL_BUDGET : raw?.callBudgetPerSession;
180
+ const total = clamp(stored, DEFAULT_BUDGETS.total, MAX_TOTAL_BUDGET);
181
+ const budgets = { total };
182
+ for (const name of SYSTEM_NAMES) {
183
+ budgets[name] = Math.min(DEFAULT_BUDGETS[name] ?? 0, total);
184
+ }
185
+
186
+ return { ...raw, schema: 2, budgets, callBudgetPerSession: undefined };
187
+ }
188
+
189
+ /**
190
+ * Schema 2 carried the command guard as a separate `guard: { enabled, startup }` pair, because it
191
+ * was a separate package with its own global configuration file. Schema 3 makes it the eighth
192
+ * system, so the stored preference becomes `systems.guard`.
193
+ *
194
+ * `guard.startup` is what migrates, not `guard.enabled`: the former is what the user chose for new
195
+ * sessions, and the latter was a session flag that happened to be written to disk. A file where the
196
+ * guard was wanted at startup but the layer itself was not produces a system that is on inside a
197
+ * layer that is off, which is inactive -- the guard used to sit outside the master switch and now
198
+ * does not. That direction is deliberate: a gate quietly becoming inactive is recoverable in one
199
+ * command, and a gate quietly becoming active is how a session stops being able to run anything.
200
+ */
201
+ function migrateToThree(raw) {
202
+ const systems = { ...(raw?.systems ?? {}), guard: raw?.guard?.startup === true };
203
+
204
+ const { guard: _guard, ...rest } = raw ?? {};
205
+
206
+ return { ...rest, schema: 3, systems };
207
+ }
208
+
209
+ function normalize(raw) {
210
+ const one = raw?.schema === 1 ? migrate(raw) : raw;
211
+ const source = one?.schema === 2 ? migrateToThree(one) : one;
212
+ if (source?.schema !== 3) {
213
+ return defaultSettings();
214
+ }
215
+
216
+ const systems = Object.fromEntries(SYSTEM_NAMES.map((name) => [name, source.systems?.[name] === true]));
217
+
218
+ return {
219
+ schema: 3,
220
+ master: source.master === true,
221
+ startup: source.startup === true,
222
+ systems,
223
+ budgets: normalizeBudgets(source.budgets),
224
+ progressNudge: NUDGE_MODES.includes(source.progressNudge) ? source.progressNudge : "notify",
225
+ };
226
+ }
227
+
228
+ export function loadSettings() {
229
+ try {
230
+ const file = settingsFile();
231
+ if (!regularFile(file, "Jev settings")) {
232
+ return defaultSettings();
233
+ }
234
+
235
+ return normalize(JSON.parse(fs.readFileSync(file, "utf8")));
236
+ } catch {
237
+ return defaultSettings();
238
+ }
239
+ }
240
+
241
+ /** Shared atomic write for every file this extension owns. */
242
+ export function writeFileAtomic(file, contents) {
243
+ const directory = path.dirname(file);
244
+ fs.mkdirSync(directory, { recursive: true, mode: 0o700 });
245
+ const temporary = path.join(directory, `.${path.basename(file)}.${randomUUID()}.tmp`);
246
+ try {
247
+ fs.writeFileSync(temporary, contents, { mode: 0o600, flag: "wx" });
248
+ fs.renameSync(temporary, file);
249
+ } finally {
250
+ fs.rmSync(temporary, { force: true });
251
+ }
252
+ }
253
+
254
+ export function saveSettings(settings) {
255
+ // A caller handing back a schema-1 shape is migrated rather than reset, so a round trip through
256
+ // an old reader cannot quietly disable the layer.
257
+ const next = normalize(settings?.schema === 1 || settings?.schema === 2 ? settings : { ...settings, schema: 3 });
258
+ const file = settingsFile();
259
+ if (fs.existsSync(file)) {
260
+ regularFile(file, "Jev settings");
261
+ }
262
+
263
+ writeFileAtomic(file, `${JSON.stringify(next, null, 4)}\n`);
264
+
265
+ return next;
266
+ }
267
+
268
+ export function settingsPath() {
269
+ return settingsFile();
270
+ }
@@ -0,0 +1,133 @@
1
+ // A config flag records an intention. It does not record that a human was told what the intention
2
+ // costs. This file holds the separate, explicit grant: the first time any system would put session
3
+ // state on the wire, a dialog names the endpoint, the shape of the data and the byte budget, and
4
+ // nothing is sent until someone says yes.
5
+ //
6
+ // Mirrors capability-policy: SpecPi's own file, atomic, mode 0600, symlinks refused, missing or
7
+ // unreadable read as "ask". No interactive UI means no send, ever.
8
+
9
+ import fs from "node:fs";
10
+ import path from "node:path";
11
+ import { jevDirectory, regularFile, writeFileAtomic } from "./config.mjs";
12
+ import { baseUrl } from "./client.mjs";
13
+ import { MAX_STATE_BYTES } from "./sanitize.mjs";
14
+
15
+ /**
16
+ * The host the bytes will actually reach, not the service they are nominally for. This was a fixed
17
+ * "api.typesafe.ai" until the default backend became OpenRouter, at which point the dialog named a
18
+ * host the data no longer went to, which is the one thing a consent dialog may never do. It is
19
+ * derived now, so each of the three selectable destinations (OpenRouter, the direct TypeSafe API,
20
+ * and a TYPESAFE_BASE_URL override) names itself, and so the rule below is true rather than
21
+ * aspirational: a grant is keyed on this string, so repointing the client really does ask again.
22
+ */
23
+ export function endpointLabel() {
24
+ const base = baseUrl();
25
+ try {
26
+ return new URL(base).host;
27
+ } catch {
28
+ // Unparseable means the fetch will fail anyway. Returning the raw value keeps the dialog
29
+ // honest and cannot collide with a host a real grant was given for.
30
+ return base;
31
+ }
32
+ }
33
+
34
+ function consentFile() {
35
+ return path.join(jevDirectory(), "consent.json");
36
+ }
37
+
38
+ export function loadConsent() {
39
+ try {
40
+ const file = consentFile();
41
+ if (!regularFile(file, "Jev consent")) {
42
+ return undefined;
43
+ }
44
+
45
+ const stored = JSON.parse(fs.readFileSync(file, "utf8"));
46
+ if (stored?.schema !== 1 || stored.granted !== true || typeof stored.endpoint !== "string") {
47
+ return undefined;
48
+ }
49
+
50
+ // A grant is for the endpoint it was given for. Repointing the client asks again.
51
+ return stored.endpoint === endpointLabel() ? stored : undefined;
52
+ } catch {
53
+ return undefined;
54
+ }
55
+ }
56
+
57
+ export function granted() {
58
+ return loadConsent() !== undefined;
59
+ }
60
+
61
+ export function saveConsent() {
62
+ const file = consentFile();
63
+ if (fs.existsSync(file)) {
64
+ regularFile(file, "Jev consent");
65
+ }
66
+
67
+ const stored = {
68
+ schema: 1,
69
+ granted: true,
70
+ endpoint: endpointLabel(),
71
+ maxStateBytes: MAX_STATE_BYTES,
72
+ grantedAt: new Date().toISOString(),
73
+ };
74
+ writeFileAtomic(file, `${JSON.stringify(stored, null, 4)}\n`);
75
+
76
+ return stored;
77
+ }
78
+
79
+ export function revokeConsent() {
80
+ try {
81
+ fs.rmSync(consentFile(), { force: true });
82
+ } catch {
83
+ // A consent file that cannot be removed still reads as granted; the master switch is the
84
+ // reliable stop, and /jev status reports both.
85
+ }
86
+ }
87
+
88
+ export function consentPath() {
89
+ return consentFile();
90
+ }
91
+
92
+ export const CONSENT_TITLE = "Allow SpecPi to send task summaries to Jev?";
93
+
94
+ export function consentBody(systemLabel) {
95
+ return [
96
+ `${systemLabel} wants to ask TypeSafe's Jev classifier a question about this session.`,
97
+ "",
98
+ `What is sent: a summary object of at most ${MAX_STATE_BYTES} bytes to ${endpointLabel()}, over HTTPS.`,
99
+ "It carries tool names, byte counts, relative paths and short descriptions.",
100
+ "It never carries file contents, command output, credentials, URLs or session history.",
101
+ "",
102
+ "Every call is recorded locally in transmissions.jsonl with a hash of exactly what was sent,",
103
+ "which you can read with /jev ledger. Turn this off at any time with /jev off.",
104
+ ].join("\n");
105
+ }
106
+
107
+ /**
108
+ * Resolve consent for a system, prompting once. Returns false without prompting when there is no
109
+ * interactive human: an advisor must never be the reason a headless run sends data.
110
+ */
111
+ export async function ensureConsent(ctx, systemLabel) {
112
+ if (granted()) {
113
+ return true;
114
+ }
115
+
116
+ if (!ctx?.hasUI || typeof ctx.ui?.confirm !== "function") {
117
+ return false;
118
+ }
119
+
120
+ const accepted = await ctx.ui.confirm(CONSENT_TITLE, consentBody(systemLabel));
121
+ if (!accepted) {
122
+ return false;
123
+ }
124
+
125
+ try {
126
+ saveConsent();
127
+ } catch {
128
+ // An unwritable grant means asking again next time, which is the safe direction.
129
+ return true;
130
+ }
131
+
132
+ return true;
133
+ }
@@ -0,0 +1,263 @@
1
+ // A probability is not a decision. This file is the only place that turns one into the other, so
2
+ // every system gates the same way and a threshold change is a one-line diff with one test to move.
3
+ //
4
+ // The three primitives do not gate alike. A Noul returns a bare probability with no confidence
5
+ // field, so a confidence test cannot be applied to it. A Choice can be confident and still be a
6
+ // coin flip between its top two options, so the margin matters as much as the confidence. A Score
7
+ // is only actionable when it sits clear of a level boundary rather than straddling one.
8
+ //
9
+ // MEASURED, not guessed. Every number here is now read off `evals/runs/jev-calibration.json`, which
10
+ // `node scripts/jev-calibrate.mjs` writes from 259 recorded eval attempts plus a fixed reachability
11
+ // pass of six synthetic cases run five times each. `tests/jev-calibration.test.mjs` pins each number
12
+ // to that file, so moving one takes new evidence rather than a new opinion.
13
+ //
14
+ // Two questions were asked of the evidence, because a threshold can fail in two different ways.
15
+ //
16
+ // 1. DOES THE CONFIDENCE FIELD SEPARATE RIGHT FROM WRONG? Measured against labels this repository
17
+ // already owns: predicting an attempt's pass (Noul), its task category (Choice) and its tier
18
+ // (Score) from behavioural metadata alone, with the labels withheld from the state.
19
+ //
20
+ // Score: yes, weakly. At scoreConfidence 0.60 and boundary 0.30 the answer is exactly right
21
+ // about two thirds of the time and within one level about 96%, on roughly a fifth of
22
+ // answers, against a 43.2% majority class. The exact figures are in CALIBRATION below,
23
+ // which a test compares against the artifact; they moved from 68.4% to 64.7% between two
24
+ // runs over the same 259 attempts, so any single decimal here is a sample, not a
25
+ // constant, and the honest summary is "a lift of about 1.5".
26
+ // Noul: no. Precision tracks the 90.7% base rate at every threshold (lift 1.01-1.02), and no
27
+ // answer to that question ever exceeded 0.80.
28
+ // Choice: no. About 30% accuracy against a 29.7% majority class, and accuracy falls as
29
+ // confidence rises. The margin changes nothing, because the top-two gap is almost always
30
+ // wide.
31
+ //
32
+ // None of the systems' pre-registered precision targets (0.75 to 0.95) is met anywhere on any of
33
+ // those curves, and the artifact records UNMET rather than a number chosen to fill the gap. The
34
+ // Score point below is the best available operating point, not a met target. That is a real
35
+ // limit on what this layer can claim, and it is published rather than smoothed over.
36
+ //
37
+ // It is also a fair reading that the proxy questions are much harder than the production ones:
38
+ // the eval state is a row of counters, while a production state carries the material being
39
+ // judged. Question 2 is what tests that, and the answer is yes.
40
+ //
41
+ // 2. CAN THE GATE EVER FIRE? This is the one that found a shipped defect. Against the fixture cases
42
+ // the real question sets separate cleanly -- a planted secret scores 0.96 and a clean report
43
+ // 0.04; a page carrying an injected instruction scores 0.97 and an ordinary one 0.04; the one
44
+ // relevant file among noise scores 1.99 while the other two score 0.01 -- but the Score gate
45
+ // shipped at confidence 0.80 with boundary 0.35, and on a maximally obvious "spent" result (a
46
+ // listing of vendor icons during a changelog edit) Jev answered 0.10-0.18 with confidence
47
+ // 0.73-0.85 across ten runs, most of them under 0.80. Boundary 0.35 demands the value sit within
48
+ // 0.15 of a level, which 0.16 and 0.17 miss. So retention could gate through to "keep this
49
+ // result" and essentially never to "this result is spent": the only branch that does anything
50
+ // was unreachable, and running the layer could never have revealed it, because a system that
51
+ // never fires looks exactly like a system whose advice was always to do nothing.
52
+ //
53
+ // compaction's `unresolved_thread` had the same problem at high 0.85: the clearest open
54
+ // investigation the fixture can express scores 0.63-0.65. It is lowered to 0.60, which is
55
+ // defensible only because of what that branch does -- add one sentence to a summariser prompt
56
+ // that is being rebuilt from scratch anyway. It is the cheapest action in the layer, so it can
57
+ // afford the loosest gate. gap keeps 0.85 because its Noul reaches 0.96 on the case that matters
58
+ // and because a firing there blocks a write.
59
+
60
+ /**
61
+ * The figures the comment above cites, in a form a test can check against the artifact. A citation
62
+ * that drifts from its source is worse than no citation, because it reads like evidence. If these
63
+ * stop matching `evals/runs/jev-calibration.json`, `tests/jev-calibration.test.mjs` fails and
64
+ * whoever re-ran the calibration has to update the prose too.
65
+ */
66
+ export const CALIBRATION = Object.freeze({
67
+ artifact: "evals/runs/jev-calibration.json",
68
+ attempts: 259,
69
+ passBaseRate: 0.907,
70
+ tierBaseRate: 0.432,
71
+ categoryBaseRate: 0.297,
72
+ // At the shipped scoreConfidence 0.60 / boundary 0.30. Tolerance is deliberate: see above.
73
+ scoreExact: 0.647,
74
+ scoreWithinOne: 0.961,
75
+ scoreCoverage: 0.197,
76
+ tolerance: 0.05,
77
+ });
78
+
79
+ export const THRESHOLDS = Object.freeze({
80
+ // One Score operating point, applied to every system, because one proxy question produced one
81
+ // curve. Four different per-system numbers would be four claims from a single measurement. The
82
+ // per-system asymmetry lives where it belongs instead: retention needs two independent answers
83
+ // to agree before it shortens anything, and sources only ever reorders.
84
+ retention: Object.freeze({
85
+ scoreConfidence: 0.6,
86
+ boundary: 0.3,
87
+ choiceConfidence: 0.75,
88
+ margin: 0.3,
89
+ // Unused by retention, which reads only the low side; kept so every system has a full set.
90
+ high: 0.9,
91
+ // Measured 0.06-0.07 on the spent case, so this clears with room.
92
+ low: 0.1,
93
+ }),
94
+ compaction: Object.freeze({
95
+ scoreConfidence: 0.6,
96
+ boundary: 0.3,
97
+ choiceConfidence: 0.75,
98
+ margin: 0.25,
99
+ // Lowered from 0.85: the clearest open investigation scores 0.63-0.65, and the action is
100
+ // one sentence added to a prompt that is being rebuilt regardless.
101
+ high: 0.6,
102
+ low: 0.15,
103
+ }),
104
+ gap: Object.freeze({
105
+ scoreConfidence: 0.6,
106
+ boundary: 0.3,
107
+ choiceConfidence: 0.75,
108
+ margin: 0.2,
109
+ // Measured 0.96 on a report carrying a machine-specific path and 0.04 on a clean one.
110
+ high: 0.85,
111
+ low: 0.15,
112
+ }),
113
+ progress: Object.freeze({
114
+ scoreConfidence: 0.6,
115
+ boundary: 0.3,
116
+ // The strictest Choice gate in the layer, because it is the only system whose action can
117
+ // change what the model does next. Running the same taxonomy over the 24 recorded failures
118
+ // left 14 of them below this bar, which is the intended behaviour: silence is the correct
119
+ // answer to a session whose trouble is not yet legible.
120
+ choiceConfidence: 0.8,
121
+ margin: 0.25,
122
+ high: 0.85,
123
+ low: 0.15,
124
+ }),
125
+ capability: Object.freeze({
126
+ scoreConfidence: 0.6,
127
+ boundary: 0.3,
128
+ choiceConfidence: 0.75,
129
+ margin: 0.2,
130
+ // Asymmetric on purpose. A false positive costs the group's schema on every request for
131
+ // the rest of the session -- a turn-1 arming measured 16% more than never arming -- plus a
132
+ // confirmation the human did not need. A false negative costs nothing: it leaves today's
133
+ // behaviour exactly as it is, and `request_capability` is still there for the moment the
134
+ // need becomes real.
135
+ //
136
+ // This shipped at 0.90 for exactly as long as it took to measure it, which is the same
137
+ // defect described above, in code written the same day, caught by the same check. On a
138
+ // request that unambiguously needs a browser -- open the pricing page at 375px and fix what
139
+ // overflows -- `needs_browser` answers 0.86-0.87, five times out of five, while the same
140
+ // question on a rename answers 0.09-0.10. 0.90 sits inside the yes cluster and rejects all
141
+ // of it; 0.85 sits below it with a margin of 0.76 to the nearest no. No Noul anywhere in
142
+ // the fixture set has ever exceeded 0.97, so "higher is safer" stops being true well before
143
+ // it stops being tempting.
144
+ high: 0.85,
145
+ low: 0.1,
146
+ }),
147
+ untrusted: Object.freeze({
148
+ scoreConfidence: 0.6,
149
+ boundary: 0.3,
150
+ choiceConfidence: 0.75,
151
+ margin: 0.2,
152
+ // A banner is cheap and a missed injection is not, so this is the one place where the
153
+ // asymmetry runs the other way from gap's. It is still 0.85 rather than lower, because a
154
+ // banner on ordinary prose is exactly the false positive that teaches a model to stop
155
+ // reading the channel -- the objection this system had to answer before it could exist.
156
+ high: 0.85,
157
+ low: 0.15,
158
+ }),
159
+ guard: Object.freeze({
160
+ // The same measured Score operating point as every other system, written out rather than
161
+ // inherited by falling through `thresholdsFor`'s default. The guard asked under the name
162
+ // "gap" for exactly as long as it took to notice that adding a `guard` entry would then have
163
+ // changed nothing -- a silent no-op on the one system whose action takes a tool call away.
164
+ scoreConfidence: 0.6,
165
+ boundary: 0.3,
166
+ choiceConfidence: 0.75,
167
+ margin: 0.2,
168
+ // Unused: this system reads only the Score side. Kept so every system has a full set, and
169
+ // so the calibration tests cover this entry like any other.
170
+ high: 0.85,
171
+ low: 0.15,
172
+ }),
173
+ sources: Object.freeze({
174
+ scoreConfidence: 0.6,
175
+ boundary: 0.3,
176
+ choiceConfidence: 0.7,
177
+ margin: 0.15,
178
+ // Not reached by any fixture case: `worth_delegating` scored 0.20-0.21 on a question that
179
+ // reads as self-contained to a human. Left at 0.85 rather than tuned down, because the only
180
+ // thing that reads it warns and never blocks, and a threshold moved to make a fixture pass
181
+ // is a threshold set by the fixture.
182
+ high: 0.85,
183
+ low: 0.15,
184
+ }),
185
+ });
186
+
187
+ export function thresholdsFor(system) {
188
+ return THRESHOLDS[system] ?? THRESHOLDS.gap;
189
+ }
190
+
191
+ /** True when the Noul is confidently yes. */
192
+ export function nounTrue(answer, system) {
193
+ const limits = thresholdsFor(system);
194
+
195
+ return answer?.kind === "noul" && answer.value >= limits.high;
196
+ }
197
+
198
+ /** True when the Noul is confidently no. Not the negation of nounTrue: the middle band is silence. */
199
+ export function nounFalse(answer, system) {
200
+ const limits = thresholdsFor(system);
201
+
202
+ return answer?.kind === "noul" && answer.value <= limits.low;
203
+ }
204
+
205
+ function topTwo(probabilities) {
206
+ const values = Object.values(probabilities ?? {})
207
+ .filter((value) => typeof value === "number")
208
+ .sort((a, b) => b - a);
209
+
210
+ return { first: values[0] ?? 0, second: values[1] ?? 0 };
211
+ }
212
+
213
+ /**
214
+ * A Choice is actionable when it is both confident and clearly separated from its runner-up.
215
+ * Without a distribution the margin cannot be checked, so the answer is treated as ungated.
216
+ *
217
+ * The reachability pass confirms all 40 Choice answers from this backend carried a distribution, so
218
+ * the margin is a live test rather than a branch that silently never runs.
219
+ */
220
+ export function choiceValue(answer, system) {
221
+ if (answer?.kind !== "choice" || typeof answer.confidence !== "number") {
222
+ return undefined;
223
+ }
224
+
225
+ const limits = thresholdsFor(system);
226
+ if (answer.confidence < limits.choiceConfidence) {
227
+ return undefined;
228
+ }
229
+
230
+ const { first, second } = topTwo(answer.probabilities);
231
+ if (answer.probabilities && first - second < limits.margin) {
232
+ return undefined;
233
+ }
234
+
235
+ return answer.value;
236
+ }
237
+
238
+ /**
239
+ * A Score is actionable when it is confident and sits clear of the nearest level boundary. The
240
+ * returned level is the rounded band; callers compare against their own rubric.
241
+ */
242
+ export function scoreLevel(answer, system) {
243
+ if (answer?.kind !== "score" || typeof answer.confidence !== "number") {
244
+ return undefined;
245
+ }
246
+
247
+ const limits = thresholdsFor(system);
248
+ if (answer.confidence < limits.scoreConfidence) {
249
+ return undefined;
250
+ }
251
+
252
+ const level = Math.round(answer.value);
253
+ if (Math.abs(answer.value - level) > 0.5 - limits.boundary) {
254
+ return undefined;
255
+ }
256
+
257
+ return level;
258
+ }
259
+
260
+ /** Raw probability for logging and calibration, with no gate applied. */
261
+ export function rawValue(answer) {
262
+ return answer?.value;
263
+ }