driftproof 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -5
- package/bin/driftproof +191 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +74 -15
- package/lib/models.js +52 -11
- package/lib/provider.js +265 -103
- package/lib/receipt.js +51 -11
- package/lib/run.js +231 -26
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +38 -7
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/provider.js
CHANGED
|
@@ -4,9 +4,12 @@
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const os = require('os');
|
|
6
6
|
const path = require('path');
|
|
7
|
-
|
|
7
|
+
// Called through the module object (cp.spawn), never destructured: the gate
|
|
8
|
+
// records spawns by replacing that property, so nothing real runs offline.
|
|
9
|
+
const cp = require('child_process');
|
|
8
10
|
const { withRetry, withTimeout } = require('./json');
|
|
9
11
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
12
|
+
const { readAnthropicApiResponse, readOpenaiApiResponse } = require('./usage');
|
|
10
13
|
const {
|
|
11
14
|
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
|
|
12
15
|
} = require('./usage');
|
|
@@ -126,15 +129,23 @@ const CODEX_OVERHEAD_NOTE =
|
|
|
126
129
|
//
|
|
127
130
|
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
128
131
|
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
129
|
-
|
|
132
|
+
// `trusted` (spec 022) selects the SAME-USER legacy spawn on the two CLI lanes.
|
|
133
|
+
// It is false unless a caller says otherwise; bin/driftproof exposes it as
|
|
134
|
+
// --trusted-skill, for skills the operator authored. Api lanes ignore it.
|
|
135
|
+
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined, trusted = false }) {
|
|
130
136
|
const surface = surfaceForModel(model);
|
|
131
137
|
const provider = providerForSurface(surface);
|
|
132
|
-
// Offline stub surface: canned completion, zero model calls. The
|
|
133
|
-
//
|
|
134
|
-
//
|
|
138
|
+
// Offline stub surface: canned completion, zero model calls. The reply says
|
|
139
|
+
// so: surface `stub`, answeredBy `stub`, no model reported, no stop reason,
|
|
140
|
+
// no isolation (nothing was spawned). Spec 026 F1: the receipt used to record
|
|
141
|
+
// the REAL surface name here and read TESTED while the text was canned; now
|
|
142
|
+
// the receipt's surface, level and answered_by are derived from what this
|
|
143
|
+
// function returned, and a stub run reads UNVERIFIED with surface stub. The
|
|
144
|
+
// requested model's provider is still reported, so a reader knows what was
|
|
145
|
+
// asked for.
|
|
135
146
|
if (stubEnabled()) {
|
|
136
147
|
const s = stubComplete({ system, prompt });
|
|
137
|
-
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
|
|
148
|
+
return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface: 'stub', provider, attempts: 1, answeredBy: 'stub', reportedModels: null, stopReason: null, isolation: 'none' };
|
|
138
149
|
}
|
|
139
150
|
|
|
140
151
|
// Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
|
|
@@ -149,9 +160,9 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
149
160
|
const laneRunner = () => {
|
|
150
161
|
switch (surface) {
|
|
151
162
|
case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
|
|
152
|
-
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
163
|
+
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
|
|
153
164
|
case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
|
|
154
|
-
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
165
|
+
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
|
|
155
166
|
default: throw new Error(`unknown surface: ${surface}`);
|
|
156
167
|
}
|
|
157
168
|
};
|
|
@@ -172,7 +183,16 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
172
183
|
try {
|
|
173
184
|
const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
|
|
174
185
|
const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
|
|
175
|
-
|
|
186
|
+
// v0.6 (spec 026 AC-1, AC-2, AC-8): a model surface answered; what it said
|
|
187
|
+
// served the call, why the reply stopped, and which spawn path was taken
|
|
188
|
+
// are what the lane could read, null where its surface reports nothing.
|
|
189
|
+
return {
|
|
190
|
+
...out, usage, wall_ms: wallMs, surface, provider, attempts,
|
|
191
|
+
answeredBy: 'model',
|
|
192
|
+
reportedModels: Array.isArray(out.reportedModels) && out.reportedModels.length ? out.reportedModels : null,
|
|
193
|
+
stopReason: typeof out.stopReason === 'string' && out.stopReason ? out.stopReason : null,
|
|
194
|
+
isolation: out.isolation || 'none',
|
|
195
|
+
};
|
|
176
196
|
} catch (e) {
|
|
177
197
|
// Surface the attempt count so a persistently-failing call can be charged for
|
|
178
198
|
// (and, for a timeout, marked failed_timeout by the runner instead of fatal).
|
|
@@ -197,50 +217,191 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
|
|
|
197
217
|
if (temperature !== undefined) params.temperature = temperature;
|
|
198
218
|
const resp = await client.messages.create(params);
|
|
199
219
|
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
200
|
-
|
|
220
|
+
// v0.6: the response names the model that served it and why it stopped.
|
|
221
|
+
return { text, usage: parseAnthropicApiUsage(resp.usage), ...readAnthropicApiResponse(resp), isolation: 'none' };
|
|
201
222
|
}
|
|
202
223
|
|
|
203
|
-
// ──
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
224
|
+
// ── isolation: the eval-user hop (spec 022) ───────────────────────────────────
|
|
225
|
+
//
|
|
226
|
+
// Every local-CLI spawn runs as a DEDICATED UNPRIVILEGED UNIX USER by default.
|
|
227
|
+
// The third-party SKILL.md a run evaluates is instructions to an agent with
|
|
228
|
+
// tools, and until spec 022 that agent ran as the operator: same uid, whole
|
|
229
|
+
// environment block, operator's cwd, operator's CLI configuration. The
|
|
230
|
+
// 2026-09-03 security audit (findings A2, N5) confirmed a live API token in the
|
|
231
|
+
// inherited environment and this host's SSH keys and CLI session tokens
|
|
232
|
+
// readable by the same uid; this was then reproduced by execution.
|
|
233
|
+
//
|
|
234
|
+
// The boundary is a uid change reached through sudo from an EMPTY environment:
|
|
235
|
+
//
|
|
236
|
+
// /usr/bin/sudo -n -u <eval-user> /usr/bin/env -i HOME=/home/<eval-user>
|
|
237
|
+
// PATH=/home/<eval-user>/.local/bin:/usr/bin:/bin bash -lc '<wrapper>'
|
|
238
|
+
// driftproof-eval-hop <cli> <cli-args...>
|
|
239
|
+
//
|
|
240
|
+
// Operator prerequisite, NOT created here (RUNBOOK.md): the user exists with a
|
|
241
|
+
// mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
|
|
242
|
+
// under ~/.local/bin, and a sudoers rule grants the operator NOPASSWD as that
|
|
243
|
+
// user. The runner refuses loudly when any of that is missing; it never creates
|
|
244
|
+
// a user, edits sudoers, or logs a CLI in.
|
|
245
|
+
//
|
|
246
|
+
// The same-user (legacy) spawn survives ONLY behind `trusted: true`, exposed by
|
|
247
|
+
// bin/driftproof as --trusted-skill, for skills the operator authored.
|
|
248
|
+
const SUDO_BIN = '/usr/bin/sudo';
|
|
249
|
+
const ENV_BIN = '/usr/bin/env';
|
|
250
|
+
const EVAL_USER_DEFAULT = 'driftproof-eval';
|
|
251
|
+
// A user name that can safely follow `sudo -u`: no leading dash, no shell
|
|
252
|
+
// metacharacter, no path separator. Anything else is refused before any spawn.
|
|
253
|
+
const EVAL_USER_RE = /^[a-z_][a-z0-9_-]{0,31}$/;
|
|
254
|
+
// The ONLY names the child environment carries, each CONSTRUCTED from the eval
|
|
255
|
+
// user's name in isolatedEnv() and never copied from process.env. Verified
|
|
256
|
+
// 2026-09-03: both CLIs run through the hop with these two alone; the login
|
|
257
|
+
// shell supplies LANG and TERM itself. Data, frozen, asserted by the gate.
|
|
258
|
+
const ISOLATED_ENV_ALLOWLIST = Object.freeze(['HOME', 'PATH']);
|
|
259
|
+
// $0 of the wrapper, so `ps` shows what the bash process is.
|
|
260
|
+
const HOP_LABEL = 'driftproof-eval-hop';
|
|
261
|
+
// The wrapper runs INSIDE the hop, as the eval user. It takes the CLI as
|
|
262
|
+
// positional parameters ("$@"), so no SKILL.md text is ever shell-interpreted.
|
|
263
|
+
// - the cwd is a fresh per-call directory under the system tmp, created by
|
|
264
|
+
// the eval user (one the parent creates is mode 700 to it);
|
|
265
|
+
// - the CLI runs under setsid in its own process group, stdin preserved
|
|
266
|
+
// (`<&0` defeats the /dev/null a non-interactive bash gives a background job);
|
|
267
|
+
// - TERM/INT/HUP, which sudo relays from the parent on timeout, kill that
|
|
268
|
+
// whole group, remove the directory and exit 143;
|
|
269
|
+
// - a normal exit removes the directory and returns the CLI's status.
|
|
270
|
+
const ISOLATED_WRAPPER = [
|
|
271
|
+
'd=$(mktemp -d -t driftproof-eval.XXXXXXXX) || exit 97',
|
|
272
|
+
'cd "$d" || exit 97',
|
|
273
|
+
'setsid -w "$@" <&0 & pid=$!',
|
|
274
|
+
'trap \'kill -TERM -- "-$pid" 2>/dev/null; sleep 1; kill -KILL -- "-$pid" 2>/dev/null; cd /; rm -rf "$d"; exit 143\' TERM INT HUP',
|
|
275
|
+
'wait "$pid"; r=$?',
|
|
276
|
+
'cd /; rm -rf "$d"; exit $r',
|
|
277
|
+
].join('; ');
|
|
278
|
+
|
|
279
|
+
// DRIFTPROOF_EVAL_USER selects WHICH user isolates, never WHETHER isolation
|
|
280
|
+
// happens (that is the trusted flag, an argv decision). Empty means default.
|
|
281
|
+
function evalUser() {
|
|
282
|
+
const raw = process.env.DRIFTPROOF_EVAL_USER;
|
|
283
|
+
const u = raw === undefined || raw === '' ? EVAL_USER_DEFAULT : raw;
|
|
284
|
+
if (!EVAL_USER_RE.test(u)) throw new Error(`DRIFTPROOF_EVAL_USER is not a plausible unix user name: ${JSON.stringify(u)}`);
|
|
285
|
+
return u;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
// The child environment: constructed, not copied. Keys are exactly the allowlist.
|
|
289
|
+
function isolatedEnv(user) {
|
|
290
|
+
const home = `/home/${user}`;
|
|
291
|
+
return { HOME: home, PATH: `${home}/.local/bin:/usr/bin:/bin` };
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// The isolation a plan records into the receipt (run.answered_by.isolation,
|
|
295
|
+
// spec 026 AC-2): the hop is `eval-user`, the legacy spawn `same-user`.
|
|
296
|
+
function isolationOf(plan) { return plan && plan.mode === 'isolated' ? 'eval-user' : 'same-user'; }
|
|
297
|
+
|
|
298
|
+
// The spawn plan for one CLI call: { mode, user, file, args, env }. Pure, so the
|
|
299
|
+
// gate can inspect what WOULD be spawned. Untrusted → the hop; trusted → the
|
|
300
|
+
// legacy same-user spawn exactly as it was (env inherited; the claude lane still
|
|
301
|
+
// strips ANTHROPIC_API_KEY so the CLI uses the subscription, not the metered key).
|
|
302
|
+
function buildSpawnPlan({ bin, args, trusted = false }) {
|
|
303
|
+
if (trusted) {
|
|
208
304
|
const env = { ...process.env };
|
|
209
|
-
delete env.ANTHROPIC_API_KEY;
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
305
|
+
if (bin === 'claude') delete env.ANTHROPIC_API_KEY;
|
|
306
|
+
return { mode: 'same-user', user: null, file: bin, args: [...args], env };
|
|
307
|
+
}
|
|
308
|
+
const user = evalUser();
|
|
309
|
+
const childEnv = isolatedEnv(user);
|
|
310
|
+
return {
|
|
311
|
+
mode: 'isolated',
|
|
312
|
+
user,
|
|
313
|
+
file: SUDO_BIN,
|
|
314
|
+
args: ['-n', '-u', user, ENV_BIN, '-i', ...ISOLATED_ENV_ALLOWLIST.map((k) => `${k}=${childEnv[k]}`), 'bash', '-lc', ISOLATED_WRAPPER, HOP_LABEL, bin, ...args],
|
|
315
|
+
// The sudo front process gets NOTHING from this process. sudo resolves
|
|
316
|
+
// /usr/bin/env itself; env -i then starts the child from empty.
|
|
317
|
+
env: {},
|
|
318
|
+
};
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Run a plan: `input` to stdin, stdout/stderr collected, settled on close.
|
|
322
|
+
// The timeout signal differs by mode. The legacy child gets SIGKILL as it
|
|
323
|
+
// always did. The hop gets SIGTERM: it is the signal the operator's uid can
|
|
324
|
+
// deliver to a root-owned sudo, and the one sudo relays to the wrapper's trap.
|
|
325
|
+
// SIGKILL would neither be permitted nor relayed, and a wrapper without the
|
|
326
|
+
// trap left the CLI orphaned with the stdout pipe open, so close never fired.
|
|
327
|
+
function runPlan(plan, { input = null, timeoutMs = 300000, graceMs = 5000 } = {}) {
|
|
328
|
+
return new Promise((resolve, reject) => {
|
|
329
|
+
let child;
|
|
330
|
+
try { child = cp.spawn(plan.file, plan.args, { env: plan.env, stdio: ['pipe', 'pipe', 'pipe'] }); }
|
|
331
|
+
catch (e) { return reject(e); }
|
|
332
|
+
let stdout = '';
|
|
333
|
+
let stderr = '';
|
|
334
|
+
let timedOut = false;
|
|
335
|
+
const killer = setTimeout(() => {
|
|
336
|
+
timedOut = true;
|
|
337
|
+
child.kill(plan.mode === 'isolated' ? 'SIGTERM' : 'SIGKILL');
|
|
338
|
+
}, timeoutMs + graceMs);
|
|
339
|
+
child.stdout.on('data', (d) => { stdout += d; });
|
|
340
|
+
child.stderr.on('data', (d) => { stderr += d; });
|
|
341
|
+
child.on('error', (e) => { clearTimeout(killer); reject(e); });
|
|
342
|
+
child.on('close', (code, signal) => { clearTimeout(killer); resolve({ code, signal, stdout, stderr, timedOut }); });
|
|
343
|
+
child.stdin.on('error', () => { /* EPIPE if the CLI exits before reading all of stdin */ });
|
|
344
|
+
if (input != null) child.stdin.write(input);
|
|
240
345
|
child.stdin.end();
|
|
241
346
|
});
|
|
242
347
|
}
|
|
243
348
|
|
|
349
|
+
// The hop with an arbitrary command in place of the CLI, for the gate: the same
|
|
350
|
+
// plan builder and the same runner, never trusted. No provider is reached.
|
|
351
|
+
function runIsolated(cmdArgv, opts = {}) {
|
|
352
|
+
const [bin, ...args] = cmdArgv;
|
|
353
|
+
return runPlan(buildSpawnPlan({ bin, args, trusted: false }), { graceMs: 0, ...opts });
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// A hop failure, said in terms of what to fix. Returns null when the failure is
|
|
357
|
+
// the CLI's own (exit status from inside the hop), which the lane reports as before.
|
|
358
|
+
function hopFailure(bin, r) {
|
|
359
|
+
const err = String(r.stderr || '').trim();
|
|
360
|
+
if (r.code === 127) {
|
|
361
|
+
return `the \`${bin}\` CLI is not on the eval user's PATH inside the isolated hop (${err.slice(0, 200)}). `
|
|
362
|
+
+ 'Install and log it in as that user (RUNBOOK.md), or pass --trusted-skill for a skill you authored.';
|
|
363
|
+
}
|
|
364
|
+
const sudoLine = err.split('\n').find((l) => /^sudo:/.test(l));
|
|
365
|
+
if (sudoLine) {
|
|
366
|
+
return `the isolated eval-user hop failed before reaching \`${bin}\`: ${sudoLine}. The eval user and its NOPASSWD sudoers `
|
|
367
|
+
+ 'rule are an operator prerequisite (RUNBOOK.md); pass --trusted-skill only for a skill you authored.';
|
|
368
|
+
}
|
|
369
|
+
return null;
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
373
|
+
async function completeClaudeCli({ system, prompt, model, timeoutMs, trusted = false }) {
|
|
374
|
+
// v0.4: `--output-format json` returns ONE JSON object carrying both the
|
|
375
|
+
// final text (`result`) and the token usage (`usage`) — the default text
|
|
376
|
+
// output carries no usage at all, which is why usage was previously null on
|
|
377
|
+
// this lane. The text is read from the parsed object; if the CLI ever emits
|
|
378
|
+
// something unparseable we fall back to the raw stdout so a run degrades to
|
|
379
|
+
// the old behaviour (text, no usage) rather than failing.
|
|
380
|
+
const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
|
|
381
|
+
if (system) args.push('--append-system-prompt', system);
|
|
382
|
+
const plan = buildSpawnPlan({ bin: 'claude', args, trusted });
|
|
383
|
+
let r;
|
|
384
|
+
try {
|
|
385
|
+
r = await runPlan(plan, { input: prompt, timeoutMs });
|
|
386
|
+
} catch (e) {
|
|
387
|
+
if (e.code === 'ENOENT') {
|
|
388
|
+
throw new Error(plan.mode === 'isolated'
|
|
389
|
+
? `the isolated hop requires ${SUDO_BIN}`
|
|
390
|
+
: "the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api.");
|
|
391
|
+
}
|
|
392
|
+
throw e;
|
|
393
|
+
}
|
|
394
|
+
if (r.code !== 0) {
|
|
395
|
+
const hop = plan.mode === 'isolated' ? hopFailure('claude', r) : null;
|
|
396
|
+
throw new Error(hop || `claude CLI exited ${r.code}: ${r.stderr.slice(0, 400)}`);
|
|
397
|
+
}
|
|
398
|
+
const parsed = parseClaudeCliJson(r.stdout);
|
|
399
|
+
if (!parsed) return { text: r.stdout.trim(), usage: null, stopReason: null, reportedModels: null, isolation: isolationOf(plan) };
|
|
400
|
+
if (parsed.isError) throw new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`);
|
|
401
|
+
// v0.6: the CLI's own stop_reason and the modelUsage map's canonical ids.
|
|
402
|
+
return { text: String(parsed.text || '').trim(), usage: parsed.usage, stopReason: parsed.stopReason, reportedModels: parsed.reportedModels, isolation: isolationOf(plan) };
|
|
403
|
+
}
|
|
404
|
+
|
|
244
405
|
// ── openai/api (Chat Completions-compatible) ───────────────────────────────────
|
|
245
406
|
// Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
|
|
246
407
|
// is configurable — env OPENAI_BASE_URL wins, else the registry's provider
|
|
@@ -278,7 +439,8 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
|
|
|
278
439
|
let parsed;
|
|
279
440
|
try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
|
|
280
441
|
const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
|
|
281
|
-
|
|
442
|
+
// v0.6: the response's model and the first choice's finish_reason.
|
|
443
|
+
return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage), ...readOpenaiApiResponse(parsed), isolation: 'none' };
|
|
282
444
|
}
|
|
283
445
|
|
|
284
446
|
// Read the OpenAI provider config (base_url + api-key env) from the registry,
|
|
@@ -294,7 +456,7 @@ function openaiProviderConfig() {
|
|
|
294
456
|
|
|
295
457
|
// ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
|
|
296
458
|
// Invocation template (from the verified recon, reference_codex_cli):
|
|
297
|
-
// codex exec --json -s read-only --skip-git-repo-check --ephemeral
|
|
459
|
+
// codex exec --json -s read-only --skip-git-repo-check --ephemeral [-m <model>] -
|
|
298
460
|
// Honoured facts:
|
|
299
461
|
// - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
|
|
300
462
|
// exec already defaults to approval_policy "never".
|
|
@@ -305,6 +467,12 @@ function openaiProviderConfig() {
|
|
|
305
467
|
// - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
|
|
306
468
|
// with_skill mode) is folded into the prompt text.
|
|
307
469
|
// - stderr emits a harmless "Reading additional input from stdin…" notice.
|
|
470
|
+
// - `-o/--output-last-message` is NEVER passed (spec 022 AC-9, approval F-1).
|
|
471
|
+
// The final message is taken from the `--json` stream on the stdout pipe
|
|
472
|
+
// this process owns. A `-o` path is a file the child writes and the parent
|
|
473
|
+
// reads back; isolated, the child is the eval user running third-party
|
|
474
|
+
// instructions, and a symlink planted at that path would have made this
|
|
475
|
+
// process read an operator-owned file into the generation.
|
|
308
476
|
const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
|
|
309
477
|
|
|
310
478
|
function codexAuthPresent(homeDir) {
|
|
@@ -315,9 +483,9 @@ function codexAuthPresent(homeDir) {
|
|
|
315
483
|
}
|
|
316
484
|
|
|
317
485
|
// Build the exact argv for a codex exec call. Exposed for the gate to assert the
|
|
318
|
-
// template (
|
|
319
|
-
function buildCodexArgs({ model
|
|
320
|
-
const args = [...CODEX_EXEC_ARGS
|
|
486
|
+
// template (that `-a` never appears, and neither does `-o`).
|
|
487
|
+
function buildCodexArgs({ model } = {}) {
|
|
488
|
+
const args = [...CODEX_EXEC_ARGS];
|
|
321
489
|
if (model) args.push('-m', resolveModel(model));
|
|
322
490
|
return args;
|
|
323
491
|
}
|
|
@@ -326,66 +494,60 @@ function buildCodexArgs({ model, outFile }) {
|
|
|
326
494
|
// means "read the prompt from stdin". The prompt is NEVER a positional argv —
|
|
327
495
|
// real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
|
|
328
496
|
// positional beginning with `--`. Exposed so the gate can lock this in.
|
|
329
|
-
function codexFinalArgs({ model
|
|
330
|
-
return [...buildCodexArgs({ model
|
|
497
|
+
function codexFinalArgs({ model } = {}) {
|
|
498
|
+
return [...buildCodexArgs({ model }), '-'];
|
|
331
499
|
}
|
|
332
500
|
|
|
333
|
-
function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
501
|
+
async function completeCodexCli({ system, prompt, model, timeoutMs, trusted = false }) {
|
|
502
|
+
// The auth presence check is a PARENT-side read of ~/.codex/auth.json, which
|
|
503
|
+
// only means something on the same-user path; the eval user's home is not
|
|
504
|
+
// readable from here, and a missing login there surfaces as codex's own error.
|
|
505
|
+
if (trusted && !codexAuthPresent()) {
|
|
506
|
+
throw new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).');
|
|
507
|
+
}
|
|
508
|
+
// codex exec has no system-prompt flag → fold the system prompt into the
|
|
509
|
+
// prompt text (this is how the SKILL.md reaches the model in with_skill mode).
|
|
510
|
+
const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
|
|
511
|
+
// The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
|
|
512
|
+
// files begin with `---` (YAML frontmatter), and a positional argument that
|
|
513
|
+
// starts with `--` is rejected by codex's arg parser ("unexpected argument
|
|
514
|
+
// '---'"). `-` as the positional tells `codex exec` to read instructions from
|
|
515
|
+
// stdin (per its --help), which is content-agnostic.
|
|
516
|
+
const plan = buildSpawnPlan({ bin: 'codex', args: codexFinalArgs({ model }), trusted });
|
|
517
|
+
// stdout is CAPTURED. `--json` streams JSONL events there: the last
|
|
518
|
+
// `item.completed`/`agent_message` carries the final text and the terminal
|
|
519
|
+
// `turn.completed` event carries this call's token usage — the only place
|
|
520
|
+
// codex reports it. Nothing is read from the filesystem after the call, on
|
|
521
|
+
// either mode: the stdout pipe is the one channel only the child we spawned
|
|
522
|
+
// can write to, and the eval user cannot make this process read a file
|
|
523
|
+
// through it (spec 022 AC-9).
|
|
524
|
+
let r;
|
|
525
|
+
try {
|
|
526
|
+
r = await runPlan(plan, { input: fullPrompt, timeoutMs });
|
|
527
|
+
} catch (e) {
|
|
528
|
+
if (e.code === 'ENOENT') {
|
|
529
|
+
throw new Error(plan.mode === 'isolated'
|
|
530
|
+
? `the isolated hop requires ${SUDO_BIN}`
|
|
531
|
+
: 'the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).');
|
|
337
532
|
}
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
const
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
// v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
|
|
350
|
-
// there, and the terminal `turn.completed` event carries this call's token
|
|
351
|
-
// usage — the only place codex reports it. The final message still comes from
|
|
352
|
-
// the -o file (cleaner than scraping the stream); the JSONL is read for usage.
|
|
353
|
-
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
354
|
-
let err = '';
|
|
355
|
-
let jsonl = '';
|
|
356
|
-
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
357
|
-
child.stdout.on('data', (d) => { jsonl += d; });
|
|
358
|
-
child.stderr.on('data', (d) => { err += d; });
|
|
359
|
-
child.on('error', (e) => {
|
|
360
|
-
clearTimeout(killer);
|
|
361
|
-
if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
|
|
362
|
-
else reject(e);
|
|
363
|
-
});
|
|
364
|
-
child.on('close', (code) => {
|
|
365
|
-
clearTimeout(killer);
|
|
366
|
-
let text = '';
|
|
367
|
-
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
368
|
-
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
369
|
-
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
370
|
-
const ev = parseCodexJsonl(jsonl);
|
|
371
|
-
// Prefer the -o file; fall back to the stream's agent_message if it is empty.
|
|
372
|
-
const finalText = String(text || ev.text || '').trim();
|
|
373
|
-
resolve({ text: finalText, usage: ev.usage });
|
|
374
|
-
});
|
|
375
|
-
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
376
|
-
child.stdin.write(fullPrompt);
|
|
377
|
-
child.stdin.end();
|
|
378
|
-
});
|
|
533
|
+
throw e;
|
|
534
|
+
}
|
|
535
|
+
if (r.code !== 0) {
|
|
536
|
+
const hop = plan.mode === 'isolated' ? hopFailure('codex', r) : null;
|
|
537
|
+
throw new Error(hop || `codex exec exited ${r.code}: ${String(r.stderr).slice(0, 400)}`);
|
|
538
|
+
}
|
|
539
|
+
const ev = parseCodexJsonl(r.stdout);
|
|
540
|
+
// codex reports neither the serving model nor a stop reason on its stream;
|
|
541
|
+
// both read null and the receipt says the surface did not report them.
|
|
542
|
+
return { text: String(ev.text || '').trim(), usage: ev.usage, stopReason: ev.stopReason, reportedModels: ev.reportedModels, isolation: isolationOf(plan) };
|
|
379
543
|
}
|
|
380
544
|
|
|
381
|
-
// A monotonically-increasing counter for temp-file uniqueness that does not use
|
|
382
|
-
// Math.random (kept deterministic-friendly for any harness that forbids it).
|
|
383
|
-
let _codexCounter = 0;
|
|
384
|
-
function codexCounter() { _codexCounter += 1; return _codexCounter; }
|
|
385
|
-
|
|
386
545
|
module.exports = {
|
|
387
546
|
complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
|
|
388
547
|
inferProvider, surfaceForModel, providerForSurface,
|
|
389
548
|
isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
|
|
390
549
|
CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
|
|
550
|
+
// spec 022 — the eval-user hop
|
|
551
|
+
SUDO_BIN, EVAL_USER_DEFAULT, ISOLATED_ENV_ALLOWLIST, ISOLATED_WRAPPER, HOP_LABEL,
|
|
552
|
+
evalUser, isolatedEnv, buildSpawnPlan, runIsolated, isolationOf,
|
|
391
553
|
};
|
package/lib/receipt.js
CHANGED
|
@@ -20,9 +20,30 @@ const SCHEMA_FILES = {
|
|
|
20
20
|
// published v0.4 receipt asserts conformance by NUMBER, and the number has to
|
|
21
21
|
// keep resolving to the schema it meant (AC-10).
|
|
22
22
|
'0.4': 'receipt.v0.4.schema.json',
|
|
23
|
-
|
|
23
|
+
// v0.5 moved the same way when v0.6 took the current pointer (spec 026): the
|
|
24
|
+
// six report-008 receipts assert v0.5 by number, and v0.6 is not additive for
|
|
25
|
+
// the validator (it requires the receipt to say what answered it), so the
|
|
26
|
+
// number has to keep resolving to the schema it meant.
|
|
27
|
+
'0.5': 'receipt.v0.5.schema.json',
|
|
28
|
+
'0.6': 'receipt.schema.json',
|
|
24
29
|
};
|
|
25
30
|
|
|
31
|
+
// THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
|
|
32
|
+
// 026 AC-6) so a reader recomputes the summary from the rows by the formula
|
|
33
|
+
// it names rather than by guessing which of the two "bands" a receipt carries.
|
|
34
|
+
// Per case, stddev is the spread of judge samples; per arm, it is this.
|
|
35
|
+
const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases';
|
|
36
|
+
|
|
37
|
+
// THE ONE PREDICATE for "this case did not complete" (spec 026 AC-3). Every
|
|
38
|
+
// reader in lib/ and scripts/ asks this, never the failed_timeout literal: a
|
|
39
|
+
// reader that excluded by that literal admitted a failed_unmeasured case (an
|
|
40
|
+
// empty generation, a judge with no score) into its statistics as if it had
|
|
41
|
+
// been measured. A case is failed when its case_status is present and is not
|
|
42
|
+
// 'ok'; the status names what was observed, and both failure statuses are
|
|
43
|
+
// recorded without fabricated samples or hashes.
|
|
44
|
+
const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
|
|
45
|
+
function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
|
|
46
|
+
|
|
26
47
|
const _validators = {};
|
|
27
48
|
// Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
|
|
28
49
|
// so the library can be required without ajv present (pure hashing utilities).
|
|
@@ -80,6 +101,27 @@ function aggregate(caseResults) {
|
|
|
80
101
|
return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
|
|
81
102
|
}
|
|
82
103
|
|
|
104
|
+
// The comparison block from the two arms' aggregates. Every null it carries is
|
|
105
|
+
// named: `delta_uncertainty_unavailable` is `no_cases` when an arm has no
|
|
106
|
+
// included case (then every score is null too: the mean of nothing is not a
|
|
107
|
+
// number) and `single_case` when an arm has one (a mean, no band). Otherwise
|
|
108
|
+
// the block is numeric and the field is absent (spec 026 AC-7).
|
|
109
|
+
function comparisonOf(aggWith, aggBase) {
|
|
110
|
+
const noCases = aggWith.case_count === 0 || aggBase.case_count === 0;
|
|
111
|
+
const single = !noCases && (aggWith.case_count < 2 || aggBase.case_count < 2);
|
|
112
|
+
const cmp = {
|
|
113
|
+
with_skill_score: aggWith.mean_score,
|
|
114
|
+
baseline_score: aggBase.mean_score,
|
|
115
|
+
delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
|
|
116
|
+
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
117
|
+
// bands. Diff uses this for the headline "within noise" vs real-move rule.
|
|
118
|
+
delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
119
|
+
};
|
|
120
|
+
if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
|
|
121
|
+
else if (single) cmp.delta_uncertainty_unavailable = 'single_case';
|
|
122
|
+
return cmp;
|
|
123
|
+
}
|
|
124
|
+
|
|
83
125
|
// Assemble a full receipt from the runner's raw pieces, seal it, and return it.
|
|
84
126
|
// skill: { name, version, contentHash }
|
|
85
127
|
// suite: { format, suiteHash, caseCount }
|
|
@@ -109,7 +151,7 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
109
151
|
// The excluded case STAYS in `results.cases`. It is removed from the mean, not
|
|
110
152
|
// from the record — deleting the evidence of a failure is a different and worse
|
|
111
153
|
// defect than averaging over it.
|
|
112
|
-
const armUnusable = (c) => c
|
|
154
|
+
const armUnusable = (c) => caseFailed(c)
|
|
113
155
|
|| (c.mean == null && c.score == null);
|
|
114
156
|
const excludedIds = new Map();
|
|
115
157
|
for (const c of cases) {
|
|
@@ -183,25 +225,23 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
183
225
|
registry: run.registry || 'unregistered',
|
|
184
226
|
transcripts: run.transcripts || 'hashes-only',
|
|
185
227
|
judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
|
|
228
|
+
// v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
|
|
229
|
+
// given and never defaulted: a receipt that does not say what answered it
|
|
230
|
+
// is refused by the schema, which is the point.
|
|
231
|
+
...(run.answered_by ? { answered_by: run.answered_by } : {}),
|
|
186
232
|
},
|
|
187
233
|
results: {
|
|
188
234
|
cases,
|
|
189
235
|
aggregates: {
|
|
190
236
|
with_skill: aggWith,
|
|
191
237
|
baseline: aggBase,
|
|
238
|
+
band_rule: BAND_RULE,
|
|
192
239
|
// Present only when something was excluded, so a clean run's receipt is
|
|
193
240
|
// unchanged and the archive does not acquire an empty field.
|
|
194
241
|
...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
|
|
195
242
|
},
|
|
196
243
|
},
|
|
197
|
-
comparison:
|
|
198
|
-
with_skill_score: aggWith.mean_score,
|
|
199
|
-
baseline_score: aggBase.mean_score,
|
|
200
|
-
delta: round(aggWith.mean_score - aggBase.mean_score),
|
|
201
|
-
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
202
|
-
// bands. Diff uses this for the headline "within noise" vs real-move rule.
|
|
203
|
-
delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
204
|
-
},
|
|
244
|
+
comparison: comparisonOf(aggWith, aggBase),
|
|
205
245
|
verification_level: verificationLevel,
|
|
206
246
|
receipt_hash: '',
|
|
207
247
|
};
|
|
@@ -226,5 +266,5 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
|
|
|
226
266
|
}
|
|
227
267
|
|
|
228
268
|
module.exports = {
|
|
229
|
-
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
|
|
269
|
+
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
|
|
230
270
|
};
|