driftproof 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/provider.js CHANGED
@@ -4,9 +4,12 @@
4
4
  const fs = require('fs');
5
5
  const os = require('os');
6
6
  const path = require('path');
7
- const { spawn } = require('child_process');
7
+ // Called through the module object (cp.spawn), never destructured: the gate
8
+ // records spawns by replacing that property, so nothing real runs offline.
9
+ const cp = require('child_process');
8
10
  const { withRetry, withTimeout } = require('./json');
9
11
  const { stubComplete, stubEnabled } = require('./stub');
12
+ const { readAnthropicApiResponse, readOpenaiApiResponse } = require('./usage');
10
13
  const {
11
14
  parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage, normalizeUsage,
12
15
  } = require('./usage');
@@ -126,15 +129,23 @@ const CODEX_OVERHEAD_NOTE =
126
129
  //
127
130
  // `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
128
131
  // themselves, so temperature is ignored there and the receipt records that fact.
129
- async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined }) {
132
+ // `trusted` (spec 022) selects the SAME-USER legacy spawn on the two CLI lanes.
133
+ // It is false unless a caller says otherwise; bin/driftproof exposes it as
134
+ // --trusted-skill, for skills the operator authored. Api lanes ignore it.
135
+ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined, trusted = false }) {
130
136
  const surface = surfaceForModel(model);
131
137
  const provider = providerForSurface(surface);
132
- // Offline stub surface: canned completion, zero model calls. The receipt still
133
- // records the real surface/provider so a stub run is not mistaken for a genuine
134
- // one at read time — only the generation/judge TEXT is canned.
138
+ // Offline stub surface: canned completion, zero model calls. The reply says
139
+ // so: surface `stub`, answeredBy `stub`, no model reported, no stop reason,
140
+ // no isolation (nothing was spawned). Spec 026 F1: the receipt used to record
141
+ // the REAL surface name here and read TESTED while the text was canned; now
142
+ // the receipt's surface, level and answered_by are derived from what this
143
+ // function returned, and a stub run reads UNVERIFIED with surface stub. The
144
+ // requested model's provider is still reported, so a reader knows what was
145
+ // asked for.
135
146
  if (stubEnabled()) {
136
147
  const s = stubComplete({ system, prompt });
137
- return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface, provider, attempts: 1 };
148
+ return { ...s, usage: normalizeUsage(s.usage), wall_ms: (s.usage && s.usage.wall_ms) || 0, surface: 'stub', provider, attempts: 1, answeredBy: 'stub', reportedModels: null, stopReason: null, isolation: 'none' };
138
149
  }
139
150
 
140
151
  // Per-surface tolerance (see retryPolicyForSurface): a cli/subscription surface
@@ -149,9 +160,9 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
149
160
  const laneRunner = () => {
150
161
  switch (surface) {
151
162
  case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
152
- case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
163
+ case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
153
164
  case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
154
- case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
165
+ case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
155
166
  default: throw new Error(`unknown surface: ${surface}`);
156
167
  }
157
168
  };
@@ -172,7 +183,16 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
172
183
  try {
173
184
  const out = await withRetry(runner, { tries: policy.tries, baseDelayMs });
174
185
  const usage = normalizeUsage({ ...(out.usage || {}), wall_ms: wallMs });
175
- return { ...out, usage, wall_ms: wallMs, surface, provider, attempts };
186
+ // v0.6 (spec 026 AC-1, AC-2, AC-8): a model surface answered; what it said
187
+ // served the call, why the reply stopped, and which spawn path was taken
188
+ // are what the lane could read, null where its surface reports nothing.
189
+ return {
190
+ ...out, usage, wall_ms: wallMs, surface, provider, attempts,
191
+ answeredBy: 'model',
192
+ reportedModels: Array.isArray(out.reportedModels) && out.reportedModels.length ? out.reportedModels : null,
193
+ stopReason: typeof out.stopReason === 'string' && out.stopReason ? out.stopReason : null,
194
+ isolation: out.isolation || 'none',
195
+ };
176
196
  } catch (e) {
177
197
  // Surface the attempt count so a persistently-failing call can be charged for
178
198
  // (and, for a timeout, marked failed_timeout by the runner instead of fatal).
@@ -197,50 +217,191 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
197
217
  if (temperature !== undefined) params.temperature = temperature;
198
218
  const resp = await client.messages.create(params);
199
219
  const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
200
- return { text, usage: parseAnthropicApiUsage(resp.usage) };
220
+ // v0.6: the response names the model that served it and why it stopped.
221
+ return { text, usage: parseAnthropicApiUsage(resp.usage), ...readAnthropicApiResponse(resp), isolation: 'none' };
201
222
  }
202
223
 
203
- // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
204
- function completeClaudeCli({ system, prompt, model, timeoutMs }) {
205
- return new Promise((resolve, reject) => {
206
- // Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
207
- // metered API key. Everything else in the env is preserved.
224
+ // ── isolation: the eval-user hop (spec 022) ───────────────────────────────────
225
+ //
226
+ // Every local-CLI spawn runs as a DEDICATED UNPRIVILEGED UNIX USER by default.
227
+ // The third-party SKILL.md a run evaluates is instructions to an agent with
228
+ // tools, and until spec 022 that agent ran as the operator: same uid, whole
229
+ // environment block, operator's cwd, operator's CLI configuration. The
230
+ // 2026-09-03 security audit (findings A2, N5) confirmed a live API token in the
231
+ // inherited environment and this host's SSH keys and CLI session tokens
232
+ // readable by the same uid; this was then reproduced by execution.
233
+ //
234
+ // The boundary is a uid change reached through sudo from an EMPTY environment:
235
+ //
236
+ // /usr/bin/sudo -n -u <eval-user> /usr/bin/env -i HOME=/home/<eval-user>
237
+ // PATH=/home/<eval-user>/.local/bin:/usr/bin:/bin bash -lc '<wrapper>'
238
+ // driftproof-eval-hop <cli> <cli-args...>
239
+ //
240
+ // Operator prerequisite, NOT created here (RUNBOOK.md): the user exists with a
241
+ // mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
242
+ // under ~/.local/bin, and a sudoers rule grants the operator NOPASSWD as that
243
+ // user. The runner refuses loudly when any of that is missing; it never creates
244
+ // a user, edits sudoers, or logs a CLI in.
245
+ //
246
+ // The same-user (legacy) spawn survives ONLY behind `trusted: true`, exposed by
247
+ // bin/driftproof as --trusted-skill, for skills the operator authored.
248
+ const SUDO_BIN = '/usr/bin/sudo';
249
+ const ENV_BIN = '/usr/bin/env';
250
+ const EVAL_USER_DEFAULT = 'driftproof-eval';
251
+ // A user name that can safely follow `sudo -u`: no leading dash, no shell
252
+ // metacharacter, no path separator. Anything else is refused before any spawn.
253
+ const EVAL_USER_RE = /^[a-z_][a-z0-9_-]{0,31}$/;
254
+ // The ONLY names the child environment carries, each CONSTRUCTED from the eval
255
+ // user's name in isolatedEnv() and never copied from process.env. Verified
256
+ // 2026-09-03: both CLIs run through the hop with these two alone; the login
257
+ // shell supplies LANG and TERM itself. Data, frozen, asserted by the gate.
258
+ const ISOLATED_ENV_ALLOWLIST = Object.freeze(['HOME', 'PATH']);
259
+ // $0 of the wrapper, so `ps` shows what the bash process is.
260
+ const HOP_LABEL = 'driftproof-eval-hop';
261
+ // The wrapper runs INSIDE the hop, as the eval user. It takes the CLI as
262
+ // positional parameters ("$@"), so no SKILL.md text is ever shell-interpreted.
263
+ // - the cwd is a fresh per-call directory under the system tmp, created by
264
+ // the eval user (one the parent creates is mode 700 to it);
265
+ // - the CLI runs under setsid in its own process group, stdin preserved
266
+ // (`<&0` defeats the /dev/null a non-interactive bash gives a background job);
267
+ // - TERM/INT/HUP, which sudo relays from the parent on timeout, kill that
268
+ // whole group, remove the directory and exit 143;
269
+ // - a normal exit removes the directory and returns the CLI's status.
270
+ const ISOLATED_WRAPPER = [
271
+ 'd=$(mktemp -d -t driftproof-eval.XXXXXXXX) || exit 97',
272
+ 'cd "$d" || exit 97',
273
+ 'setsid -w "$@" <&0 & pid=$!',
274
+ 'trap \'kill -TERM -- "-$pid" 2>/dev/null; sleep 1; kill -KILL -- "-$pid" 2>/dev/null; cd /; rm -rf "$d"; exit 143\' TERM INT HUP',
275
+ 'wait "$pid"; r=$?',
276
+ 'cd /; rm -rf "$d"; exit $r',
277
+ ].join('; ');
278
+
279
+ // DRIFTPROOF_EVAL_USER selects WHICH user isolates, never WHETHER isolation
280
+ // happens (that is the trusted flag, an argv decision). Empty means default.
281
+ function evalUser() {
282
+ const raw = process.env.DRIFTPROOF_EVAL_USER;
283
+ const u = raw === undefined || raw === '' ? EVAL_USER_DEFAULT : raw;
284
+ if (!EVAL_USER_RE.test(u)) throw new Error(`DRIFTPROOF_EVAL_USER is not a plausible unix user name: ${JSON.stringify(u)}`);
285
+ return u;
286
+ }
287
+
288
+ // The child environment: constructed, not copied. Keys are exactly the allowlist.
289
+ function isolatedEnv(user) {
290
+ const home = `/home/${user}`;
291
+ return { HOME: home, PATH: `${home}/.local/bin:/usr/bin:/bin` };
292
+ }
293
+
294
+ // The isolation a plan records into the receipt (run.answered_by.isolation,
295
+ // spec 026 AC-2): the hop is `eval-user`, the legacy spawn `same-user`.
296
+ function isolationOf(plan) { return plan && plan.mode === 'isolated' ? 'eval-user' : 'same-user'; }
297
+
298
+ // The spawn plan for one CLI call: { mode, user, file, args, env }. Pure, so the
299
+ // gate can inspect what WOULD be spawned. Untrusted → the hop; trusted → the
300
+ // legacy same-user spawn exactly as it was (env inherited; the claude lane still
301
+ // strips ANTHROPIC_API_KEY so the CLI uses the subscription, not the metered key).
302
+ function buildSpawnPlan({ bin, args, trusted = false }) {
303
+ if (trusted) {
208
304
  const env = { ...process.env };
209
- delete env.ANTHROPIC_API_KEY;
210
-
211
- // v0.4: `--output-format json` returns ONE JSON object carrying both the
212
- // final text (`result`) and the token usage (`usage`) — the default text
213
- // output carries no usage at all, which is why usage was previously null on
214
- // this lane. The text is read from the parsed object; if the CLI ever emits
215
- // something unparseable we fall back to the raw stdout so a run degrades to
216
- // the old behaviour (text, no usage) rather than failing.
217
- const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
218
- if (system) args.push('--append-system-prompt', system);
219
-
220
- const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
221
- let out = '';
222
- let err = '';
223
- const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
224
- child.stdout.on('data', (d) => { out += d; });
225
- child.stderr.on('data', (d) => { err += d; });
226
- child.on('error', (e) => {
227
- clearTimeout(killer);
228
- if (e.code === 'ENOENT') reject(new Error("the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
229
- else reject(e);
230
- });
231
- child.on('close', (code) => {
232
- clearTimeout(killer);
233
- if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
234
- const parsed = parseClaudeCliJson(out);
235
- if (!parsed) return resolve({ text: out.trim(), usage: null });
236
- if (parsed.isError) return reject(new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`));
237
- resolve({ text: String(parsed.text || '').trim(), usage: parsed.usage });
238
- });
239
- child.stdin.write(prompt);
305
+ if (bin === 'claude') delete env.ANTHROPIC_API_KEY;
306
+ return { mode: 'same-user', user: null, file: bin, args: [...args], env };
307
+ }
308
+ const user = evalUser();
309
+ const childEnv = isolatedEnv(user);
310
+ return {
311
+ mode: 'isolated',
312
+ user,
313
+ file: SUDO_BIN,
314
+ args: ['-n', '-u', user, ENV_BIN, '-i', ...ISOLATED_ENV_ALLOWLIST.map((k) => `${k}=${childEnv[k]}`), 'bash', '-lc', ISOLATED_WRAPPER, HOP_LABEL, bin, ...args],
315
+ // The sudo front process gets NOTHING from this process. sudo resolves
316
+ // /usr/bin/env itself; env -i then starts the child from empty.
317
+ env: {},
318
+ };
319
+ }
320
+
321
+ // Run a plan: `input` to stdin, stdout/stderr collected, settled on close.
322
+ // The timeout signal differs by mode. The legacy child gets SIGKILL as it
323
+ // always did. The hop gets SIGTERM: it is the signal the operator's uid can
324
+ // deliver to a root-owned sudo, and the one sudo relays to the wrapper's trap.
325
+ // SIGKILL would neither be permitted nor relayed, and a wrapper without the
326
+ // trap left the CLI orphaned with the stdout pipe open, so close never fired.
327
+ function runPlan(plan, { input = null, timeoutMs = 300000, graceMs = 5000 } = {}) {
328
+ return new Promise((resolve, reject) => {
329
+ let child;
330
+ try { child = cp.spawn(plan.file, plan.args, { env: plan.env, stdio: ['pipe', 'pipe', 'pipe'] }); }
331
+ catch (e) { return reject(e); }
332
+ let stdout = '';
333
+ let stderr = '';
334
+ let timedOut = false;
335
+ const killer = setTimeout(() => {
336
+ timedOut = true;
337
+ child.kill(plan.mode === 'isolated' ? 'SIGTERM' : 'SIGKILL');
338
+ }, timeoutMs + graceMs);
339
+ child.stdout.on('data', (d) => { stdout += d; });
340
+ child.stderr.on('data', (d) => { stderr += d; });
341
+ child.on('error', (e) => { clearTimeout(killer); reject(e); });
342
+ child.on('close', (code, signal) => { clearTimeout(killer); resolve({ code, signal, stdout, stderr, timedOut }); });
343
+ child.stdin.on('error', () => { /* EPIPE if the CLI exits before reading all of stdin */ });
344
+ if (input != null) child.stdin.write(input);
240
345
  child.stdin.end();
241
346
  });
242
347
  }
243
348
 
349
+ // The hop with an arbitrary command in place of the CLI, for the gate: the same
350
+ // plan builder and the same runner, never trusted. No provider is reached.
351
+ function runIsolated(cmdArgv, opts = {}) {
352
+ const [bin, ...args] = cmdArgv;
353
+ return runPlan(buildSpawnPlan({ bin, args, trusted: false }), { graceMs: 0, ...opts });
354
+ }
355
+
356
+ // A hop failure, said in terms of what to fix. Returns null when the failure is
357
+ // the CLI's own (exit status from inside the hop), which the lane reports as before.
358
+ function hopFailure(bin, r) {
359
+ const err = String(r.stderr || '').trim();
360
+ if (r.code === 127) {
361
+ return `the \`${bin}\` CLI is not on the eval user's PATH inside the isolated hop (${err.slice(0, 200)}). `
362
+ + 'Install and log it in as that user (RUNBOOK.md), or pass --trusted-skill for a skill you authored.';
363
+ }
364
+ const sudoLine = err.split('\n').find((l) => /^sudo:/.test(l));
365
+ if (sudoLine) {
366
+ return `the isolated eval-user hop failed before reaching \`${bin}\`: ${sudoLine}. The eval user and its NOPASSWD sudoers `
367
+ + 'rule are an operator prerequisite (RUNBOOK.md); pass --trusted-skill only for a skill you authored.';
368
+ }
369
+ return null;
370
+ }
371
+
372
+ // ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
373
+ async function completeClaudeCli({ system, prompt, model, timeoutMs, trusted = false }) {
374
+ // v0.4: `--output-format json` returns ONE JSON object carrying both the
375
+ // final text (`result`) and the token usage (`usage`) — the default text
376
+ // output carries no usage at all, which is why usage was previously null on
377
+ // this lane. The text is read from the parsed object; if the CLI ever emits
378
+ // something unparseable we fall back to the raw stdout so a run degrades to
379
+ // the old behaviour (text, no usage) rather than failing.
380
+ const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
381
+ if (system) args.push('--append-system-prompt', system);
382
+ const plan = buildSpawnPlan({ bin: 'claude', args, trusted });
383
+ let r;
384
+ try {
385
+ r = await runPlan(plan, { input: prompt, timeoutMs });
386
+ } catch (e) {
387
+ if (e.code === 'ENOENT') {
388
+ throw new Error(plan.mode === 'isolated'
389
+ ? `the isolated hop requires ${SUDO_BIN}`
390
+ : "the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api.");
391
+ }
392
+ throw e;
393
+ }
394
+ if (r.code !== 0) {
395
+ const hop = plan.mode === 'isolated' ? hopFailure('claude', r) : null;
396
+ throw new Error(hop || `claude CLI exited ${r.code}: ${r.stderr.slice(0, 400)}`);
397
+ }
398
+ const parsed = parseClaudeCliJson(r.stdout);
399
+ if (!parsed) return { text: r.stdout.trim(), usage: null, stopReason: null, reportedModels: null, isolation: isolationOf(plan) };
400
+ if (parsed.isError) throw new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`);
401
+ // v0.6: the CLI's own stop_reason and the modelUsage map's canonical ids.
402
+ return { text: String(parsed.text || '').trim(), usage: parsed.usage, stopReason: parsed.stopReason, reportedModels: parsed.reportedModels, isolation: isolationOf(plan) };
403
+ }
404
+
244
405
  // ── openai/api (Chat Completions-compatible) ───────────────────────────────────
245
406
  // Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
246
407
  // is configurable — env OPENAI_BASE_URL wins, else the registry's provider
@@ -278,7 +439,8 @@ async function completeOpenaiApi({ system, prompt, model, maxTokens, temperature
278
439
  let parsed;
279
440
  try { parsed = JSON.parse(raw); } catch (_e) { throw new Error(`openai/api: unparseable response: ${raw.slice(0, 200)}`); }
280
441
  const text = ((parsed.choices && parsed.choices[0] && parsed.choices[0].message && parsed.choices[0].message.content) || '');
281
- return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage) };
442
+ // v0.6: the response's model and the first choice's finish_reason.
443
+ return { text: String(text).trim(), usage: parseOpenaiApiUsage(parsed.usage), ...readOpenaiApiResponse(parsed), isolation: 'none' };
282
444
  }
283
445
 
284
446
  // Read the OpenAI provider config (base_url + api-key env) from the registry,
@@ -294,7 +456,7 @@ function openaiProviderConfig() {
294
456
 
295
457
  // ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
296
458
  // Invocation template (from the verified recon, reference_codex_cli):
297
- // codex exec --json -s read-only --skip-git-repo-check --ephemeral -o <tmp> [-m <model>] "PROMPT"
459
+ // codex exec --json -s read-only --skip-git-repo-check --ephemeral [-m <model>] -
298
460
  // Honoured facts:
299
461
  // - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
300
462
  // exec already defaults to approval_policy "never".
@@ -305,6 +467,12 @@ function openaiProviderConfig() {
305
467
  // - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
306
468
  // with_skill mode) is folded into the prompt text.
307
469
  // - stderr emits a harmless "Reading additional input from stdin…" notice.
470
+ // - `-o/--output-last-message` is NEVER passed (spec 022 AC-9, approval F-1).
471
+ // The final message is taken from the `--json` stream on the stdout pipe
472
+ // this process owns. A `-o` path is a file the child writes and the parent
473
+ // reads back; isolated, the child is the eval user running third-party
474
+ // instructions, and a symlink planted at that path would have made this
475
+ // process read an operator-owned file into the generation.
308
476
  const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
309
477
 
310
478
  function codexAuthPresent(homeDir) {
@@ -315,9 +483,9 @@ function codexAuthPresent(homeDir) {
315
483
  }
316
484
 
317
485
  // Build the exact argv for a codex exec call. Exposed for the gate to assert the
318
- // template (and that `-a` never appears). `outFile` receives the final message.
319
- function buildCodexArgs({ model, outFile }) {
320
- const args = [...CODEX_EXEC_ARGS, '-o', outFile];
486
+ // template (that `-a` never appears, and neither does `-o`).
487
+ function buildCodexArgs({ model } = {}) {
488
+ const args = [...CODEX_EXEC_ARGS];
321
489
  if (model) args.push('-m', resolveModel(model));
322
490
  return args;
323
491
  }
@@ -326,66 +494,60 @@ function buildCodexArgs({ model, outFile }) {
326
494
  // means "read the prompt from stdin". The prompt is NEVER a positional argv —
327
495
  // real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
328
496
  // positional beginning with `--`. Exposed so the gate can lock this in.
329
- function codexFinalArgs({ model, outFile }) {
330
- return [...buildCodexArgs({ model, outFile }), '-'];
497
+ function codexFinalArgs({ model } = {}) {
498
+ return [...buildCodexArgs({ model }), '-'];
331
499
  }
332
500
 
333
- function completeCodexCli({ system, prompt, model, timeoutMs }) {
334
- return new Promise((resolve, reject) => {
335
- if (!codexAuthPresent()) {
336
- return reject(new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).'));
501
+ async function completeCodexCli({ system, prompt, model, timeoutMs, trusted = false }) {
502
+ // The auth presence check is a PARENT-side read of ~/.codex/auth.json, which
503
+ // only means something on the same-user path; the eval user's home is not
504
+ // readable from here, and a missing login there surfaces as codex's own error.
505
+ if (trusted && !codexAuthPresent()) {
506
+ throw new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).');
507
+ }
508
+ // codex exec has no system-prompt flag → fold the system prompt into the
509
+ // prompt text (this is how the SKILL.md reaches the model in with_skill mode).
510
+ const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
511
+ // The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
512
+ // files begin with `---` (YAML frontmatter), and a positional argument that
513
+ // starts with `--` is rejected by codex's arg parser ("unexpected argument
514
+ // '---'"). `-` as the positional tells `codex exec` to read instructions from
515
+ // stdin (per its --help), which is content-agnostic.
516
+ const plan = buildSpawnPlan({ bin: 'codex', args: codexFinalArgs({ model }), trusted });
517
+ // stdout is CAPTURED. `--json` streams JSONL events there: the last
518
+ // `item.completed`/`agent_message` carries the final text and the terminal
519
+ // `turn.completed` event carries this call's token usage — the only place
520
+ // codex reports it. Nothing is read from the filesystem after the call, on
521
+ // either mode: the stdout pipe is the one channel only the child we spawned
522
+ // can write to, and the eval user cannot make this process read a file
523
+ // through it (spec 022 AC-9).
524
+ let r;
525
+ try {
526
+ r = await runPlan(plan, { input: fullPrompt, timeoutMs });
527
+ } catch (e) {
528
+ if (e.code === 'ENOENT') {
529
+ throw new Error(plan.mode === 'isolated'
530
+ ? `the isolated hop requires ${SUDO_BIN}`
531
+ : 'the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).');
337
532
  }
338
- // codex exec has no system-prompt flag → fold the system prompt into the
339
- // prompt text (this is how the SKILL.md reaches the model in with_skill mode).
340
- const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
341
- const outFile = path.join(os.tmpdir(), `driftproof-codex-${process.pid}-${Date.now()}-${Math.floor(codexCounter())}.txt`);
342
- // The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
343
- // files begin with `---` (YAML frontmatter), and a positional argument that
344
- // starts with `--` is rejected by codex's arg parser ("unexpected argument
345
- // '---'"). `-` as the positional tells `codex exec` to read instructions from
346
- // stdin (per its --help), which is content-agnostic.
347
- const args = codexFinalArgs({ model, outFile });
348
-
349
- // v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
350
- // there, and the terminal `turn.completed` event carries this call's token
351
- // usage — the only place codex reports it. The final message still comes from
352
- // the -o file (cleaner than scraping the stream); the JSONL is read for usage.
353
- const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
354
- let err = '';
355
- let jsonl = '';
356
- const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
357
- child.stdout.on('data', (d) => { jsonl += d; });
358
- child.stderr.on('data', (d) => { err += d; });
359
- child.on('error', (e) => {
360
- clearTimeout(killer);
361
- if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
362
- else reject(e);
363
- });
364
- child.on('close', (code) => {
365
- clearTimeout(killer);
366
- let text = '';
367
- try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
368
- try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
369
- if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
370
- const ev = parseCodexJsonl(jsonl);
371
- // Prefer the -o file; fall back to the stream's agent_message if it is empty.
372
- const finalText = String(text || ev.text || '').trim();
373
- resolve({ text: finalText, usage: ev.usage });
374
- });
375
- child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
376
- child.stdin.write(fullPrompt);
377
- child.stdin.end();
378
- });
533
+ throw e;
534
+ }
535
+ if (r.code !== 0) {
536
+ const hop = plan.mode === 'isolated' ? hopFailure('codex', r) : null;
537
+ throw new Error(hop || `codex exec exited ${r.code}: ${String(r.stderr).slice(0, 400)}`);
538
+ }
539
+ const ev = parseCodexJsonl(r.stdout);
540
+ // codex reports neither the serving model nor a stop reason on its stream;
541
+ // both read null and the receipt says the surface did not report them.
542
+ return { text: String(ev.text || '').trim(), usage: ev.usage, stopReason: ev.stopReason, reportedModels: ev.reportedModels, isolation: isolationOf(plan) };
379
543
  }
380
544
 
381
- // A monotonically-increasing counter for temp-file uniqueness that does not use
382
- // Math.random (kept deterministic-friendly for any harness that forbids it).
383
- let _codexCounter = 0;
384
- function codexCounter() { _codexCounter += 1; return _codexCounter; }
385
-
386
545
  module.exports = {
387
546
  complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
388
547
  inferProvider, surfaceForModel, providerForSurface,
389
548
  isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
390
549
  CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
550
+ // spec 022 — the eval-user hop
551
+ SUDO_BIN, EVAL_USER_DEFAULT, ISOLATED_ENV_ALLOWLIST, ISOLATED_WRAPPER, HOP_LABEL,
552
+ evalUser, isolatedEnv, buildSpawnPlan, runIsolated, isolationOf,
391
553
  };
package/lib/receipt.js CHANGED
@@ -20,9 +20,30 @@ const SCHEMA_FILES = {
20
20
  // published v0.4 receipt asserts conformance by NUMBER, and the number has to
21
21
  // keep resolving to the schema it meant (AC-10).
22
22
  '0.4': 'receipt.v0.4.schema.json',
23
- '0.5': 'receipt.schema.json',
23
+ // v0.5 moved the same way when v0.6 took the current pointer (spec 026): the
24
+ // six report-008 receipts assert v0.5 by number, and v0.6 is not additive for
25
+ // the validator (it requires the receipt to say what answered it), so the
26
+ // number has to keep resolving to the schema it meant.
27
+ '0.5': 'receipt.v0.5.schema.json',
28
+ '0.6': 'receipt.schema.json',
24
29
  };
25
30
 
31
+ // THE BAND RULE, stated on every receipt (results.aggregates.band_rule, spec
32
+ // 026 AC-6) so a reader recomputes the summary from the rows by the formula
33
+ // it names rather than by guessing which of the two "bands" a receipt carries.
34
+ // Per case, stddev is the spread of judge samples; per arm, it is this.
35
+ const BAND_RULE = 'per arm: the sample standard deviation (n-1) of the per-case means across the included cases; the comparison band is the quadrature sum of the two arms; null where an arm has fewer than two included cases';
36
+
37
+ // THE ONE PREDICATE for "this case did not complete" (spec 026 AC-3). Every
38
+ // reader in lib/ and scripts/ asks this, never the failed_timeout literal: a
39
+ // reader that excluded by that literal admitted a failed_unmeasured case (an
40
+ // empty generation, a judge with no score) into its statistics as if it had
41
+ // been measured. A case is failed when its case_status is present and is not
42
+ // 'ok'; the status names what was observed, and both failure statuses are
43
+ // recorded without fabricated samples or hashes.
44
+ const FAILED_STATUSES = ['failed_timeout', 'failed_unmeasured'];
45
+ function caseFailed(c) { return !!(c && c.case_status && c.case_status !== 'ok'); }
46
+
26
47
  const _validators = {};
27
48
  // Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
28
49
  // so the library can be required without ajv present (pure hashing utilities).
@@ -80,6 +101,27 @@ function aggregate(caseResults) {
80
101
  return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
81
102
  }
82
103
 
104
+ // The comparison block from the two arms' aggregates. Every null it carries is
105
+ // named: `delta_uncertainty_unavailable` is `no_cases` when an arm has no
106
+ // included case (then every score is null too: the mean of nothing is not a
107
+ // number) and `single_case` when an arm has one (a mean, no band). Otherwise
108
+ // the block is numeric and the field is absent (spec 026 AC-7).
109
+ function comparisonOf(aggWith, aggBase) {
110
+ const noCases = aggWith.case_count === 0 || aggBase.case_count === 0;
111
+ const single = !noCases && (aggWith.case_count < 2 || aggBase.case_count < 2);
112
+ const cmp = {
113
+ with_skill_score: aggWith.mean_score,
114
+ baseline_score: aggBase.mean_score,
115
+ delta: noCases ? null : round(aggWith.mean_score - aggBase.mean_score),
116
+ // Combined uncertainty of the delta: quadrature sum of the two aggregate
117
+ // bands. Diff uses this for the headline "within noise" vs real-move rule.
118
+ delta_uncertainty: noCases ? null : combineUncertainty(aggWith.stddev, aggBase.stddev),
119
+ };
120
+ if (noCases) cmp.delta_uncertainty_unavailable = 'no_cases';
121
+ else if (single) cmp.delta_uncertainty_unavailable = 'single_case';
122
+ return cmp;
123
+ }
124
+
83
125
  // Assemble a full receipt from the runner's raw pieces, seal it, and return it.
84
126
  // skill: { name, version, contentHash }
85
127
  // suite: { format, suiteHash, caseCount }
@@ -109,7 +151,7 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
109
151
  // The excluded case STAYS in `results.cases`. It is removed from the mean, not
110
152
  // from the record — deleting the evidence of a failure is a different and worse
111
153
  // defect than averaging over it.
112
- const armUnusable = (c) => c.case_status === 'failed_timeout'
154
+ const armUnusable = (c) => caseFailed(c)
113
155
  || (c.mean == null && c.score == null);
114
156
  const excludedIds = new Map();
115
157
  for (const c of cases) {
@@ -183,25 +225,23 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
183
225
  registry: run.registry || 'unregistered',
184
226
  transcripts: run.transcripts || 'hashes-only',
185
227
  judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
228
+ // v0.6 (spec 026 AC-1, AC-2): what answered. Carried from the caller as
229
+ // given and never defaulted: a receipt that does not say what answered it
230
+ // is refused by the schema, which is the point.
231
+ ...(run.answered_by ? { answered_by: run.answered_by } : {}),
186
232
  },
187
233
  results: {
188
234
  cases,
189
235
  aggregates: {
190
236
  with_skill: aggWith,
191
237
  baseline: aggBase,
238
+ band_rule: BAND_RULE,
192
239
  // Present only when something was excluded, so a clean run's receipt is
193
240
  // unchanged and the archive does not acquire an empty field.
194
241
  ...(excludedCases.length ? { excluded_cases: excludedCases } : {}),
195
242
  },
196
243
  },
197
- comparison: {
198
- with_skill_score: aggWith.mean_score,
199
- baseline_score: aggBase.mean_score,
200
- delta: round(aggWith.mean_score - aggBase.mean_score),
201
- // Combined uncertainty of the delta: quadrature sum of the two aggregate
202
- // bands. Diff uses this for the headline "within noise" vs real-move rule.
203
- delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
204
- },
244
+ comparison: comparisonOf(aggWith, aggBase),
205
245
  verification_level: verificationLevel,
206
246
  receipt_hash: '',
207
247
  };
@@ -226,5 +266,5 @@ function buildReceipt({ skill, suite, run, cases, economics = null, verification
226
266
  }
227
267
 
228
268
  module.exports = {
229
- buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
269
+ buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt, BAND_RULE, comparisonOf, caseFailed, FAILED_STATUSES,
230
270
  };