driftproof 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -3
- package/bin/driftproof +24 -4
- package/config.js +1 -1
- package/lib/judge.js +6 -4
- package/lib/provider.js +234 -96
- package/lib/run.js +11 -7
- package/lib/verdict.js +32 -6
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -211,6 +211,36 @@ Writing a suite that measures fairly is its own craft — see
|
|
|
211
211
|
|
|
212
212
|
The surface and judge settings used are recorded in every receipt.
|
|
213
213
|
|
|
214
|
+
### Isolation (the cli surfaces)
|
|
215
|
+
|
|
216
|
+
A `SKILL.md` you did not write is instructions to an agent that has tools. On
|
|
217
|
+
the two cli surfaces driftproof therefore never runs `claude` or `codex` as you.
|
|
218
|
+
Every spawn goes through a hop to a dedicated unprivileged unix user, from an
|
|
219
|
+
empty environment:
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
/usr/bin/sudo -n -u driftproof-eval /usr/bin/env -i HOME=<its home> PATH=<its ~/.local/bin>:/usr/bin:/bin bash -lc '<wrapper>' driftproof-eval-hop claude -p ...
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Only `HOME` and `PATH` reach the child, both constructed from the user name, none
|
|
226
|
+
copied from your shell. The working directory is a fresh temp directory made for
|
|
227
|
+
that one call and removed after it, including on timeout. The CLI argv is passed
|
|
228
|
+
as separate arguments, never as a shell string. What the evaluated agent cannot
|
|
229
|
+
see: your environment (tokens, keys), your home directory (SSH keys, CLI session
|
|
230
|
+
files), your current directory, and your CLI configuration.
|
|
231
|
+
|
|
232
|
+
**Operator prerequisite, set up once** (see `RUNBOOK.md`): that user exists with a
|
|
233
|
+
mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
|
|
234
|
+
under `~/.local/bin`, and a sudoers rule lets you run commands as it with
|
|
235
|
+
`NOPASSWD`. `DRIFTPROOF_EVAL_USER` names the user (default `driftproof-eval`).
|
|
236
|
+
Without the prerequisite a run fails at once, before any call, and says so.
|
|
237
|
+
|
|
238
|
+
**`--trusted-skill`** runs the legacy same-user path instead: your environment,
|
|
239
|
+
your cwd, your CLI login. It exists for exactly one case, a skill you authored
|
|
240
|
+
yourself on a machine with no eval user. Never pass it for a skill fetched from
|
|
241
|
+
anywhere else. The api surfaces are unaffected either way; they make HTTP calls
|
|
242
|
+
and spawn nothing.
|
|
243
|
+
|
|
214
244
|
### Cost guard
|
|
215
245
|
|
|
216
246
|
Sampling multiplies calls on two axes, and the second one is easy to miss. Each
|
|
@@ -241,7 +271,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
241
271
|
"model_release_date": "2025-10-01",
|
|
242
272
|
"provider": "anthropic",
|
|
243
273
|
"surface": "claude-cli",
|
|
244
|
-
"runner_version": "0.8.
|
|
274
|
+
"runner_version": "0.8.1",
|
|
245
275
|
"date_utc": "2026-07-27T…Z",
|
|
246
276
|
"registry": "registered",
|
|
247
277
|
"transcripts": "hashes-only",
|
|
@@ -315,7 +345,7 @@ jobs:
|
|
|
315
345
|
runs-on: ubuntu-latest
|
|
316
346
|
steps:
|
|
317
347
|
- uses: actions/checkout@v4
|
|
318
|
-
- uses: driftproofhq/driftproof@v0.8.
|
|
348
|
+
- uses: driftproofhq/driftproof@v0.8.1
|
|
319
349
|
with:
|
|
320
350
|
skill-dir: skills/my-skill
|
|
321
351
|
models: claude-haiku-4-5
|
|
@@ -331,7 +361,11 @@ fails the job on `REGRESSED` unless you set `fail-on-regression: 'false'`. For a
|
|
|
331
361
|
free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job env —
|
|
332
362
|
the runner returns canned receipts so the wiring can be tested without spend (this
|
|
333
363
|
is exactly how the action's own [self-test](.github/workflows/action-selftest.yml)
|
|
334
|
-
runs).
|
|
364
|
+
runs). That self-test is also the proof behind the input hardening: on every
|
|
365
|
+
run it passes one hostile value (a quote, a semicolon, `$(...)`, a backtick and
|
|
366
|
+
a newline) through each of the action's five inputs on the real runner and
|
|
367
|
+
fails unless every one is refused before anything ran, so a green check there
|
|
368
|
+
is a claim you can read, not one you have to take.
|
|
335
369
|
|
|
336
370
|
### Badge
|
|
337
371
|
|
package/bin/driftproof
CHANGED
|
@@ -10,7 +10,7 @@ const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run'
|
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
11
|
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
|
-
const { surfaceForModel, isSubscriptionSurface, resolveModel } = require('../lib/provider');
|
|
13
|
+
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
15
|
const { registryStatus } = require('../lib/models');
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
@@ -54,6 +54,10 @@ function loadRc(skillDir) {
|
|
|
54
54
|
return merged;
|
|
55
55
|
}
|
|
56
56
|
|
|
57
|
+
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
58
|
+
// positional instead of swallowing it as the flag's value.
|
|
59
|
+
const BOOLEAN_FLAGS = new Set(['trusted-skill']);
|
|
60
|
+
|
|
57
61
|
function parseArgs(argv) {
|
|
58
62
|
const positional = [];
|
|
59
63
|
const flags = {};
|
|
@@ -62,7 +66,7 @@ function parseArgs(argv) {
|
|
|
62
66
|
if (a.startsWith('--')) {
|
|
63
67
|
const key = a.slice(2);
|
|
64
68
|
const next = argv[i + 1];
|
|
65
|
-
if (next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
69
|
+
if (BOOLEAN_FLAGS.has(key) || next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
66
70
|
else { flags[key] = next; i++; }
|
|
67
71
|
} else positional.push(a);
|
|
68
72
|
}
|
|
@@ -76,7 +80,7 @@ USAGE
|
|
|
76
80
|
${PROJECT_NAME} init <dir> scaffold SKILL.md + evals/evals.json + .driftproofrc
|
|
77
81
|
${PROJECT_NAME} run <skill-dir> [--models a,b] [--samples N] [--max-cases N] [--max-calls N]
|
|
78
82
|
[--judge-model M] [--concurrency N] [--max-usd N]
|
|
79
|
-
[--keep-transcripts] [--out DIR]
|
|
83
|
+
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
80
84
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
81
85
|
${PROJECT_NAME} validate <receipt.json>
|
|
82
86
|
${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
|
|
@@ -88,6 +92,7 @@ ENV
|
|
|
88
92
|
ANTHROPIC_API_KEY required when CLAUDE_PROVIDER=api
|
|
89
93
|
DRIFTPROOF_REGISTRY path to an alternate model registry (default: the packaged config/models.json)
|
|
90
94
|
DRIFTPROOF_STUB=1 offline stub surface — canned receipts, zero model calls (for CI)
|
|
95
|
+
DRIFTPROOF_EVAL_USER the unix user the isolated hop runs the CLI as (default: driftproof-eval)
|
|
91
96
|
|
|
92
97
|
NOTES
|
|
93
98
|
- Default models list is 'haiku' only (cheap dev default).
|
|
@@ -112,6 +117,16 @@ NOTES
|
|
|
112
117
|
conservative default price.
|
|
113
118
|
- --concurrency runs that many (case,mode) tasks at once (default 1). Higher
|
|
114
119
|
values cut wall-clock on the cli surface (cold-start dominated).
|
|
120
|
+
- ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
|
|
121
|
+
runs as a dedicated unprivileged unix user (DRIFTPROOF_EVAL_USER) through
|
|
122
|
+
sudo, from an EMPTY environment (only HOME and PATH, constructed), in a fresh
|
|
123
|
+
temp working directory removed after the call. A third-party SKILL.md is
|
|
124
|
+
instructions to an agent with tools; this is what keeps it away from your
|
|
125
|
+
keys, tokens and files. Operator prerequisite: that user, its own logged-in
|
|
126
|
+
CLIs, and a NOPASSWD sudoers rule (see RUNBOOK.md).
|
|
127
|
+
- --trusted-skill runs the legacy same-user path instead: your environment,
|
|
128
|
+
your cwd, your CLI login. FOR SKILLS YOU AUTHORED YOURSELF ONLY. Never pass
|
|
129
|
+
it for a skill fetched from anywhere else.
|
|
115
130
|
- import converts another tool's results into a valid receipt with HONEST
|
|
116
131
|
epistemics: verification_level DECLARED (never TESTED), surface "external",
|
|
117
132
|
source "imported/<tool>", and NO fabricated hashes. Imported receipts are
|
|
@@ -137,6 +152,10 @@ async function cmdRun(positional, flags) {
|
|
|
137
152
|
const concurrency = flags.concurrency ? parseInt(flags.concurrency, 10) : (rc.concurrency != null ? parseInt(rc.concurrency, 10) : 1);
|
|
138
153
|
const maxUsd = flags['max-usd'] ? parseFloat(flags['max-usd']) : (rc.max_usd != null ? parseFloat(rc.max_usd) : DEV_MAX_USD);
|
|
139
154
|
const keepTranscripts = !!flags['keep-transcripts'];
|
|
155
|
+
// spec 022: the same-user spawn exists only behind this flag. Resolving the
|
|
156
|
+
// eval user here means a bad DRIFTPROOF_EVAL_USER is refused before the banner.
|
|
157
|
+
const trusted = !!flags['trusted-skill'];
|
|
158
|
+
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
140
159
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
141
160
|
fs.mkdirSync(outDir, { recursive: true });
|
|
142
161
|
|
|
@@ -166,6 +185,7 @@ async function cmdRun(positional, flags) {
|
|
|
166
185
|
console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
|
|
167
186
|
console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
|
|
168
187
|
console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
|
|
188
|
+
if (subSurfaces.length) console.log(` isolation: ${isolation}`);
|
|
169
189
|
console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
|
|
170
190
|
console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
171
191
|
if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
|
|
@@ -200,7 +220,7 @@ async function cmdRun(positional, flags) {
|
|
|
200
220
|
skill,
|
|
201
221
|
model,
|
|
202
222
|
opts: {
|
|
203
|
-
maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts,
|
|
223
|
+
maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts, trusted,
|
|
204
224
|
onProgress: (p) => {
|
|
205
225
|
if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`);
|
|
206
226
|
},
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.8.
|
|
12
|
+
const RUNNER_VERSION = '0.8.1';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
package/lib/judge.js
CHANGED
|
@@ -81,9 +81,9 @@ function judgeSettings(samples, judgeModel) {
|
|
|
81
81
|
// Grade one response once. Returns { score, reason, raw }.
|
|
82
82
|
// `raw` is the judge's verbatim output text (hashed into the receipt for
|
|
83
83
|
// transcript auditability, and optionally retained under --keep-transcripts).
|
|
84
|
-
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature }) {
|
|
84
|
+
async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature, trusted = false }) {
|
|
85
85
|
const prompt = buildJudgePrompt({ task, response, rubric });
|
|
86
|
-
const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature });
|
|
86
|
+
const { text, attempts, usage } = await complete({ system: JUDGE_SYSTEM, prompt, model, maxTokens: 400, timeoutMs, temperature, trusted });
|
|
87
87
|
let parsed;
|
|
88
88
|
try {
|
|
89
89
|
parsed = extractJsonObject(text);
|
|
@@ -104,7 +104,9 @@ async function gradeOnce({ task, response, rubric, model, timeoutMs, temperature
|
|
|
104
104
|
// JUDGE calls timed out on the api policy while running on a CLI surface, which
|
|
105
105
|
// is the second shadowing site and the one #007's prep session had not found.
|
|
106
106
|
// Passing `undefined` through lets lib/provider.js resolve the declared policy.
|
|
107
|
-
|
|
107
|
+
// `trusted` (spec 022) is plumbed to complete() untouched: the judge takes the
|
|
108
|
+
// same spawn path as the generation it grades, so --trusted-skill is whole-run.
|
|
109
|
+
async function gradeSamples({ task, response, rubric, model, samples = 5, timeoutMs, trusted = false }) {
|
|
108
110
|
const settings = judgeSettings(samples, model);
|
|
109
111
|
const scores = [];
|
|
110
112
|
const reasons = [];
|
|
@@ -114,7 +116,7 @@ async function gradeSamples({ task, response, rubric, model, samples = 5, timeou
|
|
|
114
116
|
for (let i = 0; i < samples; i++) {
|
|
115
117
|
let r;
|
|
116
118
|
try {
|
|
117
|
-
r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature });
|
|
119
|
+
r = await gradeOnce({ task, response, rubric, model, timeoutMs, temperature: settings.temperature === null ? undefined : settings.temperature, trusted });
|
|
118
120
|
} catch (e) {
|
|
119
121
|
// A judge sample that persistently failed (e.g. timed out after retries):
|
|
120
122
|
// tag the error so the runner can charge for the spend and mark the whole
|
package/lib/provider.js
CHANGED
|
@@ -4,7 +4,9 @@
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const os = require('os');
|
|
6
6
|
const path = require('path');
|
|
7
|
-
|
|
7
|
+
// Called through the module object (cp.spawn), never destructured: the gate
|
|
8
|
+
// records spawns by replacing that property, so nothing real runs offline.
|
|
9
|
+
const cp = require('child_process');
|
|
8
10
|
const { withRetry, withTimeout } = require('./json');
|
|
9
11
|
const { stubComplete, stubEnabled } = require('./stub');
|
|
10
12
|
const {
|
|
@@ -126,7 +128,10 @@ const CODEX_OVERHEAD_NOTE =
|
|
|
126
128
|
//
|
|
127
129
|
// `temperature` is honoured ONLY on api surfaces; cli surfaces control sampling
|
|
128
130
|
// themselves, so temperature is ignored there and the receipt records that fact.
|
|
129
|
-
|
|
131
|
+
// `trusted` (spec 022) selects the SAME-USER legacy spawn on the two CLI lanes.
|
|
132
|
+
// It is false unless a caller says otherwise; bin/driftproof exposes it as
|
|
133
|
+
// --trusted-skill, for skills the operator authored. Api lanes ignore it.
|
|
134
|
+
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = null, temperature = undefined, trusted = false }) {
|
|
130
135
|
const surface = surfaceForModel(model);
|
|
131
136
|
const provider = providerForSurface(surface);
|
|
132
137
|
// Offline stub surface: canned completion, zero model calls. The receipt still
|
|
@@ -149,9 +154,9 @@ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = n
|
|
|
149
154
|
const laneRunner = () => {
|
|
150
155
|
switch (surface) {
|
|
151
156
|
case 'api': return completeAnthropicApi({ system, prompt, model, maxTokens, temperature });
|
|
152
|
-
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
157
|
+
case 'claude-cli': return completeClaudeCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
|
|
153
158
|
case 'openai-api': return completeOpenaiApi({ system, prompt, model, maxTokens, temperature, timeoutMs: effTimeout });
|
|
154
|
-
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout });
|
|
159
|
+
case 'openai-cli': return completeCodexCli({ system, prompt, model, timeoutMs: effTimeout, trusted });
|
|
155
160
|
default: throw new Error(`unknown surface: ${surface}`);
|
|
156
161
|
}
|
|
157
162
|
};
|
|
@@ -200,47 +205,182 @@ async function completeAnthropicApi({ system, prompt, model, maxTokens, temperat
|
|
|
200
205
|
return { text, usage: parseAnthropicApiUsage(resp.usage) };
|
|
201
206
|
}
|
|
202
207
|
|
|
203
|
-
// ──
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
+
// ── isolation: the eval-user hop (spec 022) ───────────────────────────────────
|
|
209
|
+
//
|
|
210
|
+
// Every local-CLI spawn runs as a DEDICATED UNPRIVILEGED UNIX USER by default.
|
|
211
|
+
// The third-party SKILL.md a run evaluates is instructions to an agent with
|
|
212
|
+
// tools, and until spec 022 that agent ran as the operator: same uid, whole
|
|
213
|
+
// environment block, operator's cwd, operator's CLI configuration. The
|
|
214
|
+
// 2026-09-03 security audit (findings A2, N5) confirmed a live API token in the
|
|
215
|
+
// inherited environment and this host's SSH keys and CLI session tokens
|
|
216
|
+
// readable by the same uid; this was then reproduced by execution.
|
|
217
|
+
//
|
|
218
|
+
// The boundary is a uid change reached through sudo from an EMPTY environment:
|
|
219
|
+
//
|
|
220
|
+
// /usr/bin/sudo -n -u <eval-user> /usr/bin/env -i HOME=/home/<eval-user>
|
|
221
|
+
// PATH=/home/<eval-user>/.local/bin:/usr/bin:/bin bash -lc '<wrapper>'
|
|
222
|
+
// driftproof-eval-hop <cli> <cli-args...>
|
|
223
|
+
//
|
|
224
|
+
// Operator prerequisite, NOT created here (RUNBOOK.md): the user exists with a
|
|
225
|
+
// mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
|
|
226
|
+
// under ~/.local/bin, and a sudoers rule grants the operator NOPASSWD as that
|
|
227
|
+
// user. The runner refuses loudly when any of that is missing; it never creates
|
|
228
|
+
// a user, edits sudoers, or logs a CLI in.
|
|
229
|
+
//
|
|
230
|
+
// The same-user (legacy) spawn survives ONLY behind `trusted: true`, exposed by
|
|
231
|
+
// bin/driftproof as --trusted-skill, for skills the operator authored.
|
|
232
|
+
const SUDO_BIN = '/usr/bin/sudo';
|
|
233
|
+
const ENV_BIN = '/usr/bin/env';
|
|
234
|
+
const EVAL_USER_DEFAULT = 'driftproof-eval';
|
|
235
|
+
// A user name that can safely follow `sudo -u`: no leading dash, no shell
|
|
236
|
+
// metacharacter, no path separator. Anything else is refused before any spawn.
|
|
237
|
+
const EVAL_USER_RE = /^[a-z_][a-z0-9_-]{0,31}$/;
|
|
238
|
+
// The ONLY names the child environment carries, each CONSTRUCTED from the eval
|
|
239
|
+
// user's name in isolatedEnv() and never copied from process.env. Verified
|
|
240
|
+
// 2026-09-03: both CLIs run through the hop with these two alone; the login
|
|
241
|
+
// shell supplies LANG and TERM itself. Data, frozen, asserted by the gate.
|
|
242
|
+
const ISOLATED_ENV_ALLOWLIST = Object.freeze(['HOME', 'PATH']);
|
|
243
|
+
// $0 of the wrapper, so `ps` shows what the bash process is.
|
|
244
|
+
const HOP_LABEL = 'driftproof-eval-hop';
|
|
245
|
+
// The wrapper runs INSIDE the hop, as the eval user. It takes the CLI as
|
|
246
|
+
// positional parameters ("$@"), so no SKILL.md text is ever shell-interpreted.
|
|
247
|
+
// - the cwd is a fresh per-call directory under the system tmp, created by
|
|
248
|
+
// the eval user (one the parent creates is mode 700 to it);
|
|
249
|
+
// - the CLI runs under setsid in its own process group, stdin preserved
|
|
250
|
+
// (`<&0` defeats the /dev/null a non-interactive bash gives a background job);
|
|
251
|
+
// - TERM/INT/HUP, which sudo relays from the parent on timeout, kill that
|
|
252
|
+
// whole group, remove the directory and exit 143;
|
|
253
|
+
// - a normal exit removes the directory and returns the CLI's status.
|
|
254
|
+
const ISOLATED_WRAPPER = [
|
|
255
|
+
'd=$(mktemp -d -t driftproof-eval.XXXXXXXX) || exit 97',
|
|
256
|
+
'cd "$d" || exit 97',
|
|
257
|
+
'setsid -w "$@" <&0 & pid=$!',
|
|
258
|
+
'trap \'kill -TERM -- "-$pid" 2>/dev/null; sleep 1; kill -KILL -- "-$pid" 2>/dev/null; cd /; rm -rf "$d"; exit 143\' TERM INT HUP',
|
|
259
|
+
'wait "$pid"; r=$?',
|
|
260
|
+
'cd /; rm -rf "$d"; exit $r',
|
|
261
|
+
].join('; ');
|
|
262
|
+
|
|
263
|
+
// DRIFTPROOF_EVAL_USER selects WHICH user isolates, never WHETHER isolation
|
|
264
|
+
// happens (that is the trusted flag, an argv decision). Empty means default.
|
|
265
|
+
function evalUser() {
|
|
266
|
+
const raw = process.env.DRIFTPROOF_EVAL_USER;
|
|
267
|
+
const u = raw === undefined || raw === '' ? EVAL_USER_DEFAULT : raw;
|
|
268
|
+
if (!EVAL_USER_RE.test(u)) throw new Error(`DRIFTPROOF_EVAL_USER is not a plausible unix user name: ${JSON.stringify(u)}`);
|
|
269
|
+
return u;
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
// The child environment: constructed, not copied. Keys are exactly the allowlist.
|
|
273
|
+
function isolatedEnv(user) {
|
|
274
|
+
const home = `/home/${user}`;
|
|
275
|
+
return { HOME: home, PATH: `${home}/.local/bin:/usr/bin:/bin` };
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// The spawn plan for one CLI call: { mode, user, file, args, env }. Pure, so the
|
|
279
|
+
// gate can inspect what WOULD be spawned. Untrusted → the hop; trusted → the
|
|
280
|
+
// legacy same-user spawn exactly as it was (env inherited; the claude lane still
|
|
281
|
+
// strips ANTHROPIC_API_KEY so the CLI uses the subscription, not the metered key).
|
|
282
|
+
function buildSpawnPlan({ bin, args, trusted = false }) {
|
|
283
|
+
if (trusted) {
|
|
208
284
|
const env = { ...process.env };
|
|
209
|
-
delete env.ANTHROPIC_API_KEY;
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
285
|
+
if (bin === 'claude') delete env.ANTHROPIC_API_KEY;
|
|
286
|
+
return { mode: 'same-user', user: null, file: bin, args: [...args], env };
|
|
287
|
+
}
|
|
288
|
+
const user = evalUser();
|
|
289
|
+
const childEnv = isolatedEnv(user);
|
|
290
|
+
return {
|
|
291
|
+
mode: 'isolated',
|
|
292
|
+
user,
|
|
293
|
+
file: SUDO_BIN,
|
|
294
|
+
args: ['-n', '-u', user, ENV_BIN, '-i', ...ISOLATED_ENV_ALLOWLIST.map((k) => `${k}=${childEnv[k]}`), 'bash', '-lc', ISOLATED_WRAPPER, HOP_LABEL, bin, ...args],
|
|
295
|
+
// The sudo front process gets NOTHING from this process. sudo resolves
|
|
296
|
+
// /usr/bin/env itself; env -i then starts the child from empty.
|
|
297
|
+
env: {},
|
|
298
|
+
};
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// Run a plan: `input` to stdin, stdout/stderr collected, settled on close.
|
|
302
|
+
// The timeout signal differs by mode. The legacy child gets SIGKILL as it
|
|
303
|
+
// always did. The hop gets SIGTERM: it is the signal the operator's uid can
|
|
304
|
+
// deliver to a root-owned sudo, and the one sudo relays to the wrapper's trap.
|
|
305
|
+
// SIGKILL would neither be permitted nor relayed, and a wrapper without the
|
|
306
|
+
// trap left the CLI orphaned with the stdout pipe open, so close never fired.
|
|
307
|
+
function runPlan(plan, { input = null, timeoutMs = 300000, graceMs = 5000 } = {}) {
|
|
308
|
+
return new Promise((resolve, reject) => {
|
|
309
|
+
let child;
|
|
310
|
+
try { child = cp.spawn(plan.file, plan.args, { env: plan.env, stdio: ['pipe', 'pipe', 'pipe'] }); }
|
|
311
|
+
catch (e) { return reject(e); }
|
|
312
|
+
let stdout = '';
|
|
313
|
+
let stderr = '';
|
|
314
|
+
let timedOut = false;
|
|
315
|
+
const killer = setTimeout(() => {
|
|
316
|
+
timedOut = true;
|
|
317
|
+
child.kill(plan.mode === 'isolated' ? 'SIGTERM' : 'SIGKILL');
|
|
318
|
+
}, timeoutMs + graceMs);
|
|
319
|
+
child.stdout.on('data', (d) => { stdout += d; });
|
|
320
|
+
child.stderr.on('data', (d) => { stderr += d; });
|
|
321
|
+
child.on('error', (e) => { clearTimeout(killer); reject(e); });
|
|
322
|
+
child.on('close', (code, signal) => { clearTimeout(killer); resolve({ code, signal, stdout, stderr, timedOut }); });
|
|
323
|
+
child.stdin.on('error', () => { /* EPIPE if the CLI exits before reading all of stdin */ });
|
|
324
|
+
if (input != null) child.stdin.write(input);
|
|
240
325
|
child.stdin.end();
|
|
241
326
|
});
|
|
242
327
|
}
|
|
243
328
|
|
|
329
|
+
// The hop with an arbitrary command in place of the CLI, for the gate: the same
|
|
330
|
+
// plan builder and the same runner, never trusted. No provider is reached.
|
|
331
|
+
function runIsolated(cmdArgv, opts = {}) {
|
|
332
|
+
const [bin, ...args] = cmdArgv;
|
|
333
|
+
return runPlan(buildSpawnPlan({ bin, args, trusted: false }), { graceMs: 0, ...opts });
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
// A hop failure, said in terms of what to fix. Returns null when the failure is
|
|
337
|
+
// the CLI's own (exit status from inside the hop), which the lane reports as before.
|
|
338
|
+
function hopFailure(bin, r) {
|
|
339
|
+
const err = String(r.stderr || '').trim();
|
|
340
|
+
if (r.code === 127) {
|
|
341
|
+
return `the \`${bin}\` CLI is not on the eval user's PATH inside the isolated hop (${err.slice(0, 200)}). `
|
|
342
|
+
+ 'Install and log it in as that user (RUNBOOK.md), or pass --trusted-skill for a skill you authored.';
|
|
343
|
+
}
|
|
344
|
+
const sudoLine = err.split('\n').find((l) => /^sudo:/.test(l));
|
|
345
|
+
if (sudoLine) {
|
|
346
|
+
return `the isolated eval-user hop failed before reaching \`${bin}\`: ${sudoLine}. The eval user and its NOPASSWD sudoers `
|
|
347
|
+
+ 'rule are an operator prerequisite (RUNBOOK.md); pass --trusted-skill only for a skill you authored.';
|
|
348
|
+
}
|
|
349
|
+
return null;
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
// ── anthropic/cli (claude -p) ──────────────────────────────────────────────────
|
|
353
|
+
async function completeClaudeCli({ system, prompt, model, timeoutMs, trusted = false }) {
|
|
354
|
+
// v0.4: `--output-format json` returns ONE JSON object carrying both the
|
|
355
|
+
// final text (`result`) and the token usage (`usage`) — the default text
|
|
356
|
+
// output carries no usage at all, which is why usage was previously null on
|
|
357
|
+
// this lane. The text is read from the parsed object; if the CLI ever emits
|
|
358
|
+
// something unparseable we fall back to the raw stdout so a run degrades to
|
|
359
|
+
// the old behaviour (text, no usage) rather than failing.
|
|
360
|
+
const args = ['-p', '--output-format', 'json', '--model', resolveModel(model)];
|
|
361
|
+
if (system) args.push('--append-system-prompt', system);
|
|
362
|
+
const plan = buildSpawnPlan({ bin: 'claude', args, trusted });
|
|
363
|
+
let r;
|
|
364
|
+
try {
|
|
365
|
+
r = await runPlan(plan, { input: prompt, timeoutMs });
|
|
366
|
+
} catch (e) {
|
|
367
|
+
if (e.code === 'ENOENT') {
|
|
368
|
+
throw new Error(plan.mode === 'isolated'
|
|
369
|
+
? `the isolated hop requires ${SUDO_BIN}`
|
|
370
|
+
: "the anthropic/cli surface requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api.");
|
|
371
|
+
}
|
|
372
|
+
throw e;
|
|
373
|
+
}
|
|
374
|
+
if (r.code !== 0) {
|
|
375
|
+
const hop = plan.mode === 'isolated' ? hopFailure('claude', r) : null;
|
|
376
|
+
throw new Error(hop || `claude CLI exited ${r.code}: ${r.stderr.slice(0, 400)}`);
|
|
377
|
+
}
|
|
378
|
+
const parsed = parseClaudeCliJson(r.stdout);
|
|
379
|
+
if (!parsed) return { text: r.stdout.trim(), usage: null };
|
|
380
|
+
if (parsed.isError) throw new Error(`claude CLI reported is_error: ${String(parsed.text || '').slice(0, 300)}`);
|
|
381
|
+
return { text: String(parsed.text || '').trim(), usage: parsed.usage };
|
|
382
|
+
}
|
|
383
|
+
|
|
244
384
|
// ── openai/api (Chat Completions-compatible) ───────────────────────────────────
|
|
245
385
|
// Generic OpenAI-compatible client using global fetch (Node ≥ 18). The base_url
|
|
246
386
|
// is configurable — env OPENAI_BASE_URL wins, else the registry's provider
|
|
@@ -294,7 +434,7 @@ function openaiProviderConfig() {
|
|
|
294
434
|
|
|
295
435
|
// ── openai/cli (codex exec — ChatGPT subscription) ─────────────────────────────
|
|
296
436
|
// Invocation template (from the verified recon, reference_codex_cli):
|
|
297
|
-
// codex exec --json -s read-only --skip-git-repo-check --ephemeral
|
|
437
|
+
// codex exec --json -s read-only --skip-git-repo-check --ephemeral [-m <model>] -
|
|
298
438
|
// Honoured facts:
|
|
299
439
|
// - `-a/--ask-for-approval` is INVALID on `codex exec` — it is NEVER passed;
|
|
300
440
|
// exec already defaults to approval_policy "never".
|
|
@@ -305,6 +445,12 @@ function openaiProviderConfig() {
|
|
|
305
445
|
// - `codex exec` has no system-prompt flag, so a system prompt (the SKILL.md in
|
|
306
446
|
// with_skill mode) is folded into the prompt text.
|
|
307
447
|
// - stderr emits a harmless "Reading additional input from stdin…" notice.
|
|
448
|
+
// - `-o/--output-last-message` is NEVER passed (spec 022 AC-9, approval F-1).
|
|
449
|
+
// The final message is taken from the `--json` stream on the stdout pipe
|
|
450
|
+
// this process owns. A `-o` path is a file the child writes and the parent
|
|
451
|
+
// reads back; isolated, the child is the eval user running third-party
|
|
452
|
+
// instructions, and a symlink planted at that path would have made this
|
|
453
|
+
// process read an operator-owned file into the generation.
|
|
308
454
|
const CODEX_EXEC_ARGS = ['exec', '--json', '-s', 'read-only', '--skip-git-repo-check', '--ephemeral'];
|
|
309
455
|
|
|
310
456
|
function codexAuthPresent(homeDir) {
|
|
@@ -315,9 +461,9 @@ function codexAuthPresent(homeDir) {
|
|
|
315
461
|
}
|
|
316
462
|
|
|
317
463
|
// Build the exact argv for a codex exec call. Exposed for the gate to assert the
|
|
318
|
-
// template (
|
|
319
|
-
function buildCodexArgs({ model
|
|
320
|
-
const args = [...CODEX_EXEC_ARGS
|
|
464
|
+
// template (that `-a` never appears, and neither does `-o`).
|
|
465
|
+
function buildCodexArgs({ model } = {}) {
|
|
466
|
+
const args = [...CODEX_EXEC_ARGS];
|
|
321
467
|
if (model) args.push('-m', resolveModel(model));
|
|
322
468
|
return args;
|
|
323
469
|
}
|
|
@@ -326,66 +472,58 @@ function buildCodexArgs({ model, outFile }) {
|
|
|
326
472
|
// means "read the prompt from stdin". The prompt is NEVER a positional argv —
|
|
327
473
|
// real SKILL.md files start with `---` (YAML frontmatter) and codex rejects a
|
|
328
474
|
// positional beginning with `--`. Exposed so the gate can lock this in.
|
|
329
|
-
function codexFinalArgs({ model
|
|
330
|
-
return [...buildCodexArgs({ model
|
|
475
|
+
function codexFinalArgs({ model } = {}) {
|
|
476
|
+
return [...buildCodexArgs({ model }), '-'];
|
|
331
477
|
}
|
|
332
478
|
|
|
333
|
-
function completeCodexCli({ system, prompt, model, timeoutMs }) {
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
479
|
+
async function completeCodexCli({ system, prompt, model, timeoutMs, trusted = false }) {
|
|
480
|
+
// The auth presence check is a PARENT-side read of ~/.codex/auth.json, which
|
|
481
|
+
// only means something on the same-user path; the eval user's home is not
|
|
482
|
+
// readable from here, and a missing login there surfaces as codex's own error.
|
|
483
|
+
if (trusted && !codexAuthPresent()) {
|
|
484
|
+
throw new Error('the openai/cli surface requires codex auth (~/.codex/auth.json). Run `codex login` (or `codex login --device-auth` on a headless box).');
|
|
485
|
+
}
|
|
486
|
+
// codex exec has no system-prompt flag → fold the system prompt into the
|
|
487
|
+
// prompt text (this is how the SKILL.md reaches the model in with_skill mode).
|
|
488
|
+
const fullPrompt = system ? `${system}\n\n---\n\n${prompt}` : prompt;
|
|
489
|
+
// The prompt is delivered on STDIN, not as a positional argv: real SKILL.md
|
|
490
|
+
// files begin with `---` (YAML frontmatter), and a positional argument that
|
|
491
|
+
// starts with `--` is rejected by codex's arg parser ("unexpected argument
|
|
492
|
+
// '---'"). `-` as the positional tells `codex exec` to read instructions from
|
|
493
|
+
// stdin (per its --help), which is content-agnostic.
|
|
494
|
+
const plan = buildSpawnPlan({ bin: 'codex', args: codexFinalArgs({ model }), trusted });
|
|
495
|
+
// stdout is CAPTURED. `--json` streams JSONL events there: the last
|
|
496
|
+
// `item.completed`/`agent_message` carries the final text and the terminal
|
|
497
|
+
// `turn.completed` event carries this call's token usage — the only place
|
|
498
|
+
// codex reports it. Nothing is read from the filesystem after the call, on
|
|
499
|
+
// either mode: the stdout pipe is the one channel only the child we spawned
|
|
500
|
+
// can write to, and the eval user cannot make this process read a file
|
|
501
|
+
// through it (spec 022 AC-9).
|
|
502
|
+
let r;
|
|
503
|
+
try {
|
|
504
|
+
r = await runPlan(plan, { input: fullPrompt, timeoutMs });
|
|
505
|
+
} catch (e) {
|
|
506
|
+
if (e.code === 'ENOENT') {
|
|
507
|
+
throw new Error(plan.mode === 'isolated'
|
|
508
|
+
? `the isolated hop requires ${SUDO_BIN}`
|
|
509
|
+
: 'the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).');
|
|
337
510
|
}
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
const
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
// stdin (per its --help), which is content-agnostic.
|
|
347
|
-
const args = codexFinalArgs({ model, outFile });
|
|
348
|
-
|
|
349
|
-
// v0.4: stdout is CAPTURED (it was 'ignore'). `--json` streams JSONL events
|
|
350
|
-
// there, and the terminal `turn.completed` event carries this call's token
|
|
351
|
-
// usage — the only place codex reports it. The final message still comes from
|
|
352
|
-
// the -o file (cleaner than scraping the stream); the JSONL is read for usage.
|
|
353
|
-
const child = spawn('codex', args, { env: { ...process.env }, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
354
|
-
let err = '';
|
|
355
|
-
let jsonl = '';
|
|
356
|
-
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
357
|
-
child.stdout.on('data', (d) => { jsonl += d; });
|
|
358
|
-
child.stderr.on('data', (d) => { err += d; });
|
|
359
|
-
child.on('error', (e) => {
|
|
360
|
-
clearTimeout(killer);
|
|
361
|
-
if (e.code === 'ENOENT') reject(new Error('the openai/cli surface requires the `codex` CLI on PATH (npm i -g @openai/codex).'));
|
|
362
|
-
else reject(e);
|
|
363
|
-
});
|
|
364
|
-
child.on('close', (code) => {
|
|
365
|
-
clearTimeout(killer);
|
|
366
|
-
let text = '';
|
|
367
|
-
try { text = fs.readFileSync(outFile, 'utf8'); } catch (_e) { /* fall through */ }
|
|
368
|
-
try { fs.unlinkSync(outFile); } catch (_e) { /* best effort */ }
|
|
369
|
-
if (code !== 0) return reject(new Error(`codex exec exited ${code}: ${String(err).slice(0, 400)}`));
|
|
370
|
-
const ev = parseCodexJsonl(jsonl);
|
|
371
|
-
// Prefer the -o file; fall back to the stream's agent_message if it is empty.
|
|
372
|
-
const finalText = String(text || ev.text || '').trim();
|
|
373
|
-
resolve({ text: finalText, usage: ev.usage });
|
|
374
|
-
});
|
|
375
|
-
child.stdin.on('error', () => { /* ignore EPIPE if codex exits before reading all input */ });
|
|
376
|
-
child.stdin.write(fullPrompt);
|
|
377
|
-
child.stdin.end();
|
|
378
|
-
});
|
|
511
|
+
throw e;
|
|
512
|
+
}
|
|
513
|
+
if (r.code !== 0) {
|
|
514
|
+
const hop = plan.mode === 'isolated' ? hopFailure('codex', r) : null;
|
|
515
|
+
throw new Error(hop || `codex exec exited ${r.code}: ${String(r.stderr).slice(0, 400)}`);
|
|
516
|
+
}
|
|
517
|
+
const ev = parseCodexJsonl(r.stdout);
|
|
518
|
+
return { text: String(ev.text || '').trim(), usage: ev.usage };
|
|
379
519
|
}
|
|
380
520
|
|
|
381
|
-
// A monotonically-increasing counter for temp-file uniqueness that does not use
|
|
382
|
-
// Math.random (kept deterministic-friendly for any harness that forbids it).
|
|
383
|
-
let _codexCounter = 0;
|
|
384
|
-
function codexCounter() { _codexCounter += 1; return _codexCounter; }
|
|
385
|
-
|
|
386
521
|
module.exports = {
|
|
387
522
|
complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES,
|
|
388
523
|
inferProvider, surfaceForModel, providerForSurface,
|
|
389
524
|
isMeteredSurface, isSubscriptionSurface, retryPolicyForSurface,
|
|
390
525
|
CODEX_OVERHEAD_NOTE, CODEX_EXEC_ARGS, buildCodexArgs, codexFinalArgs, codexAuthPresent,
|
|
526
|
+
// spec 022 — the eval-user hop
|
|
527
|
+
SUDO_BIN, EVAL_USER_DEFAULT, ISOLATED_ENV_ALLOWLIST, ISOLATED_WRAPPER, HOP_LABEL,
|
|
528
|
+
evalUser, isolatedEnv, buildSpawnPlan, runIsolated,
|
|
391
529
|
};
|
package/lib/run.js
CHANGED
|
@@ -51,7 +51,7 @@ function projectCalls(caseCount, samples, draws = 1) {
|
|
|
51
51
|
// Ask the target model to perform one eval case. `withSkill` decides whether the
|
|
52
52
|
// SKILL.md is prepended as a system prompt (the whole point: measure the skill's
|
|
53
53
|
// marginal effect vs a bare baseline).
|
|
54
|
-
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
54
|
+
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs, trusted = false }) {
|
|
55
55
|
// Test seam (gate only): force a persistent timeout for a named case id so the
|
|
56
56
|
// failed_timeout path is exercised deterministically without any live call.
|
|
57
57
|
if (process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID && process.env.DRIFTPROOF_TEST_TIMEOUT_CASEID === caseObj.id) {
|
|
@@ -60,7 +60,7 @@ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
|
60
60
|
throw e;
|
|
61
61
|
}
|
|
62
62
|
const system = withSkill ? skillMd : undefined;
|
|
63
|
-
const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
63
|
+
const { text, usage, wall_ms, attempts } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs, trusted });
|
|
64
64
|
return { text, usage, wall_ms, attempts };
|
|
65
65
|
}
|
|
66
66
|
|
|
@@ -78,8 +78,8 @@ function outcomeFor(mean, stddev, threshold) {
|
|
|
78
78
|
// `generationHash` binds the graded case to the exact generation text (v0.3).
|
|
79
79
|
// Returns { caseResult, sampleTexts } — sampleTexts is transient (retained only
|
|
80
80
|
// under --keep-transcripts; never part of the receipt).
|
|
81
|
-
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples }) {
|
|
82
|
-
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs });
|
|
81
|
+
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples, trusted = false }) {
|
|
82
|
+
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs, trusted });
|
|
83
83
|
const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
|
|
84
84
|
const caseResult = {
|
|
85
85
|
id: caseObj.id,
|
|
@@ -127,7 +127,7 @@ async function mapPool(items, concurrency, fn) {
|
|
|
127
127
|
// a sealed receipt. Enforces a hard call cap; every model+judge call counts.
|
|
128
128
|
//
|
|
129
129
|
// opts: { maxCases, maxCalls, samples, judgeModel, timeoutMs, concurrency,
|
|
130
|
-
// onProgress, budget, keepTranscripts, nowIso }
|
|
130
|
+
// onProgress, budget, keepTranscripts, nowIso, trusted }
|
|
131
131
|
// budget — optional BudgetTracker; accumulates estimated per-call
|
|
132
132
|
// USD as the run proceeds and hard-stops at 1.25× the cap.
|
|
133
133
|
// keepTranscripts — when true, the run records transcripts:"retained-local"
|
|
@@ -177,6 +177,10 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
177
177
|
const onProgress = opts.onProgress || (() => {});
|
|
178
178
|
const budget = opts.budget || null;
|
|
179
179
|
const keepTranscripts = !!opts.keepTranscripts;
|
|
180
|
+
// spec 022: the SAME-USER legacy spawn is reachable only when a caller says
|
|
181
|
+
// `trusted: true` (bin/driftproof --trusted-skill). Every other caller of this
|
|
182
|
+
// function, the report scripts and the release watcher included, isolates.
|
|
183
|
+
const trusted = !!opts.trusted;
|
|
180
184
|
|
|
181
185
|
let cases = skill.suite.cases;
|
|
182
186
|
if (opts.maxCases && cases.length > opts.maxCases) cases = cases.slice(0, opts.maxCases);
|
|
@@ -222,12 +226,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
222
226
|
const drawIndex = draws.length;
|
|
223
227
|
try {
|
|
224
228
|
onProgress({ case: c.id, mode, phase: 'generate', draw: drawIndex });
|
|
225
|
-
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen });
|
|
229
|
+
const gen = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs: ctGen, trusted });
|
|
226
230
|
calls += 1;
|
|
227
231
|
if (budget) budget.add((gen.attempts || 1) * perCallCostUSD(modelId, genKind(withSkill)));
|
|
228
232
|
const generationHash = sha256(String(gen.text || ''));
|
|
229
233
|
onProgress({ case: c.id, mode, phase: 'judge', samples, draw: drawIndex });
|
|
230
|
-
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples });
|
|
234
|
+
const jr = await judgeCase({ caseObj: c, response: gen.text, generationHash, judgeModel, mode, timeoutMs: ctJudge, samples, trusted });
|
|
231
235
|
calls += samples;
|
|
232
236
|
if (budget) budget.add((jr.attempts || samples) * perCallCostUSD(judgeModel, 'judge'));
|
|
233
237
|
const draw = {
|
package/lib/verdict.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const crypto = require('crypto');
|
|
4
5
|
const { EFFECT_FLOOR } = require('../config');
|
|
5
6
|
|
|
6
7
|
// Single-receipt verdict + shields.io badge.
|
|
@@ -72,14 +73,39 @@ function badgeEndpoint(receipt, { label = 'driftproof' } = {}) {
|
|
|
72
73
|
}
|
|
73
74
|
|
|
74
75
|
// Lines suitable for appending to a GitHub Actions $GITHUB_OUTPUT file.
|
|
76
|
+
//
|
|
77
|
+
// Spec 023 (audit A6). Each entry is written in GitHub's multiline form,
|
|
78
|
+
//
|
|
79
|
+
// name<<ghadelim_<32 hex>
|
|
80
|
+
// value
|
|
81
|
+
// ghadelim_<32 hex>
|
|
82
|
+
//
|
|
83
|
+
// with a delimiter drawn fresh from crypto.randomBytes for EVERY entry, so no
|
|
84
|
+
// value can close a block early; and any value carrying \r or \n is refused
|
|
85
|
+
// with a throw before anything is written. The bare `name=value` form this used
|
|
86
|
+
// to emit let a model_id with a newline in it (any string a receipt carries; a
|
|
87
|
+
// .driftproofrc in a skill directory sets it) append a second, attacker-chosen
|
|
88
|
+
// entry, and that entry was then interpolated into the enforce step's shell.
|
|
89
|
+
// Refusing is the honest writer: a line break in a model id is not a value
|
|
90
|
+
// anybody meant to publish. Entries are joined by '\n' with no trailing newline
|
|
91
|
+
// so the caller's console.log stays the one that terminates the text.
|
|
92
|
+
function githubOutputEntry(name, value) {
|
|
93
|
+
const val = String(value);
|
|
94
|
+
if (/[\r\n]/.test(val)) {
|
|
95
|
+
throw new Error(`refusing to write $GITHUB_OUTPUT: the value of "${name}" contains a line break (${JSON.stringify(val)})`);
|
|
96
|
+
}
|
|
97
|
+
const delim = `ghadelim_${crypto.randomBytes(16).toString('hex')}`;
|
|
98
|
+
return `${name}<<${delim}\n${val}\n${delim}`;
|
|
99
|
+
}
|
|
100
|
+
|
|
75
101
|
function githubOutputLines(receipt) {
|
|
76
102
|
const v = verdictFromReceipt(receipt);
|
|
77
103
|
return [
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
].join('\n');
|
|
104
|
+
['verdict', v.verdict],
|
|
105
|
+
['delta', v.delta],
|
|
106
|
+
['message', v.message],
|
|
107
|
+
['color', v.color],
|
|
108
|
+
].map(([k, val]) => githubOutputEntry(k, val)).join('\n');
|
|
83
109
|
}
|
|
84
110
|
|
|
85
|
-
module.exports = { verdictFromReceipt, badgeEndpoint, githubOutputLines, shortModel, VERDICTS };
|
|
111
|
+
module.exports = { verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.8.
|
|
3
|
+
"version": "0.8.1",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|