driftproof 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -5
- package/bin/driftproof +191 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +74 -15
- package/lib/models.js +52 -11
- package/lib/provider.js +265 -103
- package/lib/receipt.js +51 -11
- package/lib/run.js +231 -26
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +38 -7
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/README.md
CHANGED
|
@@ -211,6 +211,36 @@ Writing a suite that measures fairly is its own craft — see
|
|
|
211
211
|
|
|
212
212
|
The surface and judge settings used are recorded in every receipt.
|
|
213
213
|
|
|
214
|
+
### Isolation (the cli surfaces)
|
|
215
|
+
|
|
216
|
+
A `SKILL.md` you did not write is instructions to an agent that has tools. On
|
|
217
|
+
the two cli surfaces driftproof therefore never runs `claude` or `codex` as you.
|
|
218
|
+
Every spawn goes through a hop to a dedicated unprivileged unix user, from an
|
|
219
|
+
empty environment:
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
/usr/bin/sudo -n -u driftproof-eval /usr/bin/env -i HOME=<its home> PATH=<its ~/.local/bin>:/usr/bin:/bin bash -lc '<wrapper>' driftproof-eval-hop claude -p ...
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Only `HOME` and `PATH` reach the child, both constructed from the user name, none
|
|
226
|
+
copied from your shell. The working directory is a fresh temp directory made for
|
|
227
|
+
that one call and removed after it, including on timeout. The CLI argv is passed
|
|
228
|
+
as separate arguments, never as a shell string. What the evaluated agent cannot
|
|
229
|
+
see: your environment (tokens, keys), your home directory (SSH keys, CLI session
|
|
230
|
+
files), your current directory, and your CLI configuration.
|
|
231
|
+
|
|
232
|
+
**Operator prerequisite, set up once** (see `RUNBOOK.md`): that user exists with a
|
|
233
|
+
mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
|
|
234
|
+
under `~/.local/bin`, and a sudoers rule lets you run commands as it with
|
|
235
|
+
`NOPASSWD`. `DRIFTPROOF_EVAL_USER` names the user (default `driftproof-eval`).
|
|
236
|
+
Without the prerequisite a run fails at once, before any call, and says so.
|
|
237
|
+
|
|
238
|
+
**`--trusted-skill`** runs the legacy same-user path instead: your environment,
|
|
239
|
+
your cwd, your CLI login. It exists for exactly one case, a skill you authored
|
|
240
|
+
yourself on a machine with no eval user. Never pass it for a skill fetched from
|
|
241
|
+
anywhere else. The api surfaces are unaffected either way; they make HTTP calls
|
|
242
|
+
and spawn nothing.
|
|
243
|
+
|
|
214
244
|
### Cost guard
|
|
215
245
|
|
|
216
246
|
Sampling multiplies calls on two axes, and the second one is easy to miss. Each
|
|
@@ -232,7 +262,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
232
262
|
|
|
233
263
|
```jsonc
|
|
234
264
|
{
|
|
235
|
-
"schema_version": "0.
|
|
265
|
+
"schema_version": "0.6",
|
|
236
266
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
237
267
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
238
268
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -241,12 +271,16 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
241
271
|
"model_release_date": "2025-10-01",
|
|
242
272
|
"provider": "anthropic",
|
|
243
273
|
"surface": "claude-cli",
|
|
244
|
-
"runner_version": "0.
|
|
274
|
+
"runner_version": "0.9.0",
|
|
245
275
|
"date_utc": "2026-07-27T…Z",
|
|
246
276
|
"registry": "registered",
|
|
247
277
|
"transcripts": "hashes-only",
|
|
248
278
|
"judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
|
|
249
|
-
"surface": "claude-cli"
|
|
279
|
+
"surface": "claude-cli", "model_id": "claude-haiku-4-5-20251001",
|
|
280
|
+
"prompt_template_hash": "…sha256 over the grading template…" },
|
|
281
|
+
"answered_by": { "kind": "model", "attested": true,
|
|
282
|
+
"reported_model": "claude-haiku-4-5-20251001",
|
|
283
|
+
"reported_models": ["claude-haiku-4-5"], "isolation": "eval-user" }
|
|
250
284
|
},
|
|
251
285
|
"results": {
|
|
252
286
|
"cases": [
|
|
@@ -315,7 +349,7 @@ jobs:
|
|
|
315
349
|
runs-on: ubuntu-latest
|
|
316
350
|
steps:
|
|
317
351
|
- uses: actions/checkout@v4
|
|
318
|
-
- uses: driftproofhq/driftproof@v0.
|
|
352
|
+
- uses: driftproofhq/driftproof@v0.9.0
|
|
319
353
|
with:
|
|
320
354
|
skill-dir: skills/my-skill
|
|
321
355
|
models: claude-haiku-4-5
|
|
@@ -331,7 +365,13 @@ fails the job on `REGRESSED` unless you set `fail-on-regression: 'false'`. For a
|
|
|
331
365
|
free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job env —
|
|
332
366
|
the runner returns canned receipts so the wiring can be tested without spend (this
|
|
333
367
|
is exactly how the action's own [self-test](.github/workflows/action-selftest.yml)
|
|
334
|
-
runs).
|
|
368
|
+
runs). A stub receipt says what it is: `verification_level` `UNVERIFIED`,
|
|
369
|
+
`run.surface` `stub`, `run.answered_by.kind` `stub`, and the verdict on it is
|
|
370
|
+
`NOT_MEASURED` — it proves the wiring, never the skill. That self-test is also the proof behind the input hardening: on every
|
|
371
|
+
run it passes one hostile value (a quote, a semicolon, `$(...)`, a backtick and
|
|
372
|
+
a newline) through each of the action's five inputs on the real runner and
|
|
373
|
+
fails unless every one is refused before anything ran, so a green check there
|
|
374
|
+
is a claim you can read, not one you have to take.
|
|
335
375
|
|
|
336
376
|
### Badge
|
|
337
377
|
|
package/bin/driftproof
CHANGED
|
@@ -6,13 +6,13 @@ const fs = require('fs');
|
|
|
6
6
|
const path = require('path');
|
|
7
7
|
const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MAX_CALLS } = require('../config');
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
|
-
const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
|
|
9
|
+
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
11
|
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
|
-
const { surfaceForModel, isSubscriptionSurface, resolveModel } = require('../lib/provider');
|
|
13
|
+
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
|
-
const { registryStatus } = require('../lib/models');
|
|
15
|
+
const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
17
17
|
const { scaffoldInit } = require('../lib/init');
|
|
18
18
|
|
|
@@ -40,6 +40,24 @@ function writeTranscripts(receipt, transcripts) {
|
|
|
40
40
|
|
|
41
41
|
const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
|
|
42
42
|
|
|
43
|
+
// ── THE DISPLAY VERIFIES BEFORE IT READS (spec 026 AC-20, F7) ────────────────
|
|
44
|
+
// badge, diff and export read a receipt somebody else may have edited. Until
|
|
45
|
+
// this, badge never checked the hash and diff and export printed a warning
|
|
46
|
+
// and proceeded, so a receipt with one digit moved and the hash left as it
|
|
47
|
+
// was rendered `passing` (028's D-5, for the CLI). A receipt whose
|
|
48
|
+
// receipt_hash does not verify is refused here: nothing on stdout, no file,
|
|
49
|
+
// the reason on stderr naming receipt_hash and the file, exit 4. The bound the
|
|
50
|
+
// threat model states does not move: the same edit followed by a re-seal
|
|
51
|
+
// (sealReceipt) verifies and renders, which is what receipt_hash is: integrity
|
|
52
|
+
// since sealing, not authenticity. The check lives here, in the three
|
|
53
|
+
// commands, and not in lib/verdict.js's writer, which the repo gate calls on
|
|
54
|
+
// synthetic unsealed receipts.
|
|
55
|
+
function refuseUnverified(receipt, file, command) {
|
|
56
|
+
if (verifyReceiptHash(receipt)) return;
|
|
57
|
+
console.error(` ✗ REFUSED (${command}): receipt_hash does not verify for ${path.basename(file)} (tampered or hand-edited since it was sealed); nothing rendered.`);
|
|
58
|
+
process.exit(4);
|
|
59
|
+
}
|
|
60
|
+
|
|
43
61
|
// Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
|
|
44
62
|
// in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
|
|
45
63
|
// flags always win over the rc; the rc wins over built-in defaults.
|
|
@@ -54,6 +72,85 @@ function loadRc(skillDir) {
|
|
|
54
72
|
return merged;
|
|
55
73
|
}
|
|
56
74
|
|
|
75
|
+
// Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
|
|
76
|
+
// positional instead of swallowing it as the flag's value.
|
|
77
|
+
const BOOLEAN_FLAGS = new Set(['trusted-skill']);
|
|
78
|
+
|
|
79
|
+
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
80
|
+
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
81
|
+
// them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
|
|
82
|
+
// same copy and AC-9 asserts the three identical, rule for rule, both ways.
|
|
83
|
+
// Consolidating into one shared module is 028's decision 1, the spec after 026.
|
|
84
|
+
//
|
|
85
|
+
// Why a contract at the door: `parseInt('abc', 10)` is NaN, both cost guards
|
|
86
|
+
// are `>` comparisons against it, both are false, and the run completed and
|
|
87
|
+
// wrote a receipt that carried no trace of it (the dollar cap never compared;
|
|
88
|
+
// the call cap silently became the dev default). `--samples abc` was quieter
|
|
89
|
+
// still: `NaN || 5` is 5. Every numeric input, from the command line or the
|
|
90
|
+
// rc, is checked against its rule BEFORE the projection is printed and before
|
|
91
|
+
// any call; a flag given with no value (the parser records `true`) and an
|
|
92
|
+
// empty string are refused the same way, not defaulted.
|
|
93
|
+
const INPUT_CONTRACT = {
|
|
94
|
+
rules: {
|
|
95
|
+
models: { re: /^[A-Za-z0-9._-]+(,[A-Za-z0-9._-]+)*$/, message: 'models: expected a comma-separated list of model ids (letters, digits, . _ -)' },
|
|
96
|
+
max_usd: { re: /^[0-9]+(\.[0-9]+)?$/, message: 'max-usd: expected a positive decimal number such as 20 or 3.53' },
|
|
97
|
+
max_usd_zero: { re: /^0+(\.0+)?$/, reject_on_match: true, message: 'max-usd: must be greater than zero' },
|
|
98
|
+
max_calls: { re: /^[1-9][0-9]*$/, message: 'max-calls: expected a positive integer with no leading zero' },
|
|
99
|
+
skill_dir_control: { re: /[\x00-\x1f\x7f]/, reject_on_match: true, message: 'skill-dir: contains a control character' },
|
|
100
|
+
},
|
|
101
|
+
// Which rule each `driftproof run` input is checked against. samples,
|
|
102
|
+
// max-cases and concurrency are positive integers and reuse max_calls' rule.
|
|
103
|
+
flags: { 'models': ['models'], 'max-usd': ['max_usd', 'max_usd_zero'], 'max-calls': ['max_calls'], 'samples': ['max_calls'], 'max-cases': ['max_calls'], 'concurrency': ['max_calls'], 'skill-dir': ['skill_dir_control'] },
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
function shownValue(v) {
|
|
107
|
+
if (v === true) return '<no value>';
|
|
108
|
+
return JSON.stringify(String(v));
|
|
109
|
+
}
|
|
110
|
+
// Refuse one input: the rule's own message, the input's name and the value it
|
|
111
|
+
// carried, exit 2. Nothing has been printed or spent when this runs.
|
|
112
|
+
function refuseInput(name, value, message, where) {
|
|
113
|
+
console.error(` ✗ REFUSED: ${message}, got ${shownValue(value)} (${where} ${name}); nothing run, no receipt written.`);
|
|
114
|
+
process.exit(2);
|
|
115
|
+
}
|
|
116
|
+
// Check one input against every rule its flag maps to. `value` is what was
|
|
117
|
+
// given: a string, a number (from the rc), `true` (a flag with no value) or
|
|
118
|
+
// an empty string; only a string that matches every rule passes.
|
|
119
|
+
function checkInput(flag, value, where, shownName = flag) {
|
|
120
|
+
const rules = INPUT_CONTRACT.flags[flag] || [];
|
|
121
|
+
const text = typeof value === 'string' ? value : (typeof value === 'number' && Number.isFinite(value)) ? String(value) : null;
|
|
122
|
+
if (text === null) refuseInput(shownName, value, `${flag}: expected a value`, where);
|
|
123
|
+
for (const key of rules) {
|
|
124
|
+
const rule = INPUT_CONTRACT.rules[key];
|
|
125
|
+
const hit = rule.re.test(text);
|
|
126
|
+
if (rule.reject_on_match ? hit : !hit) refuseInput(shownName, value, rule.message, where);
|
|
127
|
+
}
|
|
128
|
+
return text;
|
|
129
|
+
}
|
|
130
|
+
// Every numeric input the run takes, from the command line first and the rc
|
|
131
|
+
// second, checked before anything is printed. Returns the checked strings.
|
|
132
|
+
// The registry door (spec 026 AC-10, F6): every target and the judge it will
|
|
133
|
+
// run with must be a model the registry knows. lib/models.js assertRegistered
|
|
134
|
+
// is the one rule; lib/run.js asks it too on the path every caller shares, and
|
|
135
|
+
// this door is where bin/driftproof says the registry's path in its own
|
|
136
|
+
// message, before the suite loads.
|
|
137
|
+
function checkRegistered(models, judgeFor) {
|
|
138
|
+
try {
|
|
139
|
+
for (const m of models) { assertRegistered(m, 'model'); assertRegistered(judgeFor(m), 'judge model'); }
|
|
140
|
+
} catch (e) {
|
|
141
|
+
if (e && e.code === 'UNREGISTERED_MODEL') { console.error(` ✗ REFUSED (${/judge/.test(e.message) ? 'judge-model / judge_model' : 'models'}): ${e.message}`); process.exit(2); }
|
|
142
|
+
throw e;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
function checkNumericInputs(flags, rc) {
|
|
146
|
+
const out = {};
|
|
147
|
+
for (const [flag, rcKey] of [['max-calls', 'max_calls'], ['max-usd', 'max_usd'], ['samples', 'samples'], ['max-cases', 'max_cases'], ['concurrency', 'concurrency']]) {
|
|
148
|
+
if (Object.prototype.hasOwnProperty.call(flags, flag)) out[flag] = checkInput(flag, flags[flag], 'flag --');
|
|
149
|
+
else if (rc[rcKey] !== undefined && rc[rcKey] !== null) out[flag] = checkInput(flag, rc[rcKey], '.driftproofrc', rcKey);
|
|
150
|
+
}
|
|
151
|
+
return out;
|
|
152
|
+
}
|
|
153
|
+
|
|
57
154
|
function parseArgs(argv) {
|
|
58
155
|
const positional = [];
|
|
59
156
|
const flags = {};
|
|
@@ -62,7 +159,7 @@ function parseArgs(argv) {
|
|
|
62
159
|
if (a.startsWith('--')) {
|
|
63
160
|
const key = a.slice(2);
|
|
64
161
|
const next = argv[i + 1];
|
|
65
|
-
if (next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
162
|
+
if (BOOLEAN_FLAGS.has(key) || next === undefined || next.startsWith('--')) { flags[key] = true; }
|
|
66
163
|
else { flags[key] = next; i++; }
|
|
67
164
|
} else positional.push(a);
|
|
68
165
|
}
|
|
@@ -76,7 +173,7 @@ USAGE
|
|
|
76
173
|
${PROJECT_NAME} init <dir> scaffold SKILL.md + evals/evals.json + .driftproofrc
|
|
77
174
|
${PROJECT_NAME} run <skill-dir> [--models a,b] [--samples N] [--max-cases N] [--max-calls N]
|
|
78
175
|
[--judge-model M] [--concurrency N] [--max-usd N]
|
|
79
|
-
[--keep-transcripts] [--out DIR]
|
|
176
|
+
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
80
177
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
81
178
|
${PROJECT_NAME} validate <receipt.json>
|
|
82
179
|
${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
|
|
@@ -88,6 +185,7 @@ ENV
|
|
|
88
185
|
ANTHROPIC_API_KEY required when CLAUDE_PROVIDER=api
|
|
89
186
|
DRIFTPROOF_REGISTRY path to an alternate model registry (default: the packaged config/models.json)
|
|
90
187
|
DRIFTPROOF_STUB=1 offline stub surface — canned receipts, zero model calls (for CI)
|
|
188
|
+
DRIFTPROOF_EVAL_USER the unix user the isolated hop runs the CLI as (default: driftproof-eval)
|
|
91
189
|
|
|
92
190
|
NOTES
|
|
93
191
|
- Default models list is 'haiku' only (cheap dev default).
|
|
@@ -107,11 +205,22 @@ NOTES
|
|
|
107
205
|
- --keep-transcripts writes the raw generations + judge outputs to
|
|
108
206
|
transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
|
|
109
207
|
local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
|
|
110
|
-
- Prices and registry status come from config/models.json
|
|
111
|
-
|
|
112
|
-
|
|
208
|
+
- Prices and registry status come from config/models.json (or the registry
|
|
209
|
+
DRIFTPROOF_REGISTRY names). An unregistered model id is REFUSED before any call,
|
|
210
|
+
naming the id and the registry path (so is one not shaped like a model id);
|
|
211
|
+
registry:"unregistered" is an import-only receipt value, refused for a run.
|
|
113
212
|
- --concurrency runs that many (case,mode) tasks at once (default 1). Higher
|
|
114
213
|
values cut wall-clock on the cli surface (cold-start dominated).
|
|
214
|
+
- ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
|
|
215
|
+
runs as a dedicated unprivileged unix user (DRIFTPROOF_EVAL_USER) through
|
|
216
|
+
sudo, from an EMPTY environment (only HOME and PATH, constructed), in a fresh
|
|
217
|
+
temp working directory removed after the call. A third-party SKILL.md is
|
|
218
|
+
instructions to an agent with tools; this is what keeps it away from your
|
|
219
|
+
keys, tokens and files. Operator prerequisite: that user, its own logged-in
|
|
220
|
+
CLIs, and a NOPASSWD sudoers rule (see RUNBOOK.md).
|
|
221
|
+
- --trusted-skill runs the legacy same-user path instead: your environment,
|
|
222
|
+
your cwd, your CLI login. FOR SKILLS YOU AUTHORED YOURSELF ONLY. Never pass
|
|
223
|
+
it for a skill fetched from anywhere else.
|
|
115
224
|
- import converts another tool's results into a valid receipt with HONEST
|
|
116
225
|
epistemics: verification_level DECLARED (never TESTED), surface "external",
|
|
117
226
|
source "imported/<tool>", and NO fabricated hashes. Imported receipts are
|
|
@@ -129,19 +238,53 @@ async function cmdRun(positional, flags) {
|
|
|
129
238
|
|
|
130
239
|
// Per-project defaults from .driftproofrc (if any); CLI flags override these.
|
|
131
240
|
const rc = loadRc(skillDir);
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
241
|
+
// The contract, at the door (spec 026 AC-9): every input checked against
|
|
242
|
+
// its rule before the projection and before any call. A value that fails
|
|
243
|
+
// its shape is refused naming the input and the value; nothing is defaulted.
|
|
244
|
+
if (INPUT_CONTRACT.rules.skill_dir_control.re.test(String(skillDir))) refuseInput('skill-dir', skillDir, INPUT_CONTRACT.rules.skill_dir_control.message, 'argument');
|
|
245
|
+
const checked = checkNumericInputs(flags, rc);
|
|
246
|
+
const modelsRaw = Object.prototype.hasOwnProperty.call(flags, 'models') ? flags.models : (rc.models !== undefined && rc.models !== null) ? rc.models : 'haiku';
|
|
247
|
+
const models = String(modelsRaw).split(',').map((s) => s.trim()).filter(Boolean);
|
|
136
248
|
const judgeModel = flags['judge-model'] || rc.judge_model || null;
|
|
137
|
-
|
|
138
|
-
|
|
249
|
+
// Spec 026 AC-10 (F6): every model id, and the judge, must be a model the
|
|
250
|
+
// registry knows; refused here naming the input, the id and the registry's
|
|
251
|
+
// absolute path, before the suite loads, before the projection, before any
|
|
252
|
+
// call. A shape failure is refused the same way (the contract's models rule
|
|
253
|
+
// below then never sees one). The same check stands in lib/run.js for every
|
|
254
|
+
// other caller.
|
|
255
|
+
const judgeFor = (m) => (judgeModel ? resolveModel(judgeModel) : resolveModel(m));
|
|
256
|
+
checkRegistered(models, judgeFor);
|
|
257
|
+
const modelsGiven = Object.prototype.hasOwnProperty.call(flags, 'models') ? checkInput('models', flags.models, 'flag --')
|
|
258
|
+
: (rc.models !== undefined && rc.models !== null) ? checkInput('models', rc.models, '.driftproofrc', 'models') : 'haiku';
|
|
259
|
+
void modelsGiven;
|
|
260
|
+
const maxCases = checked['max-cases'] !== undefined ? parseInt(checked['max-cases'], 10) : null;
|
|
261
|
+
const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
|
|
262
|
+
const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : DEFAULT_JUDGE_SAMPLES;
|
|
263
|
+
const concurrency = checked.concurrency !== undefined ? parseInt(checked.concurrency, 10) : 1;
|
|
264
|
+
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
139
265
|
const keepTranscripts = !!flags['keep-transcripts'];
|
|
266
|
+
// spec 022: the same-user spawn exists only behind this flag.
|
|
267
|
+
//
|
|
268
|
+
// THE EVAL USER IS NOT RESOLVED HERE (F-022-4, spec 021). It used to be, on
|
|
269
|
+
// every run that was not --trusted-skill, including an api-surface-only run
|
|
270
|
+
// that spawns no CLI at all - so a hostile DRIFTPROOF_EVAL_USER refused a run
|
|
271
|
+
// that would never have read it. Fail-loud and harmless, and still wrong: a
|
|
272
|
+
// variable that governs a lane this run does not take should be neither read
|
|
273
|
+
// nor validated. It is resolved below, once a subscription surface is actually
|
|
274
|
+
// in the run, which is the same condition the isolation banner already used.
|
|
275
|
+
const trusted = !!flags['trusted-skill'];
|
|
140
276
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
141
277
|
fs.mkdirSync(outDir, { recursive: true });
|
|
142
278
|
|
|
143
279
|
const skill = loadSkill(skillDir);
|
|
144
280
|
const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
|
|
281
|
+
// Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
|
|
282
|
+
// over nothing is a receipt about nothing. Refused here, before the
|
|
283
|
+
// projection and before any call, naming the suite file.
|
|
284
|
+
if (nCases < 1) {
|
|
285
|
+
console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')} has no cases${maxCases ? ` (after --max-cases ${maxCases})` : ''}; nothing to measure, no receipt written.`);
|
|
286
|
+
process.exit(2);
|
|
287
|
+
}
|
|
145
288
|
// v0.5 draws the generation up to SAMPLING.max times per arm, so BOTH the
|
|
146
289
|
// printed projection and the dollar guard below must be scaled by it. Left at
|
|
147
290
|
// draws=1 the guard would admit a run costing up to ten times its own
|
|
@@ -155,7 +298,13 @@ async function cmdRun(positional, flags) {
|
|
|
155
298
|
// `api` surface this is real spend, so we refuse if it would exceed --max-usd.
|
|
156
299
|
// On `claude-cli` the metered spend is $0 (subscription); the figure is the
|
|
157
300
|
// hypothetical "if run on the metered API" cost — printed, never blocks.
|
|
158
|
-
|
|
301
|
+
// Spec 026 AC-10: the projection is priced with the judge the run will USE
|
|
302
|
+
// (lib/run.js: --judge-model when given, else the target itself), never a
|
|
303
|
+
// fixed haiku; each target is priced with its own judge and the figures summed.
|
|
304
|
+
const targets = models.map((m) => resolveModel(m));
|
|
305
|
+
const judges = models.map((m) => judgeFor(m));
|
|
306
|
+
const perModelCost = targets.map((t, i) => estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: [targets[i]], judgeModel: judges[i] }));
|
|
307
|
+
const cost = { totalUSD: Math.round(perModelCost.reduce((a, c) => a + c.totalUSD, 0) * 1e4) / 1e4, perModel: perModelCost.flatMap((c) => c.perModel), judges };
|
|
159
308
|
// Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
|
|
160
309
|
const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
|
|
161
310
|
const surface = surfaces.join(', ');
|
|
@@ -163,9 +312,17 @@ async function cmdRun(positional, flags) {
|
|
|
163
312
|
|
|
164
313
|
const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
|
|
165
314
|
console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
|
|
315
|
+
// Spec 026 AC-1: a stub run says so before it starts, and its receipt says so
|
|
316
|
+
// after (surface stub, answered_by.kind stub, UNVERIFIED).
|
|
317
|
+
if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer; the receipt will be UNVERIFIED and measure nothing');
|
|
166
318
|
console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
|
|
167
319
|
console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
|
|
168
320
|
console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
|
|
321
|
+
console.log(` judge: ${[...new Set(cost.judges)].join(', ')} (${judgeModel ? '--judge-model' : 'the target model itself; set --judge-model to change'})`);
|
|
322
|
+
if (subSurfaces.length) {
|
|
323
|
+
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
324
|
+
console.log(` isolation: ${isolation}`);
|
|
325
|
+
}
|
|
169
326
|
console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
|
|
170
327
|
console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
171
328
|
if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
|
|
@@ -200,13 +357,19 @@ async function cmdRun(positional, flags) {
|
|
|
200
357
|
skill,
|
|
201
358
|
model,
|
|
202
359
|
opts: {
|
|
203
|
-
maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts,
|
|
360
|
+
maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts, trusted,
|
|
204
361
|
onProgress: (p) => {
|
|
205
362
|
if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`);
|
|
206
363
|
},
|
|
207
364
|
},
|
|
208
365
|
});
|
|
209
366
|
} catch (e) {
|
|
367
|
+
if (e && e.code === 'SUBSTRATE_MISMATCH') {
|
|
368
|
+
// Spec 026 AC-2: the surface answered as a different model. The run
|
|
369
|
+
// stopped before its next call and wrote no receipt.
|
|
370
|
+
console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`);
|
|
371
|
+
process.exit(5);
|
|
372
|
+
}
|
|
210
373
|
if (e && e.code === 'BUDGET_HARDSTOP') {
|
|
211
374
|
console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
|
|
212
375
|
console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
|
|
@@ -244,7 +407,14 @@ async function cmdRun(positional, flags) {
|
|
|
244
407
|
const aw = receipt.results.aggregates.with_skill;
|
|
245
408
|
const ab = receipt.results.aggregates.baseline;
|
|
246
409
|
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
|
|
247
|
-
console.log(` →
|
|
410
|
+
console.log(` → ${answeredLine(receipt)}; verification_level ${receipt.verification_level}`);
|
|
411
|
+
// Spec 026 AC-8: how many draws were cut at the output cap and excluded.
|
|
412
|
+
const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
|
|
413
|
+
if (nTruncated) console.log(` → ${nTruncated} draw(s) truncated at the output cap: unmeasured, never judged, excluded from every band`);
|
|
414
|
+
// Spec 026 AC-6, AC-7: the band rule is named beside the band, and a band
|
|
415
|
+
// the formula could not form prints as n/a, never as 0.000.
|
|
416
|
+
if (cmp.delta == null) console.log(` → skill lift n/a (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})\n`);
|
|
417
|
+
else console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${uncertaintyStr(cmp)} (with ${band(aw.mean_score, aw.stddev)} vs base ${band(ab.mean_score, ab.stddev)}; band = sample stddev of the per-case means)\n`);
|
|
248
418
|
}
|
|
249
419
|
|
|
250
420
|
console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
|
|
@@ -256,10 +426,8 @@ function cmdDiff(positional, flags) {
|
|
|
256
426
|
const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
|
|
257
427
|
const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
|
|
258
428
|
|
|
259
|
-
//
|
|
260
|
-
for (const [p, r] of [[aPath, a], [bPath, b]])
|
|
261
|
-
if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
|
|
262
|
-
}
|
|
429
|
+
// Spec 026 AC-20: a side whose hash does not verify is refused, naming it.
|
|
430
|
+
for (const [p, r] of [[aPath, a], [bPath, b]]) refuseUnverified(r, p, 'diff');
|
|
263
431
|
|
|
264
432
|
// --mode revision inverts the axis: the skill text is the variable under test
|
|
265
433
|
// and the substrate is the control. The fields release drift merely warns about
|
|
@@ -338,6 +506,7 @@ function cmdBadge(positional, flags) {
|
|
|
338
506
|
const p = positional[0];
|
|
339
507
|
if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
|
|
340
508
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
509
|
+
refuseUnverified(receipt, p, 'badge');
|
|
341
510
|
if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
|
|
342
511
|
const badge = badgeEndpoint(receipt);
|
|
343
512
|
const json = JSON.stringify(badge, null, 2);
|
|
@@ -388,7 +557,7 @@ function cmdExport(positional, flags) {
|
|
|
388
557
|
if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
|
|
389
558
|
const { toSummaryJson } = require('../lib/export');
|
|
390
559
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
391
|
-
|
|
560
|
+
refuseUnverified(receipt, p, 'export');
|
|
392
561
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
393
562
|
const json = JSON.stringify(summary, null, 2);
|
|
394
563
|
if (flags.out) {
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.9.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -29,7 +29,30 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
29
29
|
// at run time so derived dollars stay reproducible), and the derived
|
|
30
30
|
// `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
|
|
31
31
|
// v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
|
|
32
|
-
|
|
32
|
+
// v0.5 samples the GENERATION n times per arm (per-case `generation` draw list,
|
|
33
|
+
// the across-draw band, the variance ratio, the sampling policy applied)
|
|
34
|
+
// and adds the per-suite canary. v0.4 is frozen as receipt.v0.4.schema.json.
|
|
35
|
+
// v0.6 (spec 026, receipt integrity) says WHAT ANSWERED: run.answered_by, a
|
|
36
|
+
// `stub` surface, run.judge.model_id + prompt_template_hash, per-draw
|
|
37
|
+
// stop_reason/truncated and n_truncated, case_status failed_unmeasured,
|
|
38
|
+
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
|
+
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
|
+
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.6';
|
|
42
|
+
|
|
43
|
+
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
|
+
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|
|
45
|
+
// case, prompt or rubric bound, so a 2 MB prompt went to the model at the
|
|
46
|
+
// fixed 250-token projection; lib/checks.js compiled a suite-supplied regex
|
|
47
|
+
// with no bound, and JavaScript cannot time a regex out. Each is set well
|
|
48
|
+
// above every suite this repository holds: they bind pathological input, not
|
|
49
|
+
// the archive. Named in the error a bound raises, and in AUTHORING.md.
|
|
50
|
+
const SKILL_MAX_FILES = 2000; // bundled files under a skill directory
|
|
51
|
+
const SKILL_MAX_BYTES = 32 * 1024 * 1024; // bundled bytes, all files together
|
|
52
|
+
const SKILL_MAX_DEPTH = 16; // directory depth below the skill dir
|
|
53
|
+
const SUITE_MAX_CASES = 500; // cases in one evals.json
|
|
54
|
+
const CASE_MAX_CHARS = 65536; // characters in one prompt or rubric
|
|
55
|
+
const CHECK_MAX_PATTERN = 256; // characters in one checks[] regex
|
|
33
56
|
|
|
34
57
|
// Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
|
|
35
58
|
// these. The projection is refused before any call if it exceeds the cap, on
|
|
@@ -133,6 +156,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
|
133
156
|
|
|
134
157
|
module.exports = {
|
|
135
158
|
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
159
|
+
SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
|
|
136
160
|
EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
137
161
|
GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
|
|
138
162
|
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|
package/lib/checks.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const { CHECK_MAX_PATTERN } = require('../config');
|
|
5
|
+
|
|
4
6
|
// Deterministic post-checks.
|
|
5
7
|
//
|
|
6
8
|
// A per-case eval suite may declare optional `checks[]`: structural / regex
|
|
@@ -19,6 +21,22 @@
|
|
|
19
21
|
// contains — the output includes the literal `value` substring
|
|
20
22
|
// not_contains — the output does NOT include the literal `value` substring
|
|
21
23
|
// min_length — the trimmed output is at least `value` characters long
|
|
24
|
+
//
|
|
25
|
+
// BOUNDED (spec 026 AC-13, audit A7). A suite's regex is third-party input
|
|
26
|
+
// compiled and run against model output, and JavaScript cannot time a regex
|
|
27
|
+
// out. A pattern longer than CHECK_MAX_PATTERN, or one carrying a nested
|
|
28
|
+
// quantifier of the (x+)+ family (a quantified group whose body ends in a
|
|
29
|
+
// quantifier: the shape that is exponential on every backtracking engine),
|
|
30
|
+
// is REFUSED: recorded pass: false with a reason that says so, never
|
|
31
|
+
// evaluated. The test is syntactic and narrow by design; it is not a general
|
|
32
|
+
// ReDoS detector, and the spec says so.
|
|
33
|
+
const NESTED_QUANTIFIER = /\((?:[^()\\]|\\.)*[+*}]\)\s*[+*]|\((?:[^()\\]|\\.)*\|(?:[^()\\]|\\.)*\)\s*[+*]/;
|
|
34
|
+
function refusedPattern(pattern) {
|
|
35
|
+
const p = String(pattern == null ? '' : pattern);
|
|
36
|
+
if (p.length > CHECK_MAX_PATTERN) return `refused: pattern of ${p.length} characters exceeds CHECK_MAX_PATTERN (${CHECK_MAX_PATTERN})`;
|
|
37
|
+
if (NESTED_QUANTIFIER.test(p)) return 'refused: pattern carries a nested quantifier of the (x+)+ family, which is exponential to evaluate';
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
22
40
|
|
|
23
41
|
function runOneCheck(check, output) {
|
|
24
42
|
const text = String(output || '');
|
|
@@ -40,11 +58,15 @@ function runOneCheck(check, output) {
|
|
|
40
58
|
// `checks`). Empty array when the case declares no checks.
|
|
41
59
|
function runChecks(output, checks) {
|
|
42
60
|
if (!Array.isArray(checks) || !checks.length) return [];
|
|
43
|
-
return checks.map((c) =>
|
|
44
|
-
|
|
45
|
-
kind: c && c.kind,
|
|
46
|
-
|
|
47
|
-
|
|
61
|
+
return checks.map((c) => {
|
|
62
|
+
const reason = c && c.kind === 'regex' ? refusedPattern(c.pattern) : null;
|
|
63
|
+
if (reason) return { name: String((c && c.name) || (c && c.kind) || 'check'), kind: c && c.kind, pass: false, reason };
|
|
64
|
+
return {
|
|
65
|
+
name: String((c && c.name) || (c && c.kind) || 'check'),
|
|
66
|
+
kind: c && c.kind,
|
|
67
|
+
pass: !!runOneCheck(c, output),
|
|
68
|
+
};
|
|
69
|
+
});
|
|
48
70
|
}
|
|
49
71
|
|
|
50
|
-
module.exports = { runChecks, runOneCheck };
|
|
72
|
+
module.exports = { runChecks, runOneCheck, refusedPattern, NESTED_QUANTIFIER };
|