driftproof 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -211,6 +211,36 @@ Writing a suite that measures fairly is its own craft — see
211
211
 
212
212
  The surface and judge settings used are recorded in every receipt.
213
213
 
214
+ ### Isolation (the cli surfaces)
215
+
216
+ A `SKILL.md` you did not write is instructions to an agent that has tools. On
217
+ the two cli surfaces driftproof therefore never runs `claude` or `codex` as you.
218
+ Every spawn goes through a hop to a dedicated unprivileged unix user, from an
219
+ empty environment:
220
+
221
+ ```
222
+ /usr/bin/sudo -n -u driftproof-eval /usr/bin/env -i HOME=<its home> PATH=<its ~/.local/bin>:/usr/bin:/bin bash -lc '<wrapper>' driftproof-eval-hop claude -p ...
223
+ ```
224
+
225
+ Only `HOME` and `PATH` reach the child, both constructed from the user name, none
226
+ copied from your shell. The working directory is a fresh temp directory made for
227
+ that one call and removed after it, including on timeout. The CLI argv is passed
228
+ as separate arguments, never as a shell string. What the evaluated agent cannot
229
+ see: your environment (tokens, keys), your home directory (SSH keys, CLI session
230
+ files), your current directory, and your CLI configuration.
231
+
232
+ **Operator prerequisite, set up once** (see `RUNBOOK.md`): that user exists with a
233
+ mode-700 home and no extra groups, has its own logged-in `claude` and `codex`
234
+ under `~/.local/bin`, and a sudoers rule lets you run commands as it with
235
+ `NOPASSWD`. `DRIFTPROOF_EVAL_USER` names the user (default `driftproof-eval`).
236
+ Without the prerequisite a run fails at once, before any call, and says so.
237
+
238
+ **`--trusted-skill`** runs the legacy same-user path instead: your environment,
239
+ your cwd, your CLI login. It exists for exactly one case, a skill you authored
240
+ yourself on a machine with no eval user. Never pass it for a skill fetched from
241
+ anywhere else. The api surfaces are unaffected either way; they make HTTP calls
242
+ and spawn nothing.
243
+
214
244
  ### Cost guard
215
245
 
216
246
  Sampling multiplies calls on two axes, and the second one is easy to miss. Each
@@ -232,7 +262,7 @@ A receipt is the unit of evidence — one JSON document conforming to
232
262
 
233
263
  ```jsonc
234
264
  {
235
- "schema_version": "0.5",
265
+ "schema_version": "0.6",
236
266
  "skill": { "name": "commit-message-conventions", "version": "0.2.0",
237
267
  "content_hash": "…sha256 over SKILL.md + bundled files…" },
238
268
  "suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
@@ -241,12 +271,16 @@ A receipt is the unit of evidence — one JSON document conforming to
241
271
  "model_release_date": "2025-10-01",
242
272
  "provider": "anthropic",
243
273
  "surface": "claude-cli",
244
- "runner_version": "0.8.0",
274
+ "runner_version": "0.9.0",
245
275
  "date_utc": "2026-07-27T…Z",
246
276
  "registry": "registered",
247
277
  "transcripts": "hashes-only",
248
278
  "judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
249
- "surface": "claude-cli" }
279
+ "surface": "claude-cli", "model_id": "claude-haiku-4-5-20251001",
280
+ "prompt_template_hash": "…sha256 over the grading template…" },
281
+ "answered_by": { "kind": "model", "attested": true,
282
+ "reported_model": "claude-haiku-4-5-20251001",
283
+ "reported_models": ["claude-haiku-4-5"], "isolation": "eval-user" }
250
284
  },
251
285
  "results": {
252
286
  "cases": [
@@ -315,7 +349,7 @@ jobs:
315
349
  runs-on: ubuntu-latest
316
350
  steps:
317
351
  - uses: actions/checkout@v4
318
- - uses: driftproofhq/driftproof@v0.8.0
352
+ - uses: driftproofhq/driftproof@v0.9.0
319
353
  with:
320
354
  skill-dir: skills/my-skill
321
355
  models: claude-haiku-4-5
@@ -331,7 +365,13 @@ fails the job on `REGRESSED` unless you set `fail-on-regression: 'false'`. For a
331
365
  free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job env —
332
366
  the runner returns canned receipts so the wiring can be tested without spend (this
333
367
  is exactly how the action's own [self-test](.github/workflows/action-selftest.yml)
334
- runs).
368
+ runs). A stub receipt says what it is: `verification_level` `UNVERIFIED`,
369
+ `run.surface` `stub`, `run.answered_by.kind` `stub`, and the verdict on it is
370
+ `NOT_MEASURED` — it proves the wiring, never the skill. That self-test is also the proof behind the input hardening: on every
371
+ run it passes one hostile value (a quote, a semicolon, `$(...)`, a backtick and
372
+ a newline) through each of the action's five inputs on the real runner and
373
+ fails unless every one is refused before anything ran, so a green check there
374
+ is a claim you can read, not one you have to take.
335
375
 
336
376
  ### Badge
337
377
 
package/bin/driftproof CHANGED
@@ -6,13 +6,13 @@ const fs = require('fs');
6
6
  const path = require('path');
7
7
  const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MAX_CALLS } = require('../config');
8
8
  const { loadSkill } = require('../lib/skill');
9
- const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
9
+ const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
10
10
  const { SAMPLING } = require('../lib/sampling');
11
11
  const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
12
12
  const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
13
- const { surfaceForModel, isSubscriptionSurface, resolveModel } = require('../lib/provider');
13
+ const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
14
14
  const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
15
- const { registryStatus } = require('../lib/models');
15
+ const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
16
16
  const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
17
17
  const { scaffoldInit } = require('../lib/init');
18
18
 
@@ -40,6 +40,24 @@ function writeTranscripts(receipt, transcripts) {
40
40
 
41
41
  const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
42
42
 
43
+ // ── THE DISPLAY VERIFIES BEFORE IT READS (spec 026 AC-20, F7) ────────────────
44
+ // badge, diff and export read a receipt somebody else may have edited. Until
45
+ // this, badge never checked the hash and diff and export printed a warning
46
+ // and proceeded, so a receipt with one digit moved and the hash left as it
47
+ // was rendered `passing` (028's D-5, for the CLI). A receipt whose
48
+ // receipt_hash does not verify is refused here: nothing on stdout, no file,
49
+ // the reason on stderr naming receipt_hash and the file, exit 4. The bound the
50
+ // threat model states does not move: the same edit followed by a re-seal
51
+ // (sealReceipt) verifies and renders, which is what receipt_hash is: integrity
52
+ // since sealing, not authenticity. The check lives here, in the three
53
+ // commands, and not in lib/verdict.js's writer, which the repo gate calls on
54
+ // synthetic unsealed receipts.
55
+ function refuseUnverified(receipt, file, command) {
56
+ if (verifyReceiptHash(receipt)) return;
57
+ console.error(` ✗ REFUSED (${command}): receipt_hash does not verify for ${path.basename(file)} (tampered or hand-edited since it was sealed); nothing rendered.`);
58
+ process.exit(4);
59
+ }
60
+
43
61
  // Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
44
62
  // in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
45
63
  // flags always win over the rc; the rc wins over built-in defaults.
@@ -54,6 +72,85 @@ function loadRc(skillDir) {
54
72
  return merged;
55
73
  }
56
74
 
75
+ // Flags that never take a value, so `--trusted-skill <skill-dir>` keeps the dir
76
+ // positional instead of swallowing it as the flag's value.
77
+ const BOOLEAN_FLAGS = new Set(['trusted-skill']);
78
+
79
+ // ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
80
+ // A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
81
+ // them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
82
+ // same copy and AC-9 asserts the three identical, rule for rule, both ways.
83
+ // Consolidating into one shared module is 028's decision 1, the spec after 026.
84
+ //
85
+ // Why a contract at the door: `parseInt('abc', 10)` is NaN, both cost guards
86
+ // are `>` comparisons against it, both are false, and the run completed and
87
+ // wrote a receipt that carried no trace of it (the dollar cap never compared;
88
+ // the call cap silently became the dev default). `--samples abc` was quieter
89
+ // still: `NaN || 5` is 5. Every numeric input, from the command line or the
90
+ // rc, is checked against its rule BEFORE the projection is printed and before
91
+ // any call; a flag given with no value (the parser records `true`) and an
92
+ // empty string are refused the same way, not defaulted.
93
+ const INPUT_CONTRACT = {
94
+ rules: {
95
+ models: { re: /^[A-Za-z0-9._-]+(,[A-Za-z0-9._-]+)*$/, message: 'models: expected a comma-separated list of model ids (letters, digits, . _ -)' },
96
+ max_usd: { re: /^[0-9]+(\.[0-9]+)?$/, message: 'max-usd: expected a positive decimal number such as 20 or 3.53' },
97
+ max_usd_zero: { re: /^0+(\.0+)?$/, reject_on_match: true, message: 'max-usd: must be greater than zero' },
98
+ max_calls: { re: /^[1-9][0-9]*$/, message: 'max-calls: expected a positive integer with no leading zero' },
99
+ skill_dir_control: { re: /[\x00-\x1f\x7f]/, reject_on_match: true, message: 'skill-dir: contains a control character' },
100
+ },
101
+ // Which rule each `driftproof run` input is checked against. samples,
102
+ // max-cases and concurrency are positive integers and reuse max_calls' rule.
103
+ flags: { 'models': ['models'], 'max-usd': ['max_usd', 'max_usd_zero'], 'max-calls': ['max_calls'], 'samples': ['max_calls'], 'max-cases': ['max_calls'], 'concurrency': ['max_calls'], 'skill-dir': ['skill_dir_control'] },
104
+ };
105
+
106
+ function shownValue(v) {
107
+ if (v === true) return '<no value>';
108
+ return JSON.stringify(String(v));
109
+ }
110
+ // Refuse one input: the rule's own message, the input's name and the value it
111
+ // carried, exit 2. Nothing has been printed or spent when this runs.
112
+ function refuseInput(name, value, message, where) {
113
+ console.error(` ✗ REFUSED: ${message}, got ${shownValue(value)} (${where} ${name}); nothing run, no receipt written.`);
114
+ process.exit(2);
115
+ }
116
+ // Check one input against every rule its flag maps to. `value` is what was
117
+ // given: a string, a number (from the rc), `true` (a flag with no value) or
118
+ // an empty string; only a string that matches every rule passes.
119
+ function checkInput(flag, value, where, shownName = flag) {
120
+ const rules = INPUT_CONTRACT.flags[flag] || [];
121
+ const text = typeof value === 'string' ? value : (typeof value === 'number' && Number.isFinite(value)) ? String(value) : null;
122
+ if (text === null) refuseInput(shownName, value, `${flag}: expected a value`, where);
123
+ for (const key of rules) {
124
+ const rule = INPUT_CONTRACT.rules[key];
125
+ const hit = rule.re.test(text);
126
+ if (rule.reject_on_match ? hit : !hit) refuseInput(shownName, value, rule.message, where);
127
+ }
128
+ return text;
129
+ }
130
+ // Every numeric input the run takes, from the command line first and the rc
131
+ // second, checked before anything is printed. Returns the checked strings.
132
+ // The registry door (spec 026 AC-10, F6): every target and the judge it will
133
+ // run with must be a model the registry knows. lib/models.js assertRegistered
134
+ // is the one rule; lib/run.js asks it too on the path every caller shares, and
135
+ // this door is where bin/driftproof says the registry's path in its own
136
+ // message, before the suite loads.
137
+ function checkRegistered(models, judgeFor) {
138
+ try {
139
+ for (const m of models) { assertRegistered(m, 'model'); assertRegistered(judgeFor(m), 'judge model'); }
140
+ } catch (e) {
141
+ if (e && e.code === 'UNREGISTERED_MODEL') { console.error(` ✗ REFUSED (${/judge/.test(e.message) ? 'judge-model / judge_model' : 'models'}): ${e.message}`); process.exit(2); }
142
+ throw e;
143
+ }
144
+ }
145
+ function checkNumericInputs(flags, rc) {
146
+ const out = {};
147
+ for (const [flag, rcKey] of [['max-calls', 'max_calls'], ['max-usd', 'max_usd'], ['samples', 'samples'], ['max-cases', 'max_cases'], ['concurrency', 'concurrency']]) {
148
+ if (Object.prototype.hasOwnProperty.call(flags, flag)) out[flag] = checkInput(flag, flags[flag], 'flag --');
149
+ else if (rc[rcKey] !== undefined && rc[rcKey] !== null) out[flag] = checkInput(flag, rc[rcKey], '.driftproofrc', rcKey);
150
+ }
151
+ return out;
152
+ }
153
+
57
154
  function parseArgs(argv) {
58
155
  const positional = [];
59
156
  const flags = {};
@@ -62,7 +159,7 @@ function parseArgs(argv) {
62
159
  if (a.startsWith('--')) {
63
160
  const key = a.slice(2);
64
161
  const next = argv[i + 1];
65
- if (next === undefined || next.startsWith('--')) { flags[key] = true; }
162
+ if (BOOLEAN_FLAGS.has(key) || next === undefined || next.startsWith('--')) { flags[key] = true; }
66
163
  else { flags[key] = next; i++; }
67
164
  } else positional.push(a);
68
165
  }
@@ -76,7 +173,7 @@ USAGE
76
173
  ${PROJECT_NAME} init <dir> scaffold SKILL.md + evals/evals.json + .driftproofrc
77
174
  ${PROJECT_NAME} run <skill-dir> [--models a,b] [--samples N] [--max-cases N] [--max-calls N]
78
175
  [--judge-model M] [--concurrency N] [--max-usd N]
79
- [--keep-transcripts] [--out DIR]
176
+ [--keep-transcripts] [--out DIR] [--trusted-skill]
80
177
  ${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
81
178
  ${PROJECT_NAME} validate <receipt.json>
82
179
  ${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
@@ -88,6 +185,7 @@ ENV
88
185
  ANTHROPIC_API_KEY required when CLAUDE_PROVIDER=api
89
186
  DRIFTPROOF_REGISTRY path to an alternate model registry (default: the packaged config/models.json)
90
187
  DRIFTPROOF_STUB=1 offline stub surface — canned receipts, zero model calls (for CI)
188
+ DRIFTPROOF_EVAL_USER the unix user the isolated hop runs the CLI as (default: driftproof-eval)
91
189
 
92
190
  NOTES
93
191
  - Default models list is 'haiku' only (cheap dev default).
@@ -107,11 +205,22 @@ NOTES
107
205
  - --keep-transcripts writes the raw generations + judge outputs to
108
206
  transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
109
207
  local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
110
- - Prices and registry status come from config/models.json; an unregistered
111
- model still runs but is marked registry:"unregistered" and costed at the
112
- conservative default price.
208
+ - Prices and registry status come from config/models.json (or the registry
209
+ DRIFTPROOF_REGISTRY names). An unregistered model id is REFUSED before any call,
210
+ naming the id and the registry path (so is one not shaped like a model id);
211
+ registry:"unregistered" is an import-only receipt value, refused for a run.
113
212
  - --concurrency runs that many (case,mode) tasks at once (default 1). Higher
114
213
  values cut wall-clock on the cli surface (cold-start dominated).
214
+ - ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
215
+ runs as a dedicated unprivileged unix user (DRIFTPROOF_EVAL_USER) through
216
+ sudo, from an EMPTY environment (only HOME and PATH, constructed), in a fresh
217
+ temp working directory removed after the call. A third-party SKILL.md is
218
+ instructions to an agent with tools; this is what keeps it away from your
219
+ keys, tokens and files. Operator prerequisite: that user, its own logged-in
220
+ CLIs, and a NOPASSWD sudoers rule (see RUNBOOK.md).
221
+ - --trusted-skill runs the legacy same-user path instead: your environment,
222
+ your cwd, your CLI login. FOR SKILLS YOU AUTHORED YOURSELF ONLY. Never pass
223
+ it for a skill fetched from anywhere else.
115
224
  - import converts another tool's results into a valid receipt with HONEST
116
225
  epistemics: verification_level DECLARED (never TESTED), surface "external",
117
226
  source "imported/<tool>", and NO fabricated hashes. Imported receipts are
@@ -129,19 +238,53 @@ async function cmdRun(positional, flags) {
129
238
 
130
239
  // Per-project defaults from .driftproofrc (if any); CLI flags override these.
131
240
  const rc = loadRc(skillDir);
132
- const models = String(flags.models || rc.models || 'haiku').split(',').map((s) => s.trim()).filter(Boolean);
133
- const maxCases = flags['max-cases'] ? parseInt(flags['max-cases'], 10) : (rc.max_cases != null ? parseInt(rc.max_cases, 10) : null);
134
- const maxCalls = flags['max-calls'] ? parseInt(flags['max-calls'], 10) : (rc.max_calls != null ? parseInt(rc.max_calls, 10) : DEV_MAX_CALLS);
135
- const samples = flags.samples ? parseInt(flags.samples, 10) : (rc.samples != null ? parseInt(rc.samples, 10) : DEFAULT_JUDGE_SAMPLES);
241
+ // The contract, at the door (spec 026 AC-9): every input checked against
242
+ // its rule before the projection and before any call. A value that fails
243
+ // its shape is refused naming the input and the value; nothing is defaulted.
244
+ if (INPUT_CONTRACT.rules.skill_dir_control.re.test(String(skillDir))) refuseInput('skill-dir', skillDir, INPUT_CONTRACT.rules.skill_dir_control.message, 'argument');
245
+ const checked = checkNumericInputs(flags, rc);
246
+ const modelsRaw = Object.prototype.hasOwnProperty.call(flags, 'models') ? flags.models : (rc.models !== undefined && rc.models !== null) ? rc.models : 'haiku';
247
+ const models = String(modelsRaw).split(',').map((s) => s.trim()).filter(Boolean);
136
248
  const judgeModel = flags['judge-model'] || rc.judge_model || null;
137
- const concurrency = flags.concurrency ? parseInt(flags.concurrency, 10) : (rc.concurrency != null ? parseInt(rc.concurrency, 10) : 1);
138
- const maxUsd = flags['max-usd'] ? parseFloat(flags['max-usd']) : (rc.max_usd != null ? parseFloat(rc.max_usd) : DEV_MAX_USD);
249
+ // Spec 026 AC-10 (F6): every model id, and the judge, must be a model the
250
+ // registry knows; refused here naming the input, the id and the registry's
251
+ // absolute path, before the suite loads, before the projection, before any
252
+ // call. A shape failure is refused the same way (the contract's models rule
253
+ // below then never sees one). The same check stands in lib/run.js for every
254
+ // other caller.
255
+ const judgeFor = (m) => (judgeModel ? resolveModel(judgeModel) : resolveModel(m));
256
+ checkRegistered(models, judgeFor);
257
+ const modelsGiven = Object.prototype.hasOwnProperty.call(flags, 'models') ? checkInput('models', flags.models, 'flag --')
258
+ : (rc.models !== undefined && rc.models !== null) ? checkInput('models', rc.models, '.driftproofrc', 'models') : 'haiku';
259
+ void modelsGiven;
260
+ const maxCases = checked['max-cases'] !== undefined ? parseInt(checked['max-cases'], 10) : null;
261
+ const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
262
+ const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : DEFAULT_JUDGE_SAMPLES;
263
+ const concurrency = checked.concurrency !== undefined ? parseInt(checked.concurrency, 10) : 1;
264
+ const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
139
265
  const keepTranscripts = !!flags['keep-transcripts'];
266
+ // spec 022: the same-user spawn exists only behind this flag.
267
+ //
268
+ // THE EVAL USER IS NOT RESOLVED HERE (F-022-4, spec 021). It used to be, on
269
+ // every run that was not --trusted-skill, including an api-surface-only run
270
+ // that spawns no CLI at all - so a hostile DRIFTPROOF_EVAL_USER refused a run
271
+ // that would never have read it. Fail-loud and harmless, and still wrong: a
272
+ // variable that governs a lane this run does not take should be neither read
273
+ // nor validated. It is resolved below, once a subscription surface is actually
274
+ // in the run, which is the same condition the isolation banner already used.
275
+ const trusted = !!flags['trusted-skill'];
140
276
  const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
141
277
  fs.mkdirSync(outDir, { recursive: true });
142
278
 
143
279
  const skill = loadSkill(skillDir);
144
280
  const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
281
+ // Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
282
+ // over nothing is a receipt about nothing. Refused here, before the
283
+ // projection and before any call, naming the suite file.
284
+ if (nCases < 1) {
285
+ console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')} has no cases${maxCases ? ` (after --max-cases ${maxCases})` : ''}; nothing to measure, no receipt written.`);
286
+ process.exit(2);
287
+ }
145
288
  // v0.5 draws the generation up to SAMPLING.max times per arm, so BOTH the
146
289
  // printed projection and the dollar guard below must be scaled by it. Left at
147
290
  // draws=1 the guard would admit a run costing up to ten times its own
@@ -155,7 +298,13 @@ async function cmdRun(positional, flags) {
155
298
  // `api` surface this is real spend, so we refuse if it would exceed --max-usd.
156
299
  // On `claude-cli` the metered spend is $0 (subscription); the figure is the
157
300
  // hypothetical "if run on the metered API" cost — printed, never blocks.
158
- const cost = estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: models.map((m) => require('../lib/provider').resolveModel(m)), judgeModel: judgeModel || 'haiku' });
301
+ // Spec 026 AC-10: the projection is priced with the judge the run will USE
302
+ // (lib/run.js: --judge-model when given, else the target itself), never a
303
+ // fixed haiku; each target is priced with its own judge and the figures summed.
304
+ const targets = models.map((m) => resolveModel(m));
305
+ const judges = models.map((m) => judgeFor(m));
306
+ const perModelCost = targets.map((t, i) => estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: [targets[i]], judgeModel: judges[i] }));
307
+ const cost = { totalUSD: Math.round(perModelCost.reduce((a, c) => a + c.totalUSD, 0) * 1e4) / 1e4, perModel: perModelCost.flatMap((c) => c.perModel), judges };
159
308
  // Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
160
309
  const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
161
310
  const surface = surfaces.join(', ');
@@ -163,9 +312,17 @@ async function cmdRun(positional, flags) {
163
312
 
164
313
  const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
165
314
  console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
315
+ // Spec 026 AC-1: a stub run says so before it starts, and its receipt says so
316
+ // after (surface stub, answered_by.kind stub, UNVERIFIED).
317
+ if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer; the receipt will be UNVERIFIED and measure nothing');
166
318
  console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
167
319
  console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
168
320
  console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
321
+ console.log(` judge: ${[...new Set(cost.judges)].join(', ')} (${judgeModel ? '--judge-model' : 'the target model itself; set --judge-model to change'})`);
322
+ if (subSurfaces.length) {
323
+ const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
324
+ console.log(` isolation: ${isolation}`);
325
+ }
169
326
  console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
170
327
  console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
171
328
  if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
@@ -200,13 +357,19 @@ async function cmdRun(positional, flags) {
200
357
  skill,
201
358
  model,
202
359
  opts: {
203
- maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts,
360
+ maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts, trusted,
204
361
  onProgress: (p) => {
205
362
  if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`);
206
363
  },
207
364
  },
208
365
  });
209
366
  } catch (e) {
367
+ if (e && e.code === 'SUBSTRATE_MISMATCH') {
368
+ // Spec 026 AC-2: the surface answered as a different model. The run
369
+ // stopped before its next call and wrote no receipt.
370
+ console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`);
371
+ process.exit(5);
372
+ }
210
373
  if (e && e.code === 'BUDGET_HARDSTOP') {
211
374
  console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
212
375
  console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
@@ -244,7 +407,14 @@ async function cmdRun(positional, flags) {
244
407
  const aw = receipt.results.aggregates.with_skill;
245
408
  const ab = receipt.results.aggregates.baseline;
246
409
  console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
247
- console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${cmp.delta_uncertainty.toFixed(3)} (with ${aw.mean_score.toFixed(3)} ± ${aw.stddev.toFixed(3)} vs base ${ab.mean_score.toFixed(3)} ± ${ab.stddev.toFixed(3)})\n`);
410
+ console.log(` → ${answeredLine(receipt)}; verification_level ${receipt.verification_level}`);
411
+ // Spec 026 AC-8: how many draws were cut at the output cap and excluded.
412
+ const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
413
+ if (nTruncated) console.log(` → ${nTruncated} draw(s) truncated at the output cap: unmeasured, never judged, excluded from every band`);
414
+ // Spec 026 AC-6, AC-7: the band rule is named beside the band, and a band
415
+ // the formula could not form prints as n/a, never as 0.000.
416
+ if (cmp.delta == null) console.log(` → skill lift n/a (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})\n`);
417
+ else console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${uncertaintyStr(cmp)} (with ${band(aw.mean_score, aw.stddev)} vs base ${band(ab.mean_score, ab.stddev)}; band = sample stddev of the per-case means)\n`);
248
418
  }
249
419
 
250
420
  console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
@@ -256,10 +426,8 @@ function cmdDiff(positional, flags) {
256
426
  const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
257
427
  const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
258
428
 
259
- // Warn (do not block) if either receipt fails self-verification.
260
- for (const [p, r] of [[aPath, a], [bPath, b]]) {
261
- if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
262
- }
429
+ // Spec 026 AC-20: a side whose hash does not verify is refused, naming it.
430
+ for (const [p, r] of [[aPath, a], [bPath, b]]) refuseUnverified(r, p, 'diff');
263
431
 
264
432
  // --mode revision inverts the axis: the skill text is the variable under test
265
433
  // and the substrate is the control. The fields release drift merely warns about
@@ -338,6 +506,7 @@ function cmdBadge(positional, flags) {
338
506
  const p = positional[0];
339
507
  if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
340
508
  const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
509
+ refuseUnverified(receipt, p, 'badge');
341
510
  if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
342
511
  const badge = badgeEndpoint(receipt);
343
512
  const json = JSON.stringify(badge, null, 2);
@@ -388,7 +557,7 @@ function cmdExport(positional, flags) {
388
557
  if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
389
558
  const { toSummaryJson } = require('../lib/export');
390
559
  const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
391
- if (!verifyReceiptHash(receipt)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
560
+ refuseUnverified(receipt, p, 'export');
392
561
  const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
393
562
  const json = JSON.stringify(summary, null, 2);
394
563
  if (flags.out) {
package/config.js CHANGED
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
9
9
  // Bumped whenever the runner's behaviour or receipt-generation semantics change
10
10
  // in a way that could affect results. Recorded into every receipt as
11
11
  // run.runner_version so a receipt is reproducible against a known engine.
12
- const RUNNER_VERSION = '0.8.0';
12
+ const RUNNER_VERSION = '0.9.0';
13
13
 
14
14
  // The eval format we CONSUME (we deliberately do not invent our own).
15
15
  const SUITE_FORMAT = 'agentskills.io/evals';
@@ -29,7 +29,30 @@ const SUITE_FORMAT = 'agentskills.io/evals';
29
29
  // at run time so derived dollars stay reproducible), and the derived
30
30
  // `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
31
31
  // v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
32
- const RECEIPT_SCHEMA_VERSION = '0.5';
32
+ // v0.5 samples the GENERATION n times per arm (per-case `generation` draw list,
33
+ // the across-draw band, the variance ratio, the sampling policy applied)
34
+ // and adds the per-suite canary. v0.4 is frozen as receipt.v0.4.schema.json.
35
+ // v0.6 (spec 026, receipt integrity) says WHAT ANSWERED: run.answered_by, a
36
+ // `stub` surface, run.judge.model_id + prompt_template_hash, per-draw
37
+ // stop_reason/truncated and n_truncated, case_status failed_unmeasured,
38
+ // results.aggregates.band_rule, and bands that are null where the formula
39
+ // cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
40
+ // is refused. v0.5 is frozen as receipt.v0.5.schema.json.
41
+ const RECEIPT_SCHEMA_VERSION = '0.6';
42
+
43
+ // Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
44
+ // A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
45
+ // case, prompt or rubric bound, so a 2 MB prompt went to the model at the
46
+ // fixed 250-token projection; lib/checks.js compiled a suite-supplied regex
47
+ // with no bound, and JavaScript cannot time a regex out. Each is set well
48
+ // above every suite this repository holds: they bind pathological input, not
49
+ // the archive. Named in the error a bound raises, and in AUTHORING.md.
50
+ const SKILL_MAX_FILES = 2000; // bundled files under a skill directory
51
+ const SKILL_MAX_BYTES = 32 * 1024 * 1024; // bundled bytes, all files together
52
+ const SKILL_MAX_DEPTH = 16; // directory depth below the skill dir
53
+ const SUITE_MAX_CASES = 500; // cases in one evals.json
54
+ const CASE_MAX_CHARS = 65536; // characters in one prompt or rubric
55
+ const CHECK_MAX_PATTERN = 256; // characters in one checks[] regex
33
56
 
34
57
  // Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
35
58
  // these. The projection is refused before any call if it exceeds the cap, on
@@ -133,6 +156,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
133
156
 
134
157
  module.exports = {
135
158
  PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
159
+ SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
136
160
  EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
137
161
  GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
138
162
  REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
package/lib/checks.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { CHECK_MAX_PATTERN } = require('../config');
5
+
4
6
  // Deterministic post-checks.
5
7
  //
6
8
  // A per-case eval suite may declare optional `checks[]`: structural / regex
@@ -19,6 +21,22 @@
19
21
  // contains — the output includes the literal `value` substring
20
22
  // not_contains — the output does NOT include the literal `value` substring
21
23
  // min_length — the trimmed output is at least `value` characters long
24
+ //
25
+ // BOUNDED (spec 026 AC-13, audit A7). A suite's regex is third-party input
26
+ // compiled and run against model output, and JavaScript cannot time a regex
27
+ // out. A pattern longer than CHECK_MAX_PATTERN, or one carrying a nested
28
+ // quantifier of the (x+)+ family (a quantified group whose body ends in a
29
+ // quantifier: the shape that is exponential on every backtracking engine),
30
+ // is REFUSED: recorded pass: false with a reason that says so, never
31
+ // evaluated. The test is syntactic and narrow by design; it is not a general
32
+ // ReDoS detector, and the spec says so.
33
+ const NESTED_QUANTIFIER = /\((?:[^()\\]|\\.)*[+*}]\)\s*[+*]|\((?:[^()\\]|\\.)*\|(?:[^()\\]|\\.)*\)\s*[+*]/;
34
+ function refusedPattern(pattern) {
35
+ const p = String(pattern == null ? '' : pattern);
36
+ if (p.length > CHECK_MAX_PATTERN) return `refused: pattern of ${p.length} characters exceeds CHECK_MAX_PATTERN (${CHECK_MAX_PATTERN})`;
37
+ if (NESTED_QUANTIFIER.test(p)) return 'refused: pattern carries a nested quantifier of the (x+)+ family, which is exponential to evaluate';
38
+ return null;
39
+ }
22
40
 
23
41
  function runOneCheck(check, output) {
24
42
  const text = String(output || '');
@@ -40,11 +58,15 @@ function runOneCheck(check, output) {
40
58
  // `checks`). Empty array when the case declares no checks.
41
59
  function runChecks(output, checks) {
42
60
  if (!Array.isArray(checks) || !checks.length) return [];
43
- return checks.map((c) => ({
44
- name: String((c && c.name) || (c && c.kind) || 'check'),
45
- kind: c && c.kind,
46
- pass: !!runOneCheck(c, output),
47
- }));
61
+ return checks.map((c) => {
62
+ const reason = c && c.kind === 'regex' ? refusedPattern(c.pattern) : null;
63
+ if (reason) return { name: String((c && c.name) || (c && c.kind) || 'check'), kind: c && c.kind, pass: false, reason };
64
+ return {
65
+ name: String((c && c.name) || (c && c.kind) || 'check'),
66
+ kind: c && c.kind,
67
+ pass: !!runOneCheck(c, output),
68
+ };
69
+ });
48
70
  }
49
71
 
50
- module.exports = { runChecks, runOneCheck };
72
+ module.exports = { runChecks, runOneCheck, refusedPattern, NESTED_QUANTIFIER };