driftproof 0.8.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -262,7 +262,7 @@ A receipt is the unit of evidence — one JSON document conforming to
262
262
 
263
263
  ```jsonc
264
264
  {
265
- "schema_version": "0.5",
265
+ "schema_version": "0.6",
266
266
  "skill": { "name": "commit-message-conventions", "version": "0.2.0",
267
267
  "content_hash": "…sha256 over SKILL.md + bundled files…" },
268
268
  "suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
@@ -271,12 +271,16 @@ A receipt is the unit of evidence — one JSON document conforming to
271
271
  "model_release_date": "2025-10-01",
272
272
  "provider": "anthropic",
273
273
  "surface": "claude-cli",
274
- "runner_version": "0.8.1",
274
+ "runner_version": "0.9.0",
275
275
  "date_utc": "2026-07-27T…Z",
276
276
  "registry": "registered",
277
277
  "transcripts": "hashes-only",
278
278
  "judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
279
- "surface": "claude-cli" }
279
+ "surface": "claude-cli", "model_id": "claude-haiku-4-5-20251001",
280
+ "prompt_template_hash": "…sha256 over the grading template…" },
281
+ "answered_by": { "kind": "model", "attested": true,
282
+ "reported_model": "claude-haiku-4-5-20251001",
283
+ "reported_models": ["claude-haiku-4-5"], "isolation": "eval-user" }
280
284
  },
281
285
  "results": {
282
286
  "cases": [
@@ -345,7 +349,7 @@ jobs:
345
349
  runs-on: ubuntu-latest
346
350
  steps:
347
351
  - uses: actions/checkout@v4
348
- - uses: driftproofhq/driftproof@v0.8.1
352
+ - uses: driftproofhq/driftproof@v0.9.0
349
353
  with:
350
354
  skill-dir: skills/my-skill
351
355
  models: claude-haiku-4-5
@@ -361,7 +365,9 @@ fails the job on `REGRESSED` unless you set `fail-on-regression: 'false'`. For a
361
365
  free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job env —
362
366
  the runner returns canned receipts so the wiring can be tested without spend (this
363
367
  is exactly how the action's own [self-test](.github/workflows/action-selftest.yml)
364
- runs). That self-test is also the proof behind the input hardening: on every
368
+ runs). A stub receipt says what it is: `verification_level` `UNVERIFIED`,
369
+ `run.surface` `stub`, `run.answered_by.kind` `stub`, and the verdict on it is
370
+ `NOT_MEASURED` — it proves the wiring, never the skill. That self-test is also the proof behind the input hardening: on every
365
371
  run it passes one hostile value (a quote, a semicolon, `$(...)`, a backtick and
366
372
  a newline) through each of the action's five inputs on the real runner and
367
373
  fails unless every one is refused before anything ran, so a green check there
package/bin/driftproof CHANGED
@@ -6,13 +6,13 @@ const fs = require('fs');
6
6
  const path = require('path');
7
7
  const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MAX_CALLS } = require('../config');
8
8
  const { loadSkill } = require('../lib/skill');
9
- const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
9
+ const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
10
10
  const { SAMPLING } = require('../lib/sampling');
11
11
  const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
12
12
  const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
13
13
  const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
14
14
  const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
15
- const { registryStatus } = require('../lib/models');
15
+ const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
16
16
  const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
17
17
  const { scaffoldInit } = require('../lib/init');
18
18
 
@@ -40,6 +40,24 @@ function writeTranscripts(receipt, transcripts) {
40
40
 
41
41
  const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
42
42
 
43
+ // ── THE DISPLAY VERIFIES BEFORE IT READS (spec 026 AC-20, F7) ────────────────
44
+ // badge, diff and export read a receipt somebody else may have edited. Until
45
+ // this, badge never checked the hash and diff and export printed a warning
46
+ // and proceeded, so a receipt with one digit moved and the hash left as it
47
+ // was rendered `passing` (028's D-5, for the CLI). A receipt whose
48
+ // receipt_hash does not verify is refused here: nothing on stdout, no file,
49
+ // the reason on stderr naming receipt_hash and the file, exit 4. The bound the
50
+ // threat model states does not move: the same edit followed by a re-seal
51
+ // (sealReceipt) verifies and renders, which is what receipt_hash is: integrity
52
+ // since sealing, not authenticity. The check lives here, in the three
53
+ // commands, and not in lib/verdict.js's writer, which the repo gate calls on
54
+ // synthetic unsealed receipts.
55
+ function refuseUnverified(receipt, file, command) {
56
+ if (verifyReceiptHash(receipt)) return;
57
+ console.error(` ✗ REFUSED (${command}): receipt_hash does not verify for ${path.basename(file)} (tampered or hand-edited since it was sealed); nothing rendered.`);
58
+ process.exit(4);
59
+ }
60
+
43
61
  // Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
44
62
  // in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
45
63
  // flags always win over the rc; the rc wins over built-in defaults.
@@ -58,6 +76,81 @@ function loadRc(skillDir) {
58
76
  // positional instead of swallowing it as the flag's value.
59
77
  const BOOLEAN_FLAGS = new Set(['trusted-skill']);
60
78
 
79
+ // ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
80
+ // A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
81
+ // them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
82
+ // same copy and AC-9 asserts the three identical, rule for rule, both ways.
83
+ // Consolidating into one shared module is 028's decision 1, the spec after 026.
84
+ //
85
+ // Why a contract at the door: `parseInt('abc', 10)` is NaN, both cost guards
86
+ // are `>` comparisons against it, both are false, and the run completed and
87
+ // wrote a receipt that carried no trace of it (the dollar cap never compared;
88
+ // the call cap silently became the dev default). `--samples abc` was quieter
89
+ // still: `NaN || 5` is 5. Every numeric input, from the command line or the
90
+ // rc, is checked against its rule BEFORE the projection is printed and before
91
+ // any call; a flag given with no value (the parser records `true`) and an
92
+ // empty string are refused the same way, not defaulted.
93
+ const INPUT_CONTRACT = {
94
+ rules: {
95
+ models: { re: /^[A-Za-z0-9._-]+(,[A-Za-z0-9._-]+)*$/, message: 'models: expected a comma-separated list of model ids (letters, digits, . _ -)' },
96
+ max_usd: { re: /^[0-9]+(\.[0-9]+)?$/, message: 'max-usd: expected a positive decimal number such as 20 or 3.53' },
97
+ max_usd_zero: { re: /^0+(\.0+)?$/, reject_on_match: true, message: 'max-usd: must be greater than zero' },
98
+ max_calls: { re: /^[1-9][0-9]*$/, message: 'max-calls: expected a positive integer with no leading zero' },
99
+ skill_dir_control: { re: /[\x00-\x1f\x7f]/, reject_on_match: true, message: 'skill-dir: contains a control character' },
100
+ },
101
+ // Which rule each `driftproof run` input is checked against. samples,
102
+ // max-cases and concurrency are positive integers and reuse max_calls' rule.
103
+ flags: { 'models': ['models'], 'max-usd': ['max_usd', 'max_usd_zero'], 'max-calls': ['max_calls'], 'samples': ['max_calls'], 'max-cases': ['max_calls'], 'concurrency': ['max_calls'], 'skill-dir': ['skill_dir_control'] },
104
+ };
105
+
106
+ function shownValue(v) {
107
+ if (v === true) return '<no value>';
108
+ return JSON.stringify(String(v));
109
+ }
110
+ // Refuse one input: the rule's own message, the input's name and the value it
111
+ // carried, exit 2. Nothing has been printed or spent when this runs.
112
+ function refuseInput(name, value, message, where) {
113
+ console.error(` ✗ REFUSED: ${message}, got ${shownValue(value)} (${where} ${name}); nothing run, no receipt written.`);
114
+ process.exit(2);
115
+ }
116
+ // Check one input against every rule its flag maps to. `value` is what was
117
+ // given: a string, a number (from the rc), `true` (a flag with no value) or
118
+ // an empty string; only a string that matches every rule passes.
119
+ function checkInput(flag, value, where, shownName = flag) {
120
+ const rules = INPUT_CONTRACT.flags[flag] || [];
121
+ const text = typeof value === 'string' ? value : (typeof value === 'number' && Number.isFinite(value)) ? String(value) : null;
122
+ if (text === null) refuseInput(shownName, value, `${flag}: expected a value`, where);
123
+ for (const key of rules) {
124
+ const rule = INPUT_CONTRACT.rules[key];
125
+ const hit = rule.re.test(text);
126
+ if (rule.reject_on_match ? hit : !hit) refuseInput(shownName, value, rule.message, where);
127
+ }
128
+ return text;
129
+ }
130
+ // Every numeric input the run takes, from the command line first and the rc
131
+ // second, checked before anything is printed. Returns the checked strings.
132
+ // The registry door (spec 026 AC-10, F6): every target and the judge it will
133
+ // run with must be a model the registry knows. lib/models.js assertRegistered
134
+ // is the one rule; lib/run.js asks it too on the path every caller shares, and
135
+ // this door is where bin/driftproof says the registry's path in its own
136
+ // message, before the suite loads.
137
+ function checkRegistered(models, judgeFor) {
138
+ try {
139
+ for (const m of models) { assertRegistered(m, 'model'); assertRegistered(judgeFor(m), 'judge model'); }
140
+ } catch (e) {
141
+ if (e && e.code === 'UNREGISTERED_MODEL') { console.error(` ✗ REFUSED (${/judge/.test(e.message) ? 'judge-model / judge_model' : 'models'}): ${e.message}`); process.exit(2); }
142
+ throw e;
143
+ }
144
+ }
145
+ function checkNumericInputs(flags, rc) {
146
+ const out = {};
147
+ for (const [flag, rcKey] of [['max-calls', 'max_calls'], ['max-usd', 'max_usd'], ['samples', 'samples'], ['max-cases', 'max_cases'], ['concurrency', 'concurrency']]) {
148
+ if (Object.prototype.hasOwnProperty.call(flags, flag)) out[flag] = checkInput(flag, flags[flag], 'flag --');
149
+ else if (rc[rcKey] !== undefined && rc[rcKey] !== null) out[flag] = checkInput(flag, rc[rcKey], '.driftproofrc', rcKey);
150
+ }
151
+ return out;
152
+ }
153
+
61
154
  function parseArgs(argv) {
62
155
  const positional = [];
63
156
  const flags = {};
@@ -112,9 +205,10 @@ NOTES
112
205
  - --keep-transcripts writes the raw generations + judge outputs to
113
206
  transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
114
207
  local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
115
- - Prices and registry status come from config/models.json; an unregistered
116
- model still runs but is marked registry:"unregistered" and costed at the
117
- conservative default price.
208
+ - Prices and registry status come from config/models.json (or the registry
209
+ DRIFTPROOF_REGISTRY names). An unregistered model id is REFUSED before any call,
210
+ naming the id and the registry path (so is one not shaped like a model id);
211
+ registry:"unregistered" is an import-only receipt value, refused for a run.
118
212
  - --concurrency runs that many (case,mode) tasks at once (default 1). Higher
119
213
  values cut wall-clock on the cli surface (cold-start dominated).
120
214
  - ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
@@ -144,23 +238,53 @@ async function cmdRun(positional, flags) {
144
238
 
145
239
  // Per-project defaults from .driftproofrc (if any); CLI flags override these.
146
240
  const rc = loadRc(skillDir);
147
- const models = String(flags.models || rc.models || 'haiku').split(',').map((s) => s.trim()).filter(Boolean);
148
- const maxCases = flags['max-cases'] ? parseInt(flags['max-cases'], 10) : (rc.max_cases != null ? parseInt(rc.max_cases, 10) : null);
149
- const maxCalls = flags['max-calls'] ? parseInt(flags['max-calls'], 10) : (rc.max_calls != null ? parseInt(rc.max_calls, 10) : DEV_MAX_CALLS);
150
- const samples = flags.samples ? parseInt(flags.samples, 10) : (rc.samples != null ? parseInt(rc.samples, 10) : DEFAULT_JUDGE_SAMPLES);
241
+ // The contract, at the door (spec 026 AC-9): every input checked against
242
+ // its rule before the projection and before any call. A value that fails
243
+ // its shape is refused naming the input and the value; nothing is defaulted.
244
+ if (INPUT_CONTRACT.rules.skill_dir_control.re.test(String(skillDir))) refuseInput('skill-dir', skillDir, INPUT_CONTRACT.rules.skill_dir_control.message, 'argument');
245
+ const checked = checkNumericInputs(flags, rc);
246
+ const modelsRaw = Object.prototype.hasOwnProperty.call(flags, 'models') ? flags.models : (rc.models !== undefined && rc.models !== null) ? rc.models : 'haiku';
247
+ const models = String(modelsRaw).split(',').map((s) => s.trim()).filter(Boolean);
151
248
  const judgeModel = flags['judge-model'] || rc.judge_model || null;
152
- const concurrency = flags.concurrency ? parseInt(flags.concurrency, 10) : (rc.concurrency != null ? parseInt(rc.concurrency, 10) : 1);
153
- const maxUsd = flags['max-usd'] ? parseFloat(flags['max-usd']) : (rc.max_usd != null ? parseFloat(rc.max_usd) : DEV_MAX_USD);
249
+ // Spec 026 AC-10 (F6): every model id, and the judge, must be a model the
250
+ // registry knows; refused here naming the input, the id and the registry's
251
+ // absolute path, before the suite loads, before the projection, before any
252
+ // call. A shape failure is refused the same way (the contract's models rule
253
+ // below then never sees one). The same check stands in lib/run.js for every
254
+ // other caller.
255
+ const judgeFor = (m) => (judgeModel ? resolveModel(judgeModel) : resolveModel(m));
256
+ checkRegistered(models, judgeFor);
257
+ const modelsGiven = Object.prototype.hasOwnProperty.call(flags, 'models') ? checkInput('models', flags.models, 'flag --')
258
+ : (rc.models !== undefined && rc.models !== null) ? checkInput('models', rc.models, '.driftproofrc', 'models') : 'haiku';
259
+ void modelsGiven;
260
+ const maxCases = checked['max-cases'] !== undefined ? parseInt(checked['max-cases'], 10) : null;
261
+ const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
262
+ const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : DEFAULT_JUDGE_SAMPLES;
263
+ const concurrency = checked.concurrency !== undefined ? parseInt(checked.concurrency, 10) : 1;
264
+ const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
154
265
  const keepTranscripts = !!flags['keep-transcripts'];
155
- // spec 022: the same-user spawn exists only behind this flag. Resolving the
156
- // eval user here means a bad DRIFTPROOF_EVAL_USER is refused before the banner.
266
+ // spec 022: the same-user spawn exists only behind this flag.
267
+ //
268
+ // THE EVAL USER IS NOT RESOLVED HERE (F-022-4, spec 021). It used to be, on
269
+ // every run that was not --trusted-skill, including an api-surface-only run
270
+ // that spawns no CLI at all - so a hostile DRIFTPROOF_EVAL_USER refused a run
271
+ // that would never have read it. Fail-loud and harmless, and still wrong: a
272
+ // variable that governs a lane this run does not take should be neither read
273
+ // nor validated. It is resolved below, once a subscription surface is actually
274
+ // in the run, which is the same condition the isolation banner already used.
157
275
  const trusted = !!flags['trusted-skill'];
158
- const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
159
276
  const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
160
277
  fs.mkdirSync(outDir, { recursive: true });
161
278
 
162
279
  const skill = loadSkill(skillDir);
163
280
  const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
281
+ // Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
282
+ // over nothing is a receipt about nothing. Refused here, before the
283
+ // projection and before any call, naming the suite file.
284
+ if (nCases < 1) {
285
+ console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')} has no cases${maxCases ? ` (after --max-cases ${maxCases})` : ''}; nothing to measure, no receipt written.`);
286
+ process.exit(2);
287
+ }
164
288
  // v0.5 draws the generation up to SAMPLING.max times per arm, so BOTH the
165
289
  // printed projection and the dollar guard below must be scaled by it. Left at
166
290
  // draws=1 the guard would admit a run costing up to ten times its own
@@ -174,7 +298,13 @@ async function cmdRun(positional, flags) {
174
298
  // `api` surface this is real spend, so we refuse if it would exceed --max-usd.
175
299
  // On `claude-cli` the metered spend is $0 (subscription); the figure is the
176
300
  // hypothetical "if run on the metered API" cost — printed, never blocks.
177
- const cost = estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: models.map((m) => require('../lib/provider').resolveModel(m)), judgeModel: judgeModel || 'haiku' });
301
+ // Spec 026 AC-10: the projection is priced with the judge the run will USE
302
+ // (lib/run.js: --judge-model when given, else the target itself), never a
303
+ // fixed haiku; each target is priced with its own judge and the figures summed.
304
+ const targets = models.map((m) => resolveModel(m));
305
+ const judges = models.map((m) => judgeFor(m));
306
+ const perModelCost = targets.map((t, i) => estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: [targets[i]], judgeModel: judges[i] }));
307
+ const cost = { totalUSD: Math.round(perModelCost.reduce((a, c) => a + c.totalUSD, 0) * 1e4) / 1e4, perModel: perModelCost.flatMap((c) => c.perModel), judges };
178
308
  // Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
179
309
  const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
180
310
  const surface = surfaces.join(', ');
@@ -182,10 +312,17 @@ async function cmdRun(positional, flags) {
182
312
 
183
313
  const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
184
314
  console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
315
+ // Spec 026 AC-1: a stub run says so before it starts, and its receipt says so
316
+ // after (surface stub, answered_by.kind stub, UNVERIFIED).
317
+ if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer; the receipt will be UNVERIFIED and measure nothing');
185
318
  console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
186
319
  console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
187
320
  console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
188
- if (subSurfaces.length) console.log(` isolation: ${isolation}`);
321
+ console.log(` judge: ${[...new Set(cost.judges)].join(', ')} (${judgeModel ? '--judge-model' : 'the target model itself; set --judge-model to change'})`);
322
+ if (subSurfaces.length) {
323
+ const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
324
+ console.log(` isolation: ${isolation}`);
325
+ }
189
326
  console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
190
327
  console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
191
328
  if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
@@ -227,6 +364,12 @@ async function cmdRun(positional, flags) {
227
364
  },
228
365
  });
229
366
  } catch (e) {
367
+ if (e && e.code === 'SUBSTRATE_MISMATCH') {
368
+ // Spec 026 AC-2: the surface answered as a different model. The run
369
+ // stopped before its next call and wrote no receipt.
370
+ console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`);
371
+ process.exit(5);
372
+ }
230
373
  if (e && e.code === 'BUDGET_HARDSTOP') {
231
374
  console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
232
375
  console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
@@ -264,7 +407,14 @@ async function cmdRun(positional, flags) {
264
407
  const aw = receipt.results.aggregates.with_skill;
265
408
  const ab = receipt.results.aggregates.baseline;
266
409
  console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
267
- console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${cmp.delta_uncertainty.toFixed(3)} (with ${aw.mean_score.toFixed(3)} ± ${aw.stddev.toFixed(3)} vs base ${ab.mean_score.toFixed(3)} ± ${ab.stddev.toFixed(3)})\n`);
410
+ console.log(` → ${answeredLine(receipt)}; verification_level ${receipt.verification_level}`);
411
+ // Spec 026 AC-8: how many draws were cut at the output cap and excluded.
412
+ const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
413
+ if (nTruncated) console.log(` → ${nTruncated} draw(s) truncated at the output cap: unmeasured, never judged, excluded from every band`);
414
+ // Spec 026 AC-6, AC-7: the band rule is named beside the band, and a band
415
+ // the formula could not form prints as n/a, never as 0.000.
416
+ if (cmp.delta == null) console.log(` → skill lift n/a (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})\n`);
417
+ else console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${uncertaintyStr(cmp)} (with ${band(aw.mean_score, aw.stddev)} vs base ${band(ab.mean_score, ab.stddev)}; band = sample stddev of the per-case means)\n`);
268
418
  }
269
419
 
270
420
  console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
@@ -276,10 +426,8 @@ function cmdDiff(positional, flags) {
276
426
  const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
277
427
  const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
278
428
 
279
- // Warn (do not block) if either receipt fails self-verification.
280
- for (const [p, r] of [[aPath, a], [bPath, b]]) {
281
- if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
282
- }
429
+ // Spec 026 AC-20: a side whose hash does not verify is refused, naming it.
430
+ for (const [p, r] of [[aPath, a], [bPath, b]]) refuseUnverified(r, p, 'diff');
283
431
 
284
432
  // --mode revision inverts the axis: the skill text is the variable under test
285
433
  // and the substrate is the control. The fields release drift merely warns about
@@ -358,6 +506,7 @@ function cmdBadge(positional, flags) {
358
506
  const p = positional[0];
359
507
  if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
360
508
  const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
509
+ refuseUnverified(receipt, p, 'badge');
361
510
  if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
362
511
  const badge = badgeEndpoint(receipt);
363
512
  const json = JSON.stringify(badge, null, 2);
@@ -408,7 +557,7 @@ function cmdExport(positional, flags) {
408
557
  if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
409
558
  const { toSummaryJson } = require('../lib/export');
410
559
  const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
411
- if (!verifyReceiptHash(receipt)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
560
+ refuseUnverified(receipt, p, 'export');
412
561
  const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
413
562
  const json = JSON.stringify(summary, null, 2);
414
563
  if (flags.out) {
package/config.js CHANGED
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
9
9
  // Bumped whenever the runner's behaviour or receipt-generation semantics change
10
10
  // in a way that could affect results. Recorded into every receipt as
11
11
  // run.runner_version so a receipt is reproducible against a known engine.
12
- const RUNNER_VERSION = '0.8.1';
12
+ const RUNNER_VERSION = '0.9.0';
13
13
 
14
14
  // The eval format we CONSUME (we deliberately do not invent our own).
15
15
  const SUITE_FORMAT = 'agentskills.io/evals';
@@ -29,7 +29,30 @@ const SUITE_FORMAT = 'agentskills.io/evals';
29
29
  // at run time so derived dollars stay reproducible), and the derived
30
30
  // `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
31
31
  // v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
32
- const RECEIPT_SCHEMA_VERSION = '0.5';
32
+ // v0.5 samples the GENERATION n times per arm (per-case `generation` draw list,
33
+ // the across-draw band, the variance ratio, the sampling policy applied)
34
+ // and adds the per-suite canary. v0.4 is frozen as receipt.v0.4.schema.json.
35
+ // v0.6 (spec 026, receipt integrity) says WHAT ANSWERED: run.answered_by, a
36
+ // `stub` surface, run.judge.model_id + prompt_template_hash, per-draw
37
+ // stop_reason/truncated and n_truncated, case_status failed_unmeasured,
38
+ // results.aggregates.band_rule, and bands that are null where the formula
39
+ // cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
40
+ // is refused. v0.5 is frozen as receipt.v0.5.schema.json.
41
+ const RECEIPT_SCHEMA_VERSION = '0.6';
42
+
43
+ // Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
44
+ // A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
45
+ // case, prompt or rubric bound, so a 2 MB prompt went to the model at the
46
+ // fixed 250-token projection; lib/checks.js compiled a suite-supplied regex
47
+ // with no bound, and JavaScript cannot time a regex out. Each is set well
48
+ // above every suite this repository holds: they bind pathological input, not
49
+ // the archive. Named in the error a bound raises, and in AUTHORING.md.
50
+ const SKILL_MAX_FILES = 2000; // bundled files under a skill directory
51
+ const SKILL_MAX_BYTES = 32 * 1024 * 1024; // bundled bytes, all files together
52
+ const SKILL_MAX_DEPTH = 16; // directory depth below the skill dir
53
+ const SUITE_MAX_CASES = 500; // cases in one evals.json
54
+ const CASE_MAX_CHARS = 65536; // characters in one prompt or rubric
55
+ const CHECK_MAX_PATTERN = 256; // characters in one checks[] regex
33
56
 
34
57
  // Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
35
58
  // these. The projection is refused before any call if it exceeds the cap, on
@@ -133,6 +156,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
133
156
 
134
157
  module.exports = {
135
158
  PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
159
+ SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
136
160
  EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
137
161
  GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
138
162
  REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
package/lib/checks.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { CHECK_MAX_PATTERN } = require('../config');
5
+
4
6
  // Deterministic post-checks.
5
7
  //
6
8
  // A per-case eval suite may declare optional `checks[]`: structural / regex
@@ -19,6 +21,22 @@
19
21
  // contains — the output includes the literal `value` substring
20
22
  // not_contains — the output does NOT include the literal `value` substring
21
23
  // min_length — the trimmed output is at least `value` characters long
24
+ //
25
+ // BOUNDED (spec 026 AC-13, audit A7). A suite's regex is third-party input
26
+ // compiled and run against model output, and JavaScript cannot time a regex
27
+ // out. A pattern longer than CHECK_MAX_PATTERN, or one carrying a nested
28
+ // quantifier of the (x+)+ family (a quantified group whose body ends in a
29
+ // quantifier: the shape that is exponential on every backtracking engine),
30
+ // is REFUSED: recorded pass: false with a reason that says so, never
31
+ // evaluated. The test is syntactic and narrow by design; it is not a general
32
+ // ReDoS detector, and the spec says so.
33
+ const NESTED_QUANTIFIER = /\((?:[^()\\]|\\.)*[+*}]\)\s*[+*]|\((?:[^()\\]|\\.)*\|(?:[^()\\]|\\.)*\)\s*[+*]/;
34
+ function refusedPattern(pattern) {
35
+ const p = String(pattern == null ? '' : pattern);
36
+ if (p.length > CHECK_MAX_PATTERN) return `refused: pattern of ${p.length} characters exceeds CHECK_MAX_PATTERN (${CHECK_MAX_PATTERN})`;
37
+ if (NESTED_QUANTIFIER.test(p)) return 'refused: pattern carries a nested quantifier of the (x+)+ family, which is exponential to evaluate';
38
+ return null;
39
+ }
22
40
 
23
41
  function runOneCheck(check, output) {
24
42
  const text = String(output || '');
@@ -40,11 +58,15 @@ function runOneCheck(check, output) {
40
58
  // `checks`). Empty array when the case declares no checks.
41
59
  function runChecks(output, checks) {
42
60
  if (!Array.isArray(checks) || !checks.length) return [];
43
- return checks.map((c) => ({
44
- name: String((c && c.name) || (c && c.kind) || 'check'),
45
- kind: c && c.kind,
46
- pass: !!runOneCheck(c, output),
47
- }));
61
+ return checks.map((c) => {
62
+ const reason = c && c.kind === 'regex' ? refusedPattern(c.pattern) : null;
63
+ if (reason) return { name: String((c && c.name) || (c && c.kind) || 'check'), kind: c && c.kind, pass: false, reason };
64
+ return {
65
+ name: String((c && c.name) || (c && c.kind) || 'check'),
66
+ kind: c && c.kind,
67
+ pass: !!runOneCheck(c, output),
68
+ };
69
+ });
48
70
  }
49
71
 
50
- module.exports = { runChecks, runOneCheck };
72
+ module.exports = { runChecks, runOneCheck, refusedPattern, NESTED_QUANTIFIER };
package/lib/diff.js CHANGED
@@ -59,7 +59,10 @@ function withSkillBands(receipt) {
59
59
  // which is exactly the v0.1 weakness v0.2 fixes).
60
60
  function aggWithBand(receipt) {
61
61
  const a = receipt.results.aggregates.with_skill;
62
- return { mean: a.mean_score, stddev: a.stddev || 0 };
62
+ // v0.6 (spec 026 AC-7): a null band is carried as null and rendered as what
63
+ // it is; a legacy receipt with no aggregate band keeps its 0 (the archive's
64
+ // rendering is unmoved, AC-16).
65
+ return { mean: a.mean_score, stddev: a.stddev === null ? null : (a.stddev || 0), cases: a.case_count };
63
66
  }
64
67
 
65
68
  function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
@@ -81,11 +84,41 @@ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
81
84
  // not from one case's draws).
82
85
 
83
86
  function bandStr(x) {
87
+ if (x && x.mean == null) return 'n/a (0 cases)';
88
+ if (x && x.stddev == null) return `${x.mean.toFixed(3)} ± n/a (${x.cases === 1 ? '1 case' : 'no band'})`;
84
89
  const label = x && x.source;
85
90
  return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
86
91
  }
87
92
  function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
88
93
 
94
+ // ── THE JUDGE IS PART OF THE INSTRUMENT (spec 026 AC-12, A4) ─────────────────
95
+ // Two receipts graded by different judges are two measurements with different
96
+ // instruments, and a verdict across them says nothing about the skill.
97
+ // lib/reuse.js's triage already says a changed judge means regrade; the
98
+ // differ did not ask. It asks here: run.judge.model_id (on canonical ids, so
99
+ // an alias and its dated form are one judge), run.judge.prompt_template_hash,
100
+ // run.judge.temperature, and every shared case's judge.rubric_hash. A field
101
+ // that differs names itself in the caveats and suppresses every per-case
102
+ // verdict. A pre-v0.6 receipt carries no template hash: the report says the
103
+ // template is unrecorded on that side and still compares the rest.
104
+ const canonicalJudgeId = (id) => String(id == null ? '' : id).replace(/-\d{8}$/, '');
105
+ function judgeDiffers(a, b, ids, aB, bB) {
106
+ const ja = (a.run && a.run.judge) || {}; const jb = (b.run && b.run.judge) || {};
107
+ const problems = [];
108
+ const idA = ja.model_id || (a.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
109
+ const idB = jb.model_id || (b.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
110
+ if (idA && idB && canonicalJudgeId(idA) !== canonicalJudgeId(idB)) problems.push(`run.judge.model_id differs (${idA} vs ${idB})`);
111
+ const unrecorded = [];
112
+ if (!ja.prompt_template_hash) unrecorded.push('A'); if (!jb.prompt_template_hash) unrecorded.push('B');
113
+ if (ja.prompt_template_hash && jb.prompt_template_hash && ja.prompt_template_hash !== jb.prompt_template_hash) problems.push(`run.judge.prompt_template_hash differs (${short(ja.prompt_template_hash)} vs ${short(jb.prompt_template_hash)}): the grading template is a different judge`);
114
+ if (ja.temperature !== undefined && jb.temperature !== undefined && ja.temperature !== jb.temperature) problems.push(`run.judge.temperature differs (${ja.temperature} vs ${jb.temperature})`);
115
+ const rubricA = Object.fromEntries(a.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
116
+ const rubricB = Object.fromEntries(b.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
117
+ const moved = ids.filter((id) => rubricA[id] && rubricB[id] && rubricA[id] !== rubricB[id]);
118
+ if (moved.length) problems.push(`judge.rubric_hash differs on ${moved.length} shared case(s) (${moved.slice(0, 3).map((x) => `\`${x}\``).join(', ')}${moved.length > 3 ? ', …' : ''})`);
119
+ return { problems, unrecorded };
120
+ }
121
+
89
122
  // The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
90
123
  // core lives per case (judge-sample bands), and the headline just aggregates it.
91
124
  // It deliberately does NOT run a separate band test on the aggregate mean: the
@@ -146,9 +179,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
146
179
  // improvement verdicts are never computed from evidence we did not run.
147
180
  const levelOf = (r) => r.verification_level || 'TESTED';
148
181
  const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
182
+ // Spec 026 AC-4: an INCOMPLETE receipt on either side computes no per-case
183
+ // verdict. RECEIPT.md has said so since v0.3.1; the differ did not ask.
184
+ const incompleteSides = [[labelA, a], [labelB, b]].filter(([, r]) => r.run && r.run.status === 'incomplete');
185
+ // Spec 026 AC-12: a different judge on either side suppresses the verdict.
186
+ const judge = judgeDiffers(a, b, ids, aB, bB);
187
+ const judgeProblem = judge.problems.length ? judge.problems.join('; ') : null;
149
188
  // A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
150
189
  // untested input does: no delta is asserted, and the reason travels with it.
151
- const measured = belowTested.length === 0 && !refusal;
190
+ const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
152
191
 
153
192
  const perCase = ids.map((id) => {
154
193
  const before = aB[id] || null;
@@ -250,6 +289,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
250
289
  if (belowTested.length) {
251
290
  warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
252
291
  }
292
+ if (incompleteSides.length) {
293
+ warnings.push(`verdicts NOT computed — ${incompleteSides.map(([l, r]) => `${l} is incomplete (run.status incomplete; it excluded ${r.run.failed_case_count || 0} case(s) whose arm could not be measured)`).join('; ')}. A receipt with an unmeasured arm is not evidence for a per-case verdict; its aggregates are shown as context only.`);
294
+ }
295
+ if (judgeProblem) {
296
+ warnings.push(`verdicts NOT computed — the two receipts were graded by different judges: ${judgeProblem}. A verdict across two instruments says nothing about the skill; regrade one side with the other's judge (lib/reuse.js triage: regrade).`);
297
+ }
298
+ if (judge.unrecorded.length) {
299
+ warnings.push(`judge template unrecorded on ${judge.unrecorded.map((x) => (x === 'A' ? labelA : labelB)).join(' and ')} (a pre-v0.6 receipt carries no run.judge.prompt_template_hash); the judge model id and the rubric hashes are compared, the template is not.`);
300
+ }
253
301
  // Cross-provider / cross-surface disclosure (Phase 6). A comparison across
254
302
  // providers is a skill-DURABILITY comparison across substrates, not model drift
255
303
  // over time; across surfaces, sampling control differs. Both are flagged so a
@@ -281,7 +329,11 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
281
329
  // THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
282
330
  // was printed unconditionally, so a refused pair got "0 receipt(s) below
283
331
  // TESTED" — a statement measurably false of the receipts it was given.
284
- if (refusal) {
332
+ if (incompleteSides.length) {
333
+ // An incomplete side is a fact about that receipt alone (spec 026 AC-4)
334
+ // and leads whatever else is wrong with the pair.
335
+ L.push(`**NOT MEASURED — ${incompleteSides.map(([l]) => l).join(' and ')} incomplete: a case's arm could not be measured, so no per-case verdict is computed.**`);
336
+ } else if (refusal) {
285
337
  // The reason is a complete sentence and already ends by saying no verdict
286
338
  // is asserted; prefixing that again produced "REFUSED — no verdict is
287
339
  // asserted. the baseline arm did not reproduce…" — a duplicated clause and
@@ -289,6 +341,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
289
341
  L.push(`**REFUSED — ${refusal.reason}**`);
290
342
  } else if (belowTested.length) {
291
343
  L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
344
+ } else if (judgeProblem) {
345
+ L.push(`**NOT MEASURED — different judges: ${judgeProblem}. No verdict crosses two instruments.**`);
292
346
  } else {
293
347
  L.push('**NOT MEASURED — no verdict is asserted.**');
294
348
  }
@@ -327,4 +381,4 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
327
381
  };
328
382
  }
329
383
 
330
- module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
384
+ module.exports = { buildDriftReport, withSkillBands, revisionPairProblem, judgeDiffers };
package/lib/importers.js CHANGED
@@ -17,7 +17,7 @@
17
17
  // compatibility contract) are documented in docs/interop.md.
18
18
 
19
19
  const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
20
- const { sealReceipt } = require('./receipt');
20
+ const { sealReceipt, BAND_RULE } = require('./receipt');
21
21
  const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
22
22
  const { outcomeFor } = require('./run');
23
23
  const { inferProvider } = require('./provider');
@@ -64,11 +64,16 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
64
64
  date_utc: dateUtc,
65
65
  registry: registryStatus(modelId),
66
66
  transcripts: 'none',
67
- judge: judgeBlock,
67
+ // v0.6: the source tool's grader is the judge that ran; its template is
68
+ // not ours to hash (null, the one place the schema allows it).
69
+ judge: { ...judgeBlock, model_id: (cases.find((c) => c.judge && c.judge.model_id) || { judge: { model_id: 'unknown' } }).judge.model_id, prompt_template_hash: null },
70
+ // v0.6 (spec 026 AC-1): what answered. An external tool's harness did;
71
+ // nothing here attested a model, and nothing was spawned by us.
72
+ answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
68
73
  },
69
74
  results: {
70
75
  cases,
71
- aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
76
+ aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
72
77
  },
73
78
  comparison,
74
79
  verification_level: 'DECLARED',