driftproof 0.8.1 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -5
- package/bin/driftproof +171 -22
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/README.md
CHANGED
|
@@ -262,7 +262,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
262
262
|
|
|
263
263
|
```jsonc
|
|
264
264
|
{
|
|
265
|
-
"schema_version": "0.
|
|
265
|
+
"schema_version": "0.6",
|
|
266
266
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
267
267
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
268
268
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -271,12 +271,16 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
271
271
|
"model_release_date": "2025-10-01",
|
|
272
272
|
"provider": "anthropic",
|
|
273
273
|
"surface": "claude-cli",
|
|
274
|
-
"runner_version": "0.
|
|
274
|
+
"runner_version": "0.9.0",
|
|
275
275
|
"date_utc": "2026-07-27T…Z",
|
|
276
276
|
"registry": "registered",
|
|
277
277
|
"transcripts": "hashes-only",
|
|
278
278
|
"judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
|
|
279
|
-
"surface": "claude-cli"
|
|
279
|
+
"surface": "claude-cli", "model_id": "claude-haiku-4-5-20251001",
|
|
280
|
+
"prompt_template_hash": "…sha256 over the grading template…" },
|
|
281
|
+
"answered_by": { "kind": "model", "attested": true,
|
|
282
|
+
"reported_model": "claude-haiku-4-5-20251001",
|
|
283
|
+
"reported_models": ["claude-haiku-4-5"], "isolation": "eval-user" }
|
|
280
284
|
},
|
|
281
285
|
"results": {
|
|
282
286
|
"cases": [
|
|
@@ -345,7 +349,7 @@ jobs:
|
|
|
345
349
|
runs-on: ubuntu-latest
|
|
346
350
|
steps:
|
|
347
351
|
- uses: actions/checkout@v4
|
|
348
|
-
- uses: driftproofhq/driftproof@v0.
|
|
352
|
+
- uses: driftproofhq/driftproof@v0.9.0
|
|
349
353
|
with:
|
|
350
354
|
skill-dir: skills/my-skill
|
|
351
355
|
models: claude-haiku-4-5
|
|
@@ -361,7 +365,9 @@ fails the job on `REGRESSED` unless you set `fail-on-regression: 'false'`. For a
|
|
|
361
365
|
free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job env —
|
|
362
366
|
the runner returns canned receipts so the wiring can be tested without spend (this
|
|
363
367
|
is exactly how the action's own [self-test](.github/workflows/action-selftest.yml)
|
|
364
|
-
runs).
|
|
368
|
+
runs). A stub receipt says what it is: `verification_level` `UNVERIFIED`,
|
|
369
|
+
`run.surface` `stub`, `run.answered_by.kind` `stub`, and the verdict on it is
|
|
370
|
+
`NOT_MEASURED` — it proves the wiring, never the skill. That self-test is also the proof behind the input hardening: on every
|
|
365
371
|
run it passes one hostile value (a quote, a semicolon, `$(...)`, a backtick and
|
|
366
372
|
a newline) through each of the action's five inputs on the real runner and
|
|
367
373
|
fails unless every one is refused before anything ran, so a green check there
|
package/bin/driftproof
CHANGED
|
@@ -6,13 +6,13 @@ const fs = require('fs');
|
|
|
6
6
|
const path = require('path');
|
|
7
7
|
const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MAX_CALLS } = require('../config');
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
|
-
const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
|
|
9
|
+
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
11
|
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
|
-
const { registryStatus } = require('../lib/models');
|
|
15
|
+
const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
17
17
|
const { scaffoldInit } = require('../lib/init');
|
|
18
18
|
|
|
@@ -40,6 +40,24 @@ function writeTranscripts(receipt, transcripts) {
|
|
|
40
40
|
|
|
41
41
|
const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
|
|
42
42
|
|
|
43
|
+
// ── THE DISPLAY VERIFIES BEFORE IT READS (spec 026 AC-20, F7) ────────────────
|
|
44
|
+
// badge, diff and export read a receipt somebody else may have edited. Until
|
|
45
|
+
// this, badge never checked the hash and diff and export printed a warning
|
|
46
|
+
// and proceeded, so a receipt with one digit moved and the hash left as it
|
|
47
|
+
// was rendered `passing` (028's D-5, for the CLI). A receipt whose
|
|
48
|
+
// receipt_hash does not verify is refused here: nothing on stdout, no file,
|
|
49
|
+
// the reason on stderr naming receipt_hash and the file, exit 4. The bound the
|
|
50
|
+
// threat model states does not move: the same edit followed by a re-seal
|
|
51
|
+
// (sealReceipt) verifies and renders, which is what receipt_hash is: integrity
|
|
52
|
+
// since sealing, not authenticity. The check lives here, in the three
|
|
53
|
+
// commands, and not in lib/verdict.js's writer, which the repo gate calls on
|
|
54
|
+
// synthetic unsealed receipts.
|
|
55
|
+
function refuseUnverified(receipt, file, command) {
|
|
56
|
+
if (verifyReceiptHash(receipt)) return;
|
|
57
|
+
console.error(` ✗ REFUSED (${command}): receipt_hash does not verify for ${path.basename(file)} (tampered or hand-edited since it was sealed); nothing rendered.`);
|
|
58
|
+
process.exit(4);
|
|
59
|
+
}
|
|
60
|
+
|
|
43
61
|
// Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
|
|
44
62
|
// in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
|
|
45
63
|
// flags always win over the rc; the rc wins over built-in defaults.
|
|
@@ -58,6 +76,81 @@ function loadRc(skillDir) {
|
|
|
58
76
|
// positional instead of swallowing it as the flag's value.
|
|
59
77
|
const BOOLEAN_FLAGS = new Set(['trusted-skill']);
|
|
60
78
|
|
|
79
|
+
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
80
|
+
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
81
|
+
// them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
|
|
82
|
+
// same copy and AC-9 asserts the three identical, rule for rule, both ways.
|
|
83
|
+
// Consolidating into one shared module is 028's decision 1, the spec after 026.
|
|
84
|
+
//
|
|
85
|
+
// Why a contract at the door: `parseInt('abc', 10)` is NaN, both cost guards
|
|
86
|
+
// are `>` comparisons against it, both are false, and the run completed and
|
|
87
|
+
// wrote a receipt that carried no trace of it (the dollar cap never compared;
|
|
88
|
+
// the call cap silently became the dev default). `--samples abc` was quieter
|
|
89
|
+
// still: `NaN || 5` is 5. Every numeric input, from the command line or the
|
|
90
|
+
// rc, is checked against its rule BEFORE the projection is printed and before
|
|
91
|
+
// any call; a flag given with no value (the parser records `true`) and an
|
|
92
|
+
// empty string are refused the same way, not defaulted.
|
|
93
|
+
const INPUT_CONTRACT = {
|
|
94
|
+
rules: {
|
|
95
|
+
models: { re: /^[A-Za-z0-9._-]+(,[A-Za-z0-9._-]+)*$/, message: 'models: expected a comma-separated list of model ids (letters, digits, . _ -)' },
|
|
96
|
+
max_usd: { re: /^[0-9]+(\.[0-9]+)?$/, message: 'max-usd: expected a positive decimal number such as 20 or 3.53' },
|
|
97
|
+
max_usd_zero: { re: /^0+(\.0+)?$/, reject_on_match: true, message: 'max-usd: must be greater than zero' },
|
|
98
|
+
max_calls: { re: /^[1-9][0-9]*$/, message: 'max-calls: expected a positive integer with no leading zero' },
|
|
99
|
+
skill_dir_control: { re: /[\x00-\x1f\x7f]/, reject_on_match: true, message: 'skill-dir: contains a control character' },
|
|
100
|
+
},
|
|
101
|
+
// Which rule each `driftproof run` input is checked against. samples,
|
|
102
|
+
// max-cases and concurrency are positive integers and reuse max_calls' rule.
|
|
103
|
+
flags: { 'models': ['models'], 'max-usd': ['max_usd', 'max_usd_zero'], 'max-calls': ['max_calls'], 'samples': ['max_calls'], 'max-cases': ['max_calls'], 'concurrency': ['max_calls'], 'skill-dir': ['skill_dir_control'] },
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
function shownValue(v) {
|
|
107
|
+
if (v === true) return '<no value>';
|
|
108
|
+
return JSON.stringify(String(v));
|
|
109
|
+
}
|
|
110
|
+
// Refuse one input: the rule's own message, the input's name and the value it
|
|
111
|
+
// carried, exit 2. Nothing has been printed or spent when this runs.
|
|
112
|
+
function refuseInput(name, value, message, where) {
|
|
113
|
+
console.error(` ✗ REFUSED: ${message}, got ${shownValue(value)} (${where} ${name}); nothing run, no receipt written.`);
|
|
114
|
+
process.exit(2);
|
|
115
|
+
}
|
|
116
|
+
// Check one input against every rule its flag maps to. `value` is what was
|
|
117
|
+
// given: a string, a number (from the rc), `true` (a flag with no value) or
|
|
118
|
+
// an empty string; only a string that matches every rule passes.
|
|
119
|
+
function checkInput(flag, value, where, shownName = flag) {
|
|
120
|
+
const rules = INPUT_CONTRACT.flags[flag] || [];
|
|
121
|
+
const text = typeof value === 'string' ? value : (typeof value === 'number' && Number.isFinite(value)) ? String(value) : null;
|
|
122
|
+
if (text === null) refuseInput(shownName, value, `${flag}: expected a value`, where);
|
|
123
|
+
for (const key of rules) {
|
|
124
|
+
const rule = INPUT_CONTRACT.rules[key];
|
|
125
|
+
const hit = rule.re.test(text);
|
|
126
|
+
if (rule.reject_on_match ? hit : !hit) refuseInput(shownName, value, rule.message, where);
|
|
127
|
+
}
|
|
128
|
+
return text;
|
|
129
|
+
}
|
|
130
|
+
// Every numeric input the run takes, from the command line first and the rc
|
|
131
|
+
// second, checked before anything is printed. Returns the checked strings.
|
|
132
|
+
// The registry door (spec 026 AC-10, F6): every target and the judge it will
|
|
133
|
+
// run with must be a model the registry knows. lib/models.js assertRegistered
|
|
134
|
+
// is the one rule; lib/run.js asks it too on the path every caller shares, and
|
|
135
|
+
// this door is where bin/driftproof says the registry's path in its own
|
|
136
|
+
// message, before the suite loads.
|
|
137
|
+
function checkRegistered(models, judgeFor) {
|
|
138
|
+
try {
|
|
139
|
+
for (const m of models) { assertRegistered(m, 'model'); assertRegistered(judgeFor(m), 'judge model'); }
|
|
140
|
+
} catch (e) {
|
|
141
|
+
if (e && e.code === 'UNREGISTERED_MODEL') { console.error(` ✗ REFUSED (${/judge/.test(e.message) ? 'judge-model / judge_model' : 'models'}): ${e.message}`); process.exit(2); }
|
|
142
|
+
throw e;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
function checkNumericInputs(flags, rc) {
|
|
146
|
+
const out = {};
|
|
147
|
+
for (const [flag, rcKey] of [['max-calls', 'max_calls'], ['max-usd', 'max_usd'], ['samples', 'samples'], ['max-cases', 'max_cases'], ['concurrency', 'concurrency']]) {
|
|
148
|
+
if (Object.prototype.hasOwnProperty.call(flags, flag)) out[flag] = checkInput(flag, flags[flag], 'flag --');
|
|
149
|
+
else if (rc[rcKey] !== undefined && rc[rcKey] !== null) out[flag] = checkInput(flag, rc[rcKey], '.driftproofrc', rcKey);
|
|
150
|
+
}
|
|
151
|
+
return out;
|
|
152
|
+
}
|
|
153
|
+
|
|
61
154
|
function parseArgs(argv) {
|
|
62
155
|
const positional = [];
|
|
63
156
|
const flags = {};
|
|
@@ -112,9 +205,10 @@ NOTES
|
|
|
112
205
|
- --keep-transcripts writes the raw generations + judge outputs to
|
|
113
206
|
transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
|
|
114
207
|
local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
|
|
115
|
-
- Prices and registry status come from config/models.json
|
|
116
|
-
|
|
117
|
-
|
|
208
|
+
- Prices and registry status come from config/models.json (or the registry
|
|
209
|
+
DRIFTPROOF_REGISTRY names). An unregistered model id is REFUSED before any call,
|
|
210
|
+
naming the id and the registry path (so is one not shaped like a model id);
|
|
211
|
+
registry:"unregistered" is an import-only receipt value, refused for a run.
|
|
118
212
|
- --concurrency runs that many (case,mode) tasks at once (default 1). Higher
|
|
119
213
|
values cut wall-clock on the cli surface (cold-start dominated).
|
|
120
214
|
- ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
|
|
@@ -144,23 +238,53 @@ async function cmdRun(positional, flags) {
|
|
|
144
238
|
|
|
145
239
|
// Per-project defaults from .driftproofrc (if any); CLI flags override these.
|
|
146
240
|
const rc = loadRc(skillDir);
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
241
|
+
// The contract, at the door (spec 026 AC-9): every input checked against
|
|
242
|
+
// its rule before the projection and before any call. A value that fails
|
|
243
|
+
// its shape is refused naming the input and the value; nothing is defaulted.
|
|
244
|
+
if (INPUT_CONTRACT.rules.skill_dir_control.re.test(String(skillDir))) refuseInput('skill-dir', skillDir, INPUT_CONTRACT.rules.skill_dir_control.message, 'argument');
|
|
245
|
+
const checked = checkNumericInputs(flags, rc);
|
|
246
|
+
const modelsRaw = Object.prototype.hasOwnProperty.call(flags, 'models') ? flags.models : (rc.models !== undefined && rc.models !== null) ? rc.models : 'haiku';
|
|
247
|
+
const models = String(modelsRaw).split(',').map((s) => s.trim()).filter(Boolean);
|
|
151
248
|
const judgeModel = flags['judge-model'] || rc.judge_model || null;
|
|
152
|
-
|
|
153
|
-
|
|
249
|
+
// Spec 026 AC-10 (F6): every model id, and the judge, must be a model the
|
|
250
|
+
// registry knows; refused here naming the input, the id and the registry's
|
|
251
|
+
// absolute path, before the suite loads, before the projection, before any
|
|
252
|
+
// call. A shape failure is refused the same way (the contract's models rule
|
|
253
|
+
// below then never sees one). The same check stands in lib/run.js for every
|
|
254
|
+
// other caller.
|
|
255
|
+
const judgeFor = (m) => (judgeModel ? resolveModel(judgeModel) : resolveModel(m));
|
|
256
|
+
checkRegistered(models, judgeFor);
|
|
257
|
+
const modelsGiven = Object.prototype.hasOwnProperty.call(flags, 'models') ? checkInput('models', flags.models, 'flag --')
|
|
258
|
+
: (rc.models !== undefined && rc.models !== null) ? checkInput('models', rc.models, '.driftproofrc', 'models') : 'haiku';
|
|
259
|
+
void modelsGiven;
|
|
260
|
+
const maxCases = checked['max-cases'] !== undefined ? parseInt(checked['max-cases'], 10) : null;
|
|
261
|
+
const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
|
|
262
|
+
const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : DEFAULT_JUDGE_SAMPLES;
|
|
263
|
+
const concurrency = checked.concurrency !== undefined ? parseInt(checked.concurrency, 10) : 1;
|
|
264
|
+
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
154
265
|
const keepTranscripts = !!flags['keep-transcripts'];
|
|
155
|
-
// spec 022: the same-user spawn exists only behind this flag.
|
|
156
|
-
//
|
|
266
|
+
// spec 022: the same-user spawn exists only behind this flag.
|
|
267
|
+
//
|
|
268
|
+
// THE EVAL USER IS NOT RESOLVED HERE (F-022-4, spec 021). It used to be, on
|
|
269
|
+
// every run that was not --trusted-skill, including an api-surface-only run
|
|
270
|
+
// that spawns no CLI at all - so a hostile DRIFTPROOF_EVAL_USER refused a run
|
|
271
|
+
// that would never have read it. Fail-loud and harmless, and still wrong: a
|
|
272
|
+
// variable that governs a lane this run does not take should be neither read
|
|
273
|
+
// nor validated. It is resolved below, once a subscription surface is actually
|
|
274
|
+
// in the run, which is the same condition the isolation banner already used.
|
|
157
275
|
const trusted = !!flags['trusted-skill'];
|
|
158
|
-
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
159
276
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
160
277
|
fs.mkdirSync(outDir, { recursive: true });
|
|
161
278
|
|
|
162
279
|
const skill = loadSkill(skillDir);
|
|
163
280
|
const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
|
|
281
|
+
// Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
|
|
282
|
+
// over nothing is a receipt about nothing. Refused here, before the
|
|
283
|
+
// projection and before any call, naming the suite file.
|
|
284
|
+
if (nCases < 1) {
|
|
285
|
+
console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')} has no cases${maxCases ? ` (after --max-cases ${maxCases})` : ''}; nothing to measure, no receipt written.`);
|
|
286
|
+
process.exit(2);
|
|
287
|
+
}
|
|
164
288
|
// v0.5 draws the generation up to SAMPLING.max times per arm, so BOTH the
|
|
165
289
|
// printed projection and the dollar guard below must be scaled by it. Left at
|
|
166
290
|
// draws=1 the guard would admit a run costing up to ten times its own
|
|
@@ -174,7 +298,13 @@ async function cmdRun(positional, flags) {
|
|
|
174
298
|
// `api` surface this is real spend, so we refuse if it would exceed --max-usd.
|
|
175
299
|
// On `claude-cli` the metered spend is $0 (subscription); the figure is the
|
|
176
300
|
// hypothetical "if run on the metered API" cost — printed, never blocks.
|
|
177
|
-
|
|
301
|
+
// Spec 026 AC-10: the projection is priced with the judge the run will USE
|
|
302
|
+
// (lib/run.js: --judge-model when given, else the target itself), never a
|
|
303
|
+
// fixed haiku; each target is priced with its own judge and the figures summed.
|
|
304
|
+
const targets = models.map((m) => resolveModel(m));
|
|
305
|
+
const judges = models.map((m) => judgeFor(m));
|
|
306
|
+
const perModelCost = targets.map((t, i) => estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: [targets[i]], judgeModel: judges[i] }));
|
|
307
|
+
const cost = { totalUSD: Math.round(perModelCost.reduce((a, c) => a + c.totalUSD, 0) * 1e4) / 1e4, perModel: perModelCost.flatMap((c) => c.perModel), judges };
|
|
178
308
|
// Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
|
|
179
309
|
const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
|
|
180
310
|
const surface = surfaces.join(', ');
|
|
@@ -182,10 +312,17 @@ async function cmdRun(positional, flags) {
|
|
|
182
312
|
|
|
183
313
|
const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
|
|
184
314
|
console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
|
|
315
|
+
// Spec 026 AC-1: a stub run says so before it starts, and its receipt says so
|
|
316
|
+
// after (surface stub, answered_by.kind stub, UNVERIFIED).
|
|
317
|
+
if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer; the receipt will be UNVERIFIED and measure nothing');
|
|
185
318
|
console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
|
|
186
319
|
console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
|
|
187
320
|
console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
|
|
188
|
-
|
|
321
|
+
console.log(` judge: ${[...new Set(cost.judges)].join(', ')} (${judgeModel ? '--judge-model' : 'the target model itself; set --judge-model to change'})`);
|
|
322
|
+
if (subSurfaces.length) {
|
|
323
|
+
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
324
|
+
console.log(` isolation: ${isolation}`);
|
|
325
|
+
}
|
|
189
326
|
console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
|
|
190
327
|
console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
191
328
|
if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
|
|
@@ -227,6 +364,12 @@ async function cmdRun(positional, flags) {
|
|
|
227
364
|
},
|
|
228
365
|
});
|
|
229
366
|
} catch (e) {
|
|
367
|
+
if (e && e.code === 'SUBSTRATE_MISMATCH') {
|
|
368
|
+
// Spec 026 AC-2: the surface answered as a different model. The run
|
|
369
|
+
// stopped before its next call and wrote no receipt.
|
|
370
|
+
console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`);
|
|
371
|
+
process.exit(5);
|
|
372
|
+
}
|
|
230
373
|
if (e && e.code === 'BUDGET_HARDSTOP') {
|
|
231
374
|
console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
|
|
232
375
|
console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
|
|
@@ -264,7 +407,14 @@ async function cmdRun(positional, flags) {
|
|
|
264
407
|
const aw = receipt.results.aggregates.with_skill;
|
|
265
408
|
const ab = receipt.results.aggregates.baseline;
|
|
266
409
|
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
|
|
267
|
-
console.log(` →
|
|
410
|
+
console.log(` → ${answeredLine(receipt)}; verification_level ${receipt.verification_level}`);
|
|
411
|
+
// Spec 026 AC-8: how many draws were cut at the output cap and excluded.
|
|
412
|
+
const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
|
|
413
|
+
if (nTruncated) console.log(` → ${nTruncated} draw(s) truncated at the output cap: unmeasured, never judged, excluded from every band`);
|
|
414
|
+
// Spec 026 AC-6, AC-7: the band rule is named beside the band, and a band
|
|
415
|
+
// the formula could not form prints as n/a, never as 0.000.
|
|
416
|
+
if (cmp.delta == null) console.log(` → skill lift n/a (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})\n`);
|
|
417
|
+
else console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${uncertaintyStr(cmp)} (with ${band(aw.mean_score, aw.stddev)} vs base ${band(ab.mean_score, ab.stddev)}; band = sample stddev of the per-case means)\n`);
|
|
268
418
|
}
|
|
269
419
|
|
|
270
420
|
console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
|
|
@@ -276,10 +426,8 @@ function cmdDiff(positional, flags) {
|
|
|
276
426
|
const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
|
|
277
427
|
const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
|
|
278
428
|
|
|
279
|
-
//
|
|
280
|
-
for (const [p, r] of [[aPath, a], [bPath, b]])
|
|
281
|
-
if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
|
|
282
|
-
}
|
|
429
|
+
// Spec 026 AC-20: a side whose hash does not verify is refused, naming it.
|
|
430
|
+
for (const [p, r] of [[aPath, a], [bPath, b]]) refuseUnverified(r, p, 'diff');
|
|
283
431
|
|
|
284
432
|
// --mode revision inverts the axis: the skill text is the variable under test
|
|
285
433
|
// and the substrate is the control. The fields release drift merely warns about
|
|
@@ -358,6 +506,7 @@ function cmdBadge(positional, flags) {
|
|
|
358
506
|
const p = positional[0];
|
|
359
507
|
if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
|
|
360
508
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
509
|
+
refuseUnverified(receipt, p, 'badge');
|
|
361
510
|
if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
|
|
362
511
|
const badge = badgeEndpoint(receipt);
|
|
363
512
|
const json = JSON.stringify(badge, null, 2);
|
|
@@ -408,7 +557,7 @@ function cmdExport(positional, flags) {
|
|
|
408
557
|
if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
|
|
409
558
|
const { toSummaryJson } = require('../lib/export');
|
|
410
559
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
411
|
-
|
|
560
|
+
refuseUnverified(receipt, p, 'export');
|
|
412
561
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
413
562
|
const json = JSON.stringify(summary, null, 2);
|
|
414
563
|
if (flags.out) {
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.9.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -29,7 +29,30 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
29
29
|
// at run time so derived dollars stay reproducible), and the derived
|
|
30
30
|
// `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
|
|
31
31
|
// v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
|
|
32
|
-
|
|
32
|
+
// v0.5 samples the GENERATION n times per arm (per-case `generation` draw list,
|
|
33
|
+
// the across-draw band, the variance ratio, the sampling policy applied)
|
|
34
|
+
// and adds the per-suite canary. v0.4 is frozen as receipt.v0.4.schema.json.
|
|
35
|
+
// v0.6 (spec 026, receipt integrity) says WHAT ANSWERED: run.answered_by, a
|
|
36
|
+
// `stub` surface, run.judge.model_id + prompt_template_hash, per-draw
|
|
37
|
+
// stop_reason/truncated and n_truncated, case_status failed_unmeasured,
|
|
38
|
+
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
|
+
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
|
+
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.6';
|
|
42
|
+
|
|
43
|
+
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
|
+
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|
|
45
|
+
// case, prompt or rubric bound, so a 2 MB prompt went to the model at the
|
|
46
|
+
// fixed 250-token projection; lib/checks.js compiled a suite-supplied regex
|
|
47
|
+
// with no bound, and JavaScript cannot time a regex out. Each is set well
|
|
48
|
+
// above every suite this repository holds: they bind pathological input, not
|
|
49
|
+
// the archive. Named in the error a bound raises, and in AUTHORING.md.
|
|
50
|
+
const SKILL_MAX_FILES = 2000; // bundled files under a skill directory
|
|
51
|
+
const SKILL_MAX_BYTES = 32 * 1024 * 1024; // bundled bytes, all files together
|
|
52
|
+
const SKILL_MAX_DEPTH = 16; // directory depth below the skill dir
|
|
53
|
+
const SUITE_MAX_CASES = 500; // cases in one evals.json
|
|
54
|
+
const CASE_MAX_CHARS = 65536; // characters in one prompt or rubric
|
|
55
|
+
const CHECK_MAX_PATTERN = 256; // characters in one checks[] regex
|
|
33
56
|
|
|
34
57
|
// Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
|
|
35
58
|
// these. The projection is refused before any call if it exceeds the cap, on
|
|
@@ -133,6 +156,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
|
133
156
|
|
|
134
157
|
module.exports = {
|
|
135
158
|
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
159
|
+
SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
|
|
136
160
|
EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
137
161
|
GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
|
|
138
162
|
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|
package/lib/checks.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const { CHECK_MAX_PATTERN } = require('../config');
|
|
5
|
+
|
|
4
6
|
// Deterministic post-checks.
|
|
5
7
|
//
|
|
6
8
|
// A per-case eval suite may declare optional `checks[]`: structural / regex
|
|
@@ -19,6 +21,22 @@
|
|
|
19
21
|
// contains — the output includes the literal `value` substring
|
|
20
22
|
// not_contains — the output does NOT include the literal `value` substring
|
|
21
23
|
// min_length — the trimmed output is at least `value` characters long
|
|
24
|
+
//
|
|
25
|
+
// BOUNDED (spec 026 AC-13, audit A7). A suite's regex is third-party input
|
|
26
|
+
// compiled and run against model output, and JavaScript cannot time a regex
|
|
27
|
+
// out. A pattern longer than CHECK_MAX_PATTERN, or one carrying a nested
|
|
28
|
+
// quantifier of the (x+)+ family (a quantified group whose body ends in a
|
|
29
|
+
// quantifier: the shape that is exponential on every backtracking engine),
|
|
30
|
+
// is REFUSED: recorded pass: false with a reason that says so, never
|
|
31
|
+
// evaluated. The test is syntactic and narrow by design; it is not a general
|
|
32
|
+
// ReDoS detector, and the spec says so.
|
|
33
|
+
const NESTED_QUANTIFIER = /\((?:[^()\\]|\\.)*[+*}]\)\s*[+*]|\((?:[^()\\]|\\.)*\|(?:[^()\\]|\\.)*\)\s*[+*]/;
|
|
34
|
+
function refusedPattern(pattern) {
|
|
35
|
+
const p = String(pattern == null ? '' : pattern);
|
|
36
|
+
if (p.length > CHECK_MAX_PATTERN) return `refused: pattern of ${p.length} characters exceeds CHECK_MAX_PATTERN (${CHECK_MAX_PATTERN})`;
|
|
37
|
+
if (NESTED_QUANTIFIER.test(p)) return 'refused: pattern carries a nested quantifier of the (x+)+ family, which is exponential to evaluate';
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
22
40
|
|
|
23
41
|
function runOneCheck(check, output) {
|
|
24
42
|
const text = String(output || '');
|
|
@@ -40,11 +58,15 @@ function runOneCheck(check, output) {
|
|
|
40
58
|
// `checks`). Empty array when the case declares no checks.
|
|
41
59
|
function runChecks(output, checks) {
|
|
42
60
|
if (!Array.isArray(checks) || !checks.length) return [];
|
|
43
|
-
return checks.map((c) =>
|
|
44
|
-
|
|
45
|
-
kind: c && c.kind,
|
|
46
|
-
|
|
47
|
-
|
|
61
|
+
return checks.map((c) => {
|
|
62
|
+
const reason = c && c.kind === 'regex' ? refusedPattern(c.pattern) : null;
|
|
63
|
+
if (reason) return { name: String((c && c.name) || (c && c.kind) || 'check'), kind: c && c.kind, pass: false, reason };
|
|
64
|
+
return {
|
|
65
|
+
name: String((c && c.name) || (c && c.kind) || 'check'),
|
|
66
|
+
kind: c && c.kind,
|
|
67
|
+
pass: !!runOneCheck(c, output),
|
|
68
|
+
};
|
|
69
|
+
});
|
|
48
70
|
}
|
|
49
71
|
|
|
50
|
-
module.exports = { runChecks, runOneCheck };
|
|
72
|
+
module.exports = { runChecks, runOneCheck, refusedPattern, NESTED_QUANTIFIER };
|
package/lib/diff.js
CHANGED
|
@@ -59,7 +59,10 @@ function withSkillBands(receipt) {
|
|
|
59
59
|
// which is exactly the v0.1 weakness v0.2 fixes).
|
|
60
60
|
function aggWithBand(receipt) {
|
|
61
61
|
const a = receipt.results.aggregates.with_skill;
|
|
62
|
-
|
|
62
|
+
// v0.6 (spec 026 AC-7): a null band is carried as null and rendered as what
|
|
63
|
+
// it is; a legacy receipt with no aggregate band keeps its 0 (the archive's
|
|
64
|
+
// rendering is unmoved, AC-16).
|
|
65
|
+
return { mean: a.mean_score, stddev: a.stddev === null ? null : (a.stddev || 0), cases: a.case_count };
|
|
63
66
|
}
|
|
64
67
|
|
|
65
68
|
function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
|
|
@@ -81,11 +84,41 @@ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
|
|
|
81
84
|
// not from one case's draws).
|
|
82
85
|
|
|
83
86
|
function bandStr(x) {
|
|
87
|
+
if (x && x.mean == null) return 'n/a (0 cases)';
|
|
88
|
+
if (x && x.stddev == null) return `${x.mean.toFixed(3)} ± n/a (${x.cases === 1 ? '1 case' : 'no band'})`;
|
|
84
89
|
const label = x && x.source;
|
|
85
90
|
return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
|
|
86
91
|
}
|
|
87
92
|
function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
|
|
88
93
|
|
|
94
|
+
// ── THE JUDGE IS PART OF THE INSTRUMENT (spec 026 AC-12, A4) ─────────────────
|
|
95
|
+
// Two receipts graded by different judges are two measurements with different
|
|
96
|
+
// instruments, and a verdict across them says nothing about the skill.
|
|
97
|
+
// lib/reuse.js's triage already says a changed judge means regrade; the
|
|
98
|
+
// differ did not ask. It asks here: run.judge.model_id (on canonical ids, so
|
|
99
|
+
// an alias and its dated form are one judge), run.judge.prompt_template_hash,
|
|
100
|
+
// run.judge.temperature, and every shared case's judge.rubric_hash. A field
|
|
101
|
+
// that differs names itself in the caveats and suppresses every per-case
|
|
102
|
+
// verdict. A pre-v0.6 receipt carries no template hash: the report says the
|
|
103
|
+
// template is unrecorded on that side and still compares the rest.
|
|
104
|
+
const canonicalJudgeId = (id) => String(id == null ? '' : id).replace(/-\d{8}$/, '');
|
|
105
|
+
function judgeDiffers(a, b, ids, aB, bB) {
|
|
106
|
+
const ja = (a.run && a.run.judge) || {}; const jb = (b.run && b.run.judge) || {};
|
|
107
|
+
const problems = [];
|
|
108
|
+
const idA = ja.model_id || (a.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
109
|
+
const idB = jb.model_id || (b.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
110
|
+
if (idA && idB && canonicalJudgeId(idA) !== canonicalJudgeId(idB)) problems.push(`run.judge.model_id differs (${idA} vs ${idB})`);
|
|
111
|
+
const unrecorded = [];
|
|
112
|
+
if (!ja.prompt_template_hash) unrecorded.push('A'); if (!jb.prompt_template_hash) unrecorded.push('B');
|
|
113
|
+
if (ja.prompt_template_hash && jb.prompt_template_hash && ja.prompt_template_hash !== jb.prompt_template_hash) problems.push(`run.judge.prompt_template_hash differs (${short(ja.prompt_template_hash)} vs ${short(jb.prompt_template_hash)}): the grading template is a different judge`);
|
|
114
|
+
if (ja.temperature !== undefined && jb.temperature !== undefined && ja.temperature !== jb.temperature) problems.push(`run.judge.temperature differs (${ja.temperature} vs ${jb.temperature})`);
|
|
115
|
+
const rubricA = Object.fromEntries(a.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
116
|
+
const rubricB = Object.fromEntries(b.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
117
|
+
const moved = ids.filter((id) => rubricA[id] && rubricB[id] && rubricA[id] !== rubricB[id]);
|
|
118
|
+
if (moved.length) problems.push(`judge.rubric_hash differs on ${moved.length} shared case(s) (${moved.slice(0, 3).map((x) => `\`${x}\``).join(', ')}${moved.length > 3 ? ', …' : ''})`);
|
|
119
|
+
return { problems, unrecorded };
|
|
120
|
+
}
|
|
121
|
+
|
|
89
122
|
// The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
|
|
90
123
|
// core lives per case (judge-sample bands), and the headline just aggregates it.
|
|
91
124
|
// It deliberately does NOT run a separate band test on the aggregate mean: the
|
|
@@ -146,9 +179,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
146
179
|
// improvement verdicts are never computed from evidence we did not run.
|
|
147
180
|
const levelOf = (r) => r.verification_level || 'TESTED';
|
|
148
181
|
const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
|
|
182
|
+
// Spec 026 AC-4: an INCOMPLETE receipt on either side computes no per-case
|
|
183
|
+
// verdict. RECEIPT.md has said so since v0.3.1; the differ did not ask.
|
|
184
|
+
const incompleteSides = [[labelA, a], [labelB, b]].filter(([, r]) => r.run && r.run.status === 'incomplete');
|
|
185
|
+
// Spec 026 AC-12: a different judge on either side suppresses the verdict.
|
|
186
|
+
const judge = judgeDiffers(a, b, ids, aB, bB);
|
|
187
|
+
const judgeProblem = judge.problems.length ? judge.problems.join('; ') : null;
|
|
149
188
|
// A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
|
|
150
189
|
// untested input does: no delta is asserted, and the reason travels with it.
|
|
151
|
-
const measured = belowTested.length === 0 && !refusal;
|
|
190
|
+
const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
|
|
152
191
|
|
|
153
192
|
const perCase = ids.map((id) => {
|
|
154
193
|
const before = aB[id] || null;
|
|
@@ -250,6 +289,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
250
289
|
if (belowTested.length) {
|
|
251
290
|
warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
|
|
252
291
|
}
|
|
292
|
+
if (incompleteSides.length) {
|
|
293
|
+
warnings.push(`verdicts NOT computed — ${incompleteSides.map(([l, r]) => `${l} is incomplete (run.status incomplete; it excluded ${r.run.failed_case_count || 0} case(s) whose arm could not be measured)`).join('; ')}. A receipt with an unmeasured arm is not evidence for a per-case verdict; its aggregates are shown as context only.`);
|
|
294
|
+
}
|
|
295
|
+
if (judgeProblem) {
|
|
296
|
+
warnings.push(`verdicts NOT computed — the two receipts were graded by different judges: ${judgeProblem}. A verdict across two instruments says nothing about the skill; regrade one side with the other's judge (lib/reuse.js triage: regrade).`);
|
|
297
|
+
}
|
|
298
|
+
if (judge.unrecorded.length) {
|
|
299
|
+
warnings.push(`judge template unrecorded on ${judge.unrecorded.map((x) => (x === 'A' ? labelA : labelB)).join(' and ')} (a pre-v0.6 receipt carries no run.judge.prompt_template_hash); the judge model id and the rubric hashes are compared, the template is not.`);
|
|
300
|
+
}
|
|
253
301
|
// Cross-provider / cross-surface disclosure (Phase 6). A comparison across
|
|
254
302
|
// providers is a skill-DURABILITY comparison across substrates, not model drift
|
|
255
303
|
// over time; across surfaces, sampling control differs. Both are flagged so a
|
|
@@ -281,7 +329,11 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
281
329
|
// THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
|
|
282
330
|
// was printed unconditionally, so a refused pair got "0 receipt(s) below
|
|
283
331
|
// TESTED" — a statement measurably false of the receipts it was given.
|
|
284
|
-
if (
|
|
332
|
+
if (incompleteSides.length) {
|
|
333
|
+
// An incomplete side is a fact about that receipt alone (spec 026 AC-4)
|
|
334
|
+
// and leads whatever else is wrong with the pair.
|
|
335
|
+
L.push(`**NOT MEASURED — ${incompleteSides.map(([l]) => l).join(' and ')} incomplete: a case's arm could not be measured, so no per-case verdict is computed.**`);
|
|
336
|
+
} else if (refusal) {
|
|
285
337
|
// The reason is a complete sentence and already ends by saying no verdict
|
|
286
338
|
// is asserted; prefixing that again produced "REFUSED — no verdict is
|
|
287
339
|
// asserted. the baseline arm did not reproduce…" — a duplicated clause and
|
|
@@ -289,6 +341,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
289
341
|
L.push(`**REFUSED — ${refusal.reason}**`);
|
|
290
342
|
} else if (belowTested.length) {
|
|
291
343
|
L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
|
|
344
|
+
} else if (judgeProblem) {
|
|
345
|
+
L.push(`**NOT MEASURED — different judges: ${judgeProblem}. No verdict crosses two instruments.**`);
|
|
292
346
|
} else {
|
|
293
347
|
L.push('**NOT MEASURED — no verdict is asserted.**');
|
|
294
348
|
}
|
|
@@ -327,4 +381,4 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
327
381
|
};
|
|
328
382
|
}
|
|
329
383
|
|
|
330
|
-
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
|
|
384
|
+
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem, judgeDiffers };
|
package/lib/importers.js
CHANGED
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
// compatibility contract) are documented in docs/interop.md.
|
|
18
18
|
|
|
19
19
|
const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
|
|
20
|
-
const { sealReceipt } = require('./receipt');
|
|
20
|
+
const { sealReceipt, BAND_RULE } = require('./receipt');
|
|
21
21
|
const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
|
|
22
22
|
const { outcomeFor } = require('./run');
|
|
23
23
|
const { inferProvider } = require('./provider');
|
|
@@ -64,11 +64,16 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
|
|
|
64
64
|
date_utc: dateUtc,
|
|
65
65
|
registry: registryStatus(modelId),
|
|
66
66
|
transcripts: 'none',
|
|
67
|
-
|
|
67
|
+
// v0.6: the source tool's grader is the judge that ran; its template is
|
|
68
|
+
// not ours to hash (null, the one place the schema allows it).
|
|
69
|
+
judge: { ...judgeBlock, model_id: (cases.find((c) => c.judge && c.judge.model_id) || { judge: { model_id: 'unknown' } }).judge.model_id, prompt_template_hash: null },
|
|
70
|
+
// v0.6 (spec 026 AC-1): what answered. An external tool's harness did;
|
|
71
|
+
// nothing here attested a model, and nothing was spawned by us.
|
|
72
|
+
answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
|
|
68
73
|
},
|
|
69
74
|
results: {
|
|
70
75
|
cases,
|
|
71
|
-
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
|
|
76
|
+
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
|
|
72
77
|
},
|
|
73
78
|
comparison,
|
|
74
79
|
verification_level: 'DECLARED',
|