driftproof 0.8.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +87 -16
- package/bin/driftproof +289 -24
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/decision.js +425 -0
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/runner.js +35 -3
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/README.md
CHANGED
|
@@ -158,6 +158,22 @@ npx driftproof <cmd> # no install — always the published version
|
|
|
158
158
|
npm install -g driftproof # or install the CLI globally
|
|
159
159
|
```
|
|
160
160
|
|
|
161
|
+
Inside Claude Code, there is a third path — the same CLI, reached from a slash
|
|
162
|
+
command:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
claude plugin marketplace add driftproofhq/driftproof
|
|
166
|
+
claude plugin install driftproof@driftproofhq
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
That installs `/driftproof:init`, `/driftproof:run` and `/driftproof:badge`.
|
|
170
|
+
The CLI is the product and it runs on every surface Driftproof measures, while
|
|
171
|
+
the plugin is one install path for Claude Code users: it builds one argument
|
|
172
|
+
vector, hands it to the pinned runner, and writes the receipt that runner would
|
|
173
|
+
have written from the same arguments. The plugin's version is the runner version
|
|
174
|
+
it pins, and `/driftproof:run` spends your Claude Code subscription rather than
|
|
175
|
+
an API key. It measures; it never edits a skill.
|
|
176
|
+
|
|
161
177
|
The only runtime dependency is `ajv` (schema validation); `@anthropic-ai/sdk` is
|
|
162
178
|
optional and pulled in only for `CLAUDE_PROVIDER=api`. The CLI resolves its spec,
|
|
163
179
|
schema, and model registry from inside the package, so it runs the same from a
|
|
@@ -262,7 +278,7 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
262
278
|
|
|
263
279
|
```jsonc
|
|
264
280
|
{
|
|
265
|
-
"schema_version": "0.
|
|
281
|
+
"schema_version": "0.6",
|
|
266
282
|
"skill": { "name": "commit-message-conventions", "version": "0.2.0",
|
|
267
283
|
"content_hash": "…sha256 over SKILL.md + bundled files…" },
|
|
268
284
|
"suite": { "format": "agentskills.io/evals", "suite_hash": "…", "case_count": 10 },
|
|
@@ -271,12 +287,16 @@ A receipt is the unit of evidence — one JSON document conforming to
|
|
|
271
287
|
"model_release_date": "2025-10-01",
|
|
272
288
|
"provider": "anthropic",
|
|
273
289
|
"surface": "claude-cli",
|
|
274
|
-
"runner_version": "0.
|
|
290
|
+
"runner_version": "0.10.0",
|
|
275
291
|
"date_utc": "2026-07-27T…Z",
|
|
276
292
|
"registry": "registered",
|
|
277
293
|
"transcripts": "hashes-only",
|
|
278
294
|
"judge": { "samples": 5, "temperature": null, "sampling": "surface-controlled",
|
|
279
|
-
"surface": "claude-cli"
|
|
295
|
+
"surface": "claude-cli", "model_id": "claude-haiku-4-5-20251001",
|
|
296
|
+
"prompt_template_hash": "…sha256 over the grading template…" },
|
|
297
|
+
"answered_by": { "kind": "model", "attested": true,
|
|
298
|
+
"reported_model": "claude-haiku-4-5-20251001",
|
|
299
|
+
"reported_models": ["claude-haiku-4-5"], "isolation": "eval-user" }
|
|
280
300
|
},
|
|
281
301
|
"results": {
|
|
282
302
|
"cases": [
|
|
@@ -345,27 +365,78 @@ jobs:
|
|
|
345
365
|
runs-on: ubuntu-latest
|
|
346
366
|
steps:
|
|
347
367
|
- uses: actions/checkout@v4
|
|
348
|
-
- uses: driftproofhq/driftproof@v0.
|
|
368
|
+
- uses: driftproofhq/driftproof@v0.10.0
|
|
349
369
|
with:
|
|
350
370
|
skill-dir: skills/my-skill
|
|
351
371
|
models: claude-haiku-4-5
|
|
352
372
|
api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
353
373
|
# max-usd: <n> # override the dollar budget (default: DEV_MAX_USD in config.js)
|
|
354
374
|
# max-calls: <n> # override the per-model call cap (default: DEV_MAX_CALLS)
|
|
355
|
-
# fail-on-regression: 'true' # (default) fail the job
|
|
375
|
+
# fail-on-regression: 'true' # (default) fail the job on a measured regression
|
|
376
|
+
# (a requested model with no readable receipt fails the job either way)
|
|
356
377
|
```
|
|
357
378
|
|
|
358
|
-
The action runs the suite
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
is
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
379
|
+
The action runs the suite on **every** id in `models`, writes one receipt per
|
|
380
|
+
model, uploads the whole receipt directory as a build artifact, and renders the
|
|
381
|
+
run on three surfaces: the **badge**, a **job summary** carrying one row per
|
|
382
|
+
requested model, and the **check title** of the enforcement step.
|
|
383
|
+
|
|
384
|
+
**The decision is taken over every receipt the run produced.** Each requested
|
|
385
|
+
model gets one decision state, and the run's `verdict` is the **worst** of them —
|
|
386
|
+
worst first, in this order:
|
|
387
|
+
|
|
388
|
+
| `verdict` | decision state | when |
|
|
389
|
+
|---|---|---|
|
|
390
|
+
| `REGRESSED` | regression | measured, and the skill hurt: `delta <= -EFFECT_FLOOR` |
|
|
391
|
+
| `REFUSED` | refused | a requested model produced **no readable receipt** |
|
|
392
|
+
| `INCONCLUSIVE` | inconclusive | the run did not complete, or carries no numeric delta |
|
|
393
|
+
| `NOT_MEASURED` | not measured | the receipt is below `TESTED`, or nothing in it says a model answered |
|
|
394
|
+
| `NO_EFFECT` | no detected effect | measured, and `\|delta\| < EFFECT_FLOOR` |
|
|
395
|
+
| `PASSED` | helped | measured, and the skill helped: `delta >= EFFECT_FLOOR` |
|
|
396
|
+
|
|
397
|
+
So on a multi-model run a regression on **any** one of them governs the verdict,
|
|
398
|
+
and the badge names the model it came from. Earlier releases decided the job from
|
|
399
|
+
whichever receipt's filename sorted **first**, which is a property of the model id
|
|
400
|
+
and not of the run — so if a Driftproof check was ever green on a multi-model run,
|
|
401
|
+
that result was only ever about one of your models, and is worth re-running before
|
|
402
|
+
you rely on it.
|
|
403
|
+
|
|
404
|
+
**What fails the job.** `fail-on-regression` governs **regression verdicts
|
|
405
|
+
only**: set it to `'false'` and a measured regression warns instead of failing.
|
|
406
|
+
`REFUSED` is not a verdict about your skill — it is the run failing to produce
|
|
407
|
+
one for a model you asked for — so it **fails the job whatever
|
|
408
|
+
`fail-on-regression` says**. A workflow that sets `fail-on-regression: 'false'`
|
|
409
|
+
to keep the check non-blocking can still be failed this way, deliberately: a
|
|
410
|
+
model that was not measured did not pass. `INCONCLUSIVE` and `NOT_MEASURED` do
|
|
411
|
+
not fail the job on their own — a run that could not measure is not evidence that
|
|
412
|
+
the skill hurt — but they **never render as success** on any of the three
|
|
413
|
+
surfaces: not in the badge, not in the summary row, and not in the check title,
|
|
414
|
+
which carries a `::warning` naming the state and the models it came from.
|
|
415
|
+
|
|
416
|
+
Step outputs: `verdict`, `delta` (of the model the worst decision came from),
|
|
417
|
+
`worst_state`, `regressed_models`, `missing_models` and `receipts_dir`. Each is
|
|
418
|
+
described in [`action.yml`](action.yml).
|
|
419
|
+
|
|
420
|
+
For a free CI dry-run with **zero model calls**, set `DRIFTPROOF_STUB=1` in the job
|
|
421
|
+
env — the runner returns canned receipts so the wiring can be tested without spend
|
|
422
|
+
(this is exactly how the action's own
|
|
423
|
+
[self-test](.github/workflows/action-selftest.yml) runs). A stub receipt says what
|
|
424
|
+
it is: `verification_level` `UNVERIFIED`, `run.surface` `stub`,
|
|
425
|
+
`run.answered_by.kind` `stub`, and the verdict on it is `NOT_MEASURED` — it proves
|
|
426
|
+
the wiring, never the skill, and the badge, the summary row and the check title
|
|
427
|
+
all say so rather than exiting quietly green.
|
|
428
|
+
|
|
429
|
+
That self-test is also the proof behind the **Action's** input hardening, and it
|
|
430
|
+
is worth saying exactly which surface that covers. On every CI run the Action
|
|
431
|
+
self-test sends one hostile value through each of the Action's five inputs — a
|
|
432
|
+
quote, a semicolon, `$(...)`, a backtick and a newline — on the real runner, and
|
|
433
|
+
the Action fails unless every one of them is refused before anything ran, so a
|
|
434
|
+
green check there is a claim you can read rather than one you have to take.
|
|
435
|
+
Those refusals live in `action/lib.sh`, and what they protect is the Action
|
|
436
|
+
surface. The CLI has carried its own input contract at its own door since
|
|
437
|
+
0.9.0 — the same rules, held identical to the Action's by the gate — so
|
|
438
|
+
`npx driftproof` refuses a malformed cap or model id before it projects a run.
|
|
439
|
+
Neither statement covers the other. Said of the Action, of CI, or of the runner, "hostile input is refused" is only ever true of the one surface it was measured on.
|
|
369
440
|
|
|
370
441
|
### Badge
|
|
371
442
|
|
package/bin/driftproof
CHANGED
|
@@ -6,14 +6,15 @@ const fs = require('fs');
|
|
|
6
6
|
const path = require('path');
|
|
7
7
|
const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD, DEV_MAX_CALLS } = require('../config');
|
|
8
8
|
const { loadSkill } = require('../lib/skill');
|
|
9
|
-
const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
|
|
9
|
+
const { runSkillOnModel, summarizeReceipt, projectCalls, answeredLine, band, uncertaintyStr } = require('../lib/run');
|
|
10
10
|
const { SAMPLING } = require('../lib/sampling');
|
|
11
11
|
const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
|
|
12
12
|
const { buildDriftReport, revisionPairProblem } = require('../lib/diff');
|
|
13
13
|
const { surfaceForModel, isSubscriptionSurface, resolveModel, evalUser } = require('../lib/provider');
|
|
14
14
|
const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
|
|
15
|
-
const { registryStatus } = require('../lib/models');
|
|
15
|
+
const { registryStatus, assertRegistered, REGISTRY_PATH } = require('../lib/models');
|
|
16
16
|
const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
|
|
17
|
+
const decision = require('../lib/decision');
|
|
17
18
|
const { scaffoldInit } = require('../lib/init');
|
|
18
19
|
|
|
19
20
|
// Output dirs default to the USER's current directory, not the package dir, so a
|
|
@@ -40,6 +41,24 @@ function writeTranscripts(receipt, transcripts) {
|
|
|
40
41
|
|
|
41
42
|
const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
|
|
42
43
|
|
|
44
|
+
// ── THE DISPLAY VERIFIES BEFORE IT READS (spec 026 AC-20, F7) ────────────────
|
|
45
|
+
// badge, diff and export read a receipt somebody else may have edited. Until
|
|
46
|
+
// this, badge never checked the hash and diff and export printed a warning
|
|
47
|
+
// and proceeded, so a receipt with one digit moved and the hash left as it
|
|
48
|
+
// was rendered `passing` (028's D-5, for the CLI). A receipt whose
|
|
49
|
+
// receipt_hash does not verify is refused here: nothing on stdout, no file,
|
|
50
|
+
// the reason on stderr naming receipt_hash and the file, exit 4. The bound the
|
|
51
|
+
// threat model states does not move: the same edit followed by a re-seal
|
|
52
|
+
// (sealReceipt) verifies and renders, which is what receipt_hash is: integrity
|
|
53
|
+
// since sealing, not authenticity. The check lives here, in the three
|
|
54
|
+
// commands, and not in lib/verdict.js's writer, which the repo gate calls on
|
|
55
|
+
// synthetic unsealed receipts.
|
|
56
|
+
function refuseUnverified(receipt, file, command) {
|
|
57
|
+
if (verifyReceiptHash(receipt)) return;
|
|
58
|
+
console.error(` ✗ REFUSED (${command}): receipt_hash does not verify for ${path.basename(file)} (tampered or hand-edited since it was sealed); nothing rendered.`);
|
|
59
|
+
process.exit(4);
|
|
60
|
+
}
|
|
61
|
+
|
|
43
62
|
// Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
|
|
44
63
|
// in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
|
|
45
64
|
// flags always win over the rc; the rc wins over built-in defaults.
|
|
@@ -58,6 +77,81 @@ function loadRc(skillDir) {
|
|
|
58
77
|
// positional instead of swallowing it as the flag's value.
|
|
59
78
|
const BOOLEAN_FLAGS = new Set(['trusted-skill']);
|
|
60
79
|
|
|
80
|
+
// ── THE INPUT CONTRACT (spec 026 AC-9, F5) ────────────────────────────────────
|
|
81
|
+
// A COPY of action/lib.sh's RE_* rules, in the shape spec 028's fixture holds
|
|
82
|
+
// them; specs/026-receipt-integrity/fixtures/input-contract.json carries the
|
|
83
|
+
// same copy and AC-9 asserts the three identical, rule for rule, both ways.
|
|
84
|
+
// Consolidating into one shared module is 028's decision 1, the spec after 026.
|
|
85
|
+
//
|
|
86
|
+
// Why a contract at the door: `parseInt('abc', 10)` is NaN, both cost guards
|
|
87
|
+
// are `>` comparisons against it, both are false, and the run completed and
|
|
88
|
+
// wrote a receipt that carried no trace of it (the dollar cap never compared;
|
|
89
|
+
// the call cap silently became the dev default). `--samples abc` was quieter
|
|
90
|
+
// still: `NaN || 5` is 5. Every numeric input, from the command line or the
|
|
91
|
+
// rc, is checked against its rule BEFORE the projection is printed and before
|
|
92
|
+
// any call; a flag given with no value (the parser records `true`) and an
|
|
93
|
+
// empty string are refused the same way, not defaulted.
|
|
94
|
+
const INPUT_CONTRACT = {
|
|
95
|
+
rules: {
|
|
96
|
+
models: { re: /^[A-Za-z0-9._-]+(,[A-Za-z0-9._-]+)*$/, message: 'models: expected a comma-separated list of model ids (letters, digits, . _ -)' },
|
|
97
|
+
max_usd: { re: /^[0-9]+(\.[0-9]+)?$/, message: 'max-usd: expected a positive decimal number such as 20 or 3.53' },
|
|
98
|
+
max_usd_zero: { re: /^0+(\.0+)?$/, reject_on_match: true, message: 'max-usd: must be greater than zero' },
|
|
99
|
+
max_calls: { re: /^[1-9][0-9]*$/, message: 'max-calls: expected a positive integer with no leading zero' },
|
|
100
|
+
skill_dir_control: { re: /[\x00-\x1f\x7f]/, reject_on_match: true, message: 'skill-dir: contains a control character' },
|
|
101
|
+
},
|
|
102
|
+
// Which rule each `driftproof run` input is checked against. samples,
|
|
103
|
+
// max-cases and concurrency are positive integers and reuse max_calls' rule.
|
|
104
|
+
flags: { 'models': ['models'], 'max-usd': ['max_usd', 'max_usd_zero'], 'max-calls': ['max_calls'], 'samples': ['max_calls'], 'max-cases': ['max_calls'], 'concurrency': ['max_calls'], 'skill-dir': ['skill_dir_control'] },
|
|
105
|
+
};
|
|
106
|
+
|
|
107
|
+
function shownValue(v) {
|
|
108
|
+
if (v === true) return '<no value>';
|
|
109
|
+
return JSON.stringify(String(v));
|
|
110
|
+
}
|
|
111
|
+
// Refuse one input: the rule's own message, the input's name and the value it
|
|
112
|
+
// carried, exit 2. Nothing has been printed or spent when this runs.
|
|
113
|
+
function refuseInput(name, value, message, where) {
|
|
114
|
+
console.error(` ✗ REFUSED: ${message}, got ${shownValue(value)} (${where} ${name}); nothing run, no receipt written.`);
|
|
115
|
+
process.exit(2);
|
|
116
|
+
}
|
|
117
|
+
// Check one input against every rule its flag maps to. `value` is what was
|
|
118
|
+
// given: a string, a number (from the rc), `true` (a flag with no value) or
|
|
119
|
+
// an empty string; only a string that matches every rule passes.
|
|
120
|
+
function checkInput(flag, value, where, shownName = flag) {
|
|
121
|
+
const rules = INPUT_CONTRACT.flags[flag] || [];
|
|
122
|
+
const text = typeof value === 'string' ? value : (typeof value === 'number' && Number.isFinite(value)) ? String(value) : null;
|
|
123
|
+
if (text === null) refuseInput(shownName, value, `${flag}: expected a value`, where);
|
|
124
|
+
for (const key of rules) {
|
|
125
|
+
const rule = INPUT_CONTRACT.rules[key];
|
|
126
|
+
const hit = rule.re.test(text);
|
|
127
|
+
if (rule.reject_on_match ? hit : !hit) refuseInput(shownName, value, rule.message, where);
|
|
128
|
+
}
|
|
129
|
+
return text;
|
|
130
|
+
}
|
|
131
|
+
// Every numeric input the run takes, from the command line first and the rc
|
|
132
|
+
// second, checked before anything is printed. Returns the checked strings.
|
|
133
|
+
// The registry door (spec 026 AC-10, F6): every target and the judge it will
|
|
134
|
+
// run with must be a model the registry knows. lib/models.js assertRegistered
|
|
135
|
+
// is the one rule; lib/run.js asks it too on the path every caller shares, and
|
|
136
|
+
// this door is where bin/driftproof says the registry's path in its own
|
|
137
|
+
// message, before the suite loads.
|
|
138
|
+
function checkRegistered(models, judgeFor) {
|
|
139
|
+
try {
|
|
140
|
+
for (const m of models) { assertRegistered(m, 'model'); assertRegistered(judgeFor(m), 'judge model'); }
|
|
141
|
+
} catch (e) {
|
|
142
|
+
if (e && e.code === 'UNREGISTERED_MODEL') { console.error(` ✗ REFUSED (${/judge/.test(e.message) ? 'judge-model / judge_model' : 'models'}): ${e.message}`); process.exit(2); }
|
|
143
|
+
throw e;
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
function checkNumericInputs(flags, rc) {
|
|
147
|
+
const out = {};
|
|
148
|
+
for (const [flag, rcKey] of [['max-calls', 'max_calls'], ['max-usd', 'max_usd'], ['samples', 'samples'], ['max-cases', 'max_cases'], ['concurrency', 'concurrency']]) {
|
|
149
|
+
if (Object.prototype.hasOwnProperty.call(flags, flag)) out[flag] = checkInput(flag, flags[flag], 'flag --');
|
|
150
|
+
else if (rc[rcKey] !== undefined && rc[rcKey] !== null) out[flag] = checkInput(flag, rc[rcKey], '.driftproofrc', rcKey);
|
|
151
|
+
}
|
|
152
|
+
return out;
|
|
153
|
+
}
|
|
154
|
+
|
|
61
155
|
function parseArgs(argv) {
|
|
62
156
|
const positional = [];
|
|
63
157
|
const flags = {};
|
|
@@ -83,7 +177,9 @@ USAGE
|
|
|
83
177
|
[--keep-transcripts] [--out DIR] [--trusted-skill]
|
|
84
178
|
${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE] [--mode release|revision]
|
|
85
179
|
${PROJECT_NAME} validate <receipt.json>
|
|
86
|
-
${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
|
|
180
|
+
${PROJECT_NAME} badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]
|
|
181
|
+
${PROJECT_NAME} decide <receipts-dir> --models a,b [--github-output] [--badge FILE]
|
|
182
|
+
[--summary FILE] [--enforce] [--fail-on-regression true|false]
|
|
87
183
|
${PROJECT_NAME} import <results.json> --from agent-skills-eval|skillgrade [--out DIR]
|
|
88
184
|
${PROJECT_NAME} export <receipt.json> [--to summary-json] [--report-url URL] [--out FILE]
|
|
89
185
|
|
|
@@ -112,9 +208,10 @@ NOTES
|
|
|
112
208
|
- --keep-transcripts writes the raw generations + judge outputs to
|
|
113
209
|
transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
|
|
114
210
|
local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
|
|
115
|
-
- Prices and registry status come from config/models.json
|
|
116
|
-
|
|
117
|
-
|
|
211
|
+
- Prices and registry status come from config/models.json (or the registry
|
|
212
|
+
DRIFTPROOF_REGISTRY names). An unregistered model id is REFUSED before any call,
|
|
213
|
+
naming the id and the registry path (so is one not shaped like a model id);
|
|
214
|
+
registry:"unregistered" is an import-only receipt value, refused for a run.
|
|
118
215
|
- --concurrency runs that many (case,mode) tasks at once (default 1). Higher
|
|
119
216
|
values cut wall-clock on the cli surface (cold-start dominated).
|
|
120
217
|
- ISOLATION (default). On the two cli surfaces every \`claude\` / \`codex\` spawn
|
|
@@ -144,23 +241,53 @@ async function cmdRun(positional, flags) {
|
|
|
144
241
|
|
|
145
242
|
// Per-project defaults from .driftproofrc (if any); CLI flags override these.
|
|
146
243
|
const rc = loadRc(skillDir);
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
244
|
+
// The contract, at the door (spec 026 AC-9): every input checked against
|
|
245
|
+
// its rule before the projection and before any call. A value that fails
|
|
246
|
+
// its shape is refused naming the input and the value; nothing is defaulted.
|
|
247
|
+
if (INPUT_CONTRACT.rules.skill_dir_control.re.test(String(skillDir))) refuseInput('skill-dir', skillDir, INPUT_CONTRACT.rules.skill_dir_control.message, 'argument');
|
|
248
|
+
const checked = checkNumericInputs(flags, rc);
|
|
249
|
+
const modelsRaw = Object.prototype.hasOwnProperty.call(flags, 'models') ? flags.models : (rc.models !== undefined && rc.models !== null) ? rc.models : 'haiku';
|
|
250
|
+
const models = String(modelsRaw).split(',').map((s) => s.trim()).filter(Boolean);
|
|
151
251
|
const judgeModel = flags['judge-model'] || rc.judge_model || null;
|
|
152
|
-
|
|
153
|
-
|
|
252
|
+
// Spec 026 AC-10 (F6): every model id, and the judge, must be a model the
|
|
253
|
+
// registry knows; refused here naming the input, the id and the registry's
|
|
254
|
+
// absolute path, before the suite loads, before the projection, before any
|
|
255
|
+
// call. A shape failure is refused the same way (the contract's models rule
|
|
256
|
+
// below then never sees one). The same check stands in lib/run.js for every
|
|
257
|
+
// other caller.
|
|
258
|
+
const judgeFor = (m) => (judgeModel ? resolveModel(judgeModel) : resolveModel(m));
|
|
259
|
+
checkRegistered(models, judgeFor);
|
|
260
|
+
const modelsGiven = Object.prototype.hasOwnProperty.call(flags, 'models') ? checkInput('models', flags.models, 'flag --')
|
|
261
|
+
: (rc.models !== undefined && rc.models !== null) ? checkInput('models', rc.models, '.driftproofrc', 'models') : 'haiku';
|
|
262
|
+
void modelsGiven;
|
|
263
|
+
const maxCases = checked['max-cases'] !== undefined ? parseInt(checked['max-cases'], 10) : null;
|
|
264
|
+
const maxCalls = checked['max-calls'] !== undefined ? parseInt(checked['max-calls'], 10) : DEV_MAX_CALLS;
|
|
265
|
+
const samples = checked.samples !== undefined ? parseInt(checked.samples, 10) : DEFAULT_JUDGE_SAMPLES;
|
|
266
|
+
const concurrency = checked.concurrency !== undefined ? parseInt(checked.concurrency, 10) : 1;
|
|
267
|
+
const maxUsd = checked['max-usd'] !== undefined ? parseFloat(checked['max-usd']) : DEV_MAX_USD;
|
|
154
268
|
const keepTranscripts = !!flags['keep-transcripts'];
|
|
155
|
-
// spec 022: the same-user spawn exists only behind this flag.
|
|
156
|
-
//
|
|
269
|
+
// spec 022: the same-user spawn exists only behind this flag.
|
|
270
|
+
//
|
|
271
|
+
// THE EVAL USER IS NOT RESOLVED HERE (F-022-4, spec 021). It used to be, on
|
|
272
|
+
// every run that was not --trusted-skill, including an api-surface-only run
|
|
273
|
+
// that spawns no CLI at all - so a hostile DRIFTPROOF_EVAL_USER refused a run
|
|
274
|
+
// that would never have read it. Fail-loud and harmless, and still wrong: a
|
|
275
|
+
// variable that governs a lane this run does not take should be neither read
|
|
276
|
+
// nor validated. It is resolved below, once a subscription surface is actually
|
|
277
|
+
// in the run, which is the same condition the isolation banner already used.
|
|
157
278
|
const trusted = !!flags['trusted-skill'];
|
|
158
|
-
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
159
279
|
const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
|
|
160
280
|
fs.mkdirSync(outDir, { recursive: true });
|
|
161
281
|
|
|
162
282
|
const skill = loadSkill(skillDir);
|
|
163
283
|
const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
|
|
284
|
+
// Spec 026 AC-5 (F2): a suite with no cases measures nothing, and a receipt
|
|
285
|
+
// over nothing is a receipt about nothing. Refused here, before the
|
|
286
|
+
// projection and before any call, naming the suite file.
|
|
287
|
+
if (nCases < 1) {
|
|
288
|
+
console.error(` ✗ REFUSED: the suite at ${path.join(path.resolve(skillDir), 'evals', 'evals.json')} has no cases${maxCases ? ` (after --max-cases ${maxCases})` : ''}; nothing to measure, no receipt written.`);
|
|
289
|
+
process.exit(2);
|
|
290
|
+
}
|
|
164
291
|
// v0.5 draws the generation up to SAMPLING.max times per arm, so BOTH the
|
|
165
292
|
// printed projection and the dollar guard below must be scaled by it. Left at
|
|
166
293
|
// draws=1 the guard would admit a run costing up to ten times its own
|
|
@@ -174,7 +301,13 @@ async function cmdRun(positional, flags) {
|
|
|
174
301
|
// `api` surface this is real spend, so we refuse if it would exceed --max-usd.
|
|
175
302
|
// On `claude-cli` the metered spend is $0 (subscription); the figure is the
|
|
176
303
|
// hypothetical "if run on the metered API" cost — printed, never blocks.
|
|
177
|
-
|
|
304
|
+
// Spec 026 AC-10: the projection is priced with the judge the run will USE
|
|
305
|
+
// (lib/run.js: --judge-model when given, else the target itself), never a
|
|
306
|
+
// fixed haiku; each target is priced with its own judge and the figures summed.
|
|
307
|
+
const targets = models.map((m) => resolveModel(m));
|
|
308
|
+
const judges = models.map((m) => judgeFor(m));
|
|
309
|
+
const perModelCost = targets.map((t, i) => estimateRunCostUSD({ caseCount: nCases, draws: SAMPLING.max, samples, models: [targets[i]], judgeModel: judges[i] }));
|
|
310
|
+
const cost = { totalUSD: Math.round(perModelCost.reduce((a, c) => a + c.totalUSD, 0) * 1e4) / 1e4, perModel: perModelCost.flatMap((c) => c.perModel), judges };
|
|
178
311
|
// Surface is per-model now (a run may mix an Anthropic and an OpenAI target).
|
|
179
312
|
const surfaces = [...new Set(models.map((m) => surfaceForModel(m)))];
|
|
180
313
|
const surface = surfaces.join(', ');
|
|
@@ -182,10 +315,17 @@ async function cmdRun(positional, flags) {
|
|
|
182
315
|
|
|
183
316
|
const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
|
|
184
317
|
console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
|
|
318
|
+
// Spec 026 AC-1: a stub run says so before it starts, and its receipt says so
|
|
319
|
+
// after (surface stub, answered_by.kind stub, UNVERIFIED).
|
|
320
|
+
if (process.env.DRIFTPROOF_STUB === '1') console.log(' STUB RUN (DRIFTPROOF_STUB=1): nothing will answer; the receipt will be UNVERIFIED and measure nothing');
|
|
185
321
|
console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
|
|
186
322
|
console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
|
|
187
323
|
console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
|
|
188
|
-
|
|
324
|
+
console.log(` judge: ${[...new Set(cost.judges)].join(', ')} (${judgeModel ? '--judge-model' : 'the target model itself; set --judge-model to change'})`);
|
|
325
|
+
if (subSurfaces.length) {
|
|
326
|
+
const isolation = trusted ? 'same-user (--trusted-skill: self-authored skill)' : `eval-user (${evalUser()})`;
|
|
327
|
+
console.log(` isolation: ${isolation}`);
|
|
328
|
+
}
|
|
189
329
|
console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
|
|
190
330
|
console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
|
|
191
331
|
if (subSurfaces.length) console.log(` actual metered spend on ${subSurfaces.join(', ')}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
|
|
@@ -227,6 +367,12 @@ async function cmdRun(positional, flags) {
|
|
|
227
367
|
},
|
|
228
368
|
});
|
|
229
369
|
} catch (e) {
|
|
370
|
+
if (e && e.code === 'SUBSTRATE_MISMATCH') {
|
|
371
|
+
// Spec 026 AC-2: the surface answered as a different model. The run
|
|
372
|
+
// stopped before its next call and wrote no receipt.
|
|
373
|
+
console.error(`\n ✗ REFUSED (substrate mismatch): ${e.message}`);
|
|
374
|
+
process.exit(5);
|
|
375
|
+
}
|
|
230
376
|
if (e && e.code === 'BUDGET_HARDSTOP') {
|
|
231
377
|
console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
|
|
232
378
|
console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
|
|
@@ -264,7 +410,14 @@ async function cmdRun(positional, flags) {
|
|
|
264
410
|
const aw = receipt.results.aggregates.with_skill;
|
|
265
411
|
const ab = receipt.results.aggregates.baseline;
|
|
266
412
|
console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
|
|
267
|
-
console.log(` →
|
|
413
|
+
console.log(` → ${answeredLine(receipt)}; verification_level ${receipt.verification_level}`);
|
|
414
|
+
// Spec 026 AC-8: how many draws were cut at the output cap and excluded.
|
|
415
|
+
const nTruncated = receipt.results.cases.reduce((a, c) => a + (((c.generation || {}).n_truncated) || 0), 0);
|
|
416
|
+
if (nTruncated) console.log(` → ${nTruncated} draw(s) truncated at the output cap: unmeasured, never judged, excluded from every band`);
|
|
417
|
+
// Spec 026 AC-6, AC-7: the band rule is named beside the band, and a band
|
|
418
|
+
// the formula could not form prints as n/a, never as 0.000.
|
|
419
|
+
if (cmp.delta == null) console.log(` → skill lift n/a (${cmp.delta_uncertainty_unavailable === 'no_cases' ? 'no case was included on an arm' : 'no comparison'})\n`);
|
|
420
|
+
else console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${uncertaintyStr(cmp)} (with ${band(aw.mean_score, aw.stddev)} vs base ${band(ab.mean_score, ab.stddev)}; band = sample stddev of the per-case means)\n`);
|
|
268
421
|
}
|
|
269
422
|
|
|
270
423
|
console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
|
|
@@ -276,10 +429,8 @@ function cmdDiff(positional, flags) {
|
|
|
276
429
|
const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
|
|
277
430
|
const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
|
|
278
431
|
|
|
279
|
-
//
|
|
280
|
-
for (const [p, r] of [[aPath, a], [bPath, b]])
|
|
281
|
-
if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
|
|
282
|
-
}
|
|
432
|
+
// Spec 026 AC-20: a side whose hash does not verify is refused, naming it.
|
|
433
|
+
for (const [p, r] of [[aPath, a], [bPath, b]]) refuseUnverified(r, p, 'diff');
|
|
283
434
|
|
|
284
435
|
// --mode revision inverts the axis: the skill text is the variable under test
|
|
285
436
|
// and the substrate is the control. The fields release drift merely warns about
|
|
@@ -356,8 +507,14 @@ See AUTHORING.md for how to write a fair suite.`);
|
|
|
356
507
|
// --github-output prints verdict/delta/message/color as key=value lines.
|
|
357
508
|
function cmdBadge(positional, flags) {
|
|
358
509
|
const p = positional[0];
|
|
359
|
-
if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
|
|
510
|
+
if (!p) { console.error('usage: driftproof badge <receipt.json|receipts-dir> [--models a,b] [--out FILE] [--github-output]'); process.exit(2); }
|
|
511
|
+
// SPEC 030 AC-3. A DIRECTORY renders the worst decision across every receipt
|
|
512
|
+
// in it, naming the model that state came from. A single receipt renders
|
|
513
|
+
// exactly what it always did - lib/verdict.js is untouched and still answers
|
|
514
|
+
// the one-receipt question, which AC-3's control asserts.
|
|
515
|
+
if (fs.existsSync(p) && fs.statSync(p).isDirectory()) return cmdBadgeSet(p, flags);
|
|
360
516
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
517
|
+
refuseUnverified(receipt, p, 'badge');
|
|
361
518
|
if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
|
|
362
519
|
const badge = badgeEndpoint(receipt);
|
|
363
520
|
const json = JSON.stringify(badge, null, 2);
|
|
@@ -372,6 +529,113 @@ function cmdBadge(positional, flags) {
|
|
|
372
529
|
}
|
|
373
530
|
}
|
|
374
531
|
|
|
532
|
+
// The badge over a receipt SET (spec 030 AC-3): the worst decision, the model
|
|
533
|
+
// it came from, and how many models share it. Every receipt in the directory is
|
|
534
|
+
// read; when --models is given the requested list governs, so a model with no
|
|
535
|
+
// receipt is `refused` and can be the worst state.
|
|
536
|
+
function cmdBadgeSet(dir, flags) {
|
|
537
|
+
const files = decision.receiptFiles(dir);
|
|
538
|
+
if (files.length === 0) {
|
|
539
|
+
console.error(` \u2717 REFUSED (badge): ${dir} holds no receipts; a badge over nothing is a badge about nothing.`);
|
|
540
|
+
process.exit(2);
|
|
541
|
+
}
|
|
542
|
+
// Every receipt is verified before it is displayed, exactly as the
|
|
543
|
+
// single-receipt path does (spec 026 AC-20): a set is not a way around the
|
|
544
|
+
// check, and the file that fails is named.
|
|
545
|
+
for (const f of files) {
|
|
546
|
+
const r = JSON.parse(fs.readFileSync(f, 'utf8'));
|
|
547
|
+
refuseUnverified(r, f, 'badge');
|
|
548
|
+
}
|
|
549
|
+
const models = flags.models || files
|
|
550
|
+
.map((f) => JSON.parse(fs.readFileSync(f, 'utf8')))
|
|
551
|
+
.map((r) => r.run && r.run.model_id).filter(Boolean).join(',');
|
|
552
|
+
const d = decision.decideSet(dir, models);
|
|
553
|
+
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); return; }
|
|
554
|
+
const badge = decision.badgeEndpointForSet(d);
|
|
555
|
+
const json = JSON.stringify(badge, null, 2);
|
|
556
|
+
if (flags.out) {
|
|
557
|
+
const out = path.resolve(flags.out);
|
|
558
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
559
|
+
fs.writeFileSync(out, json + '\n');
|
|
560
|
+
console.log(`badge written to ${flags.out} (${d.worst}: ${badge.message}, ${badge.color})`);
|
|
561
|
+
} else {
|
|
562
|
+
console.log(json);
|
|
563
|
+
}
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// The decision over a RUN: every receipt it produced, against every model it was
|
|
567
|
+
// asked for (spec 030 AC-1, AC-2, AC-4).
|
|
568
|
+
//
|
|
569
|
+
// `badge` answers about one receipt and still does. This answers about the set,
|
|
570
|
+
// which is what the GitHub Action needs and what `ls … | head -1` was standing
|
|
571
|
+
// in for. Modes compose: --github-output emits the step outputs, --summary
|
|
572
|
+
// writes the job-summary table, --enforce decides the job and sets the exit
|
|
573
|
+
// code. With none of them it prints the decision as JSON.
|
|
574
|
+
function cmdDecide(positional, flags) {
|
|
575
|
+
const dir = positional[0];
|
|
576
|
+
if (!dir || !flags.models) {
|
|
577
|
+
console.error('usage: driftproof decide <receipts-dir> --models a,b [--github-output] [--badge FILE] [--summary FILE] [--enforce] [--fail-on-regression true|false]');
|
|
578
|
+
process.exit(2);
|
|
579
|
+
}
|
|
580
|
+
if (!fs.existsSync(dir) || !fs.statSync(dir).isDirectory()) {
|
|
581
|
+
console.error(` \u2717 REFUSED (decide): ${dir} is not a directory; a run that wrote no receipt directory decided nothing.`);
|
|
582
|
+
process.exit(2);
|
|
583
|
+
}
|
|
584
|
+
// ── THE SEAL IS VERIFIED BEFORE ANY DECISION (spec 026 AC-20, spec 030 AC-3)
|
|
585
|
+
//
|
|
586
|
+
// Before spec 030 the action rendered through `driftproof badge <receipt>`,
|
|
587
|
+
// which calls refuseUnverified and exits 4 on a receipt_hash mismatch. All
|
|
588
|
+
// four surfaces the action writes now go through THIS command, and until
|
|
589
|
+
// this it verified nothing: a receipt with one field hand-edited after
|
|
590
|
+
// sealing rendered `passing`, `brightgreen`, exit 0 - the check that had
|
|
591
|
+
// stood between the receipts and the job was not carried across to the
|
|
592
|
+
// command that replaced its caller. The failure direction is FAIL-OPEN,
|
|
593
|
+
// which is the one direction an adopter cannot detect from their own CI.
|
|
594
|
+
//
|
|
595
|
+
// Every receipt in the set, before anything is read for a decision and
|
|
596
|
+
// before any file is written, exactly as cmdBadgeSet does - a set is not a
|
|
597
|
+
// way around the check, and the file that fails is named.
|
|
598
|
+
//
|
|
599
|
+
// A receipt that cannot be PARSED is left to the decision, not refused here:
|
|
600
|
+
// `decideSet` distinguishes a receipt that is absent from one that exists and
|
|
601
|
+
// is unreadable and fails closed on both, naming which (spec 030 AC-2,
|
|
602
|
+
// absence-vs-unreadable). Refusing it here would collapse that distinction
|
|
603
|
+
// into an exit code.
|
|
604
|
+
for (const f of decision.receiptFiles(dir)) {
|
|
605
|
+
let receipt;
|
|
606
|
+
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (_e) { continue; }
|
|
607
|
+
refuseUnverified(receipt, f, 'decide');
|
|
608
|
+
}
|
|
609
|
+
const failOnRegression = String(flags['fail-on-regression'] ?? 'true') !== 'false';
|
|
610
|
+
const d = decision.decideSet(dir, flags.models, { failOnRegression });
|
|
611
|
+
if (d.rows.length === 0) {
|
|
612
|
+
console.error(' \u2717 REFUSED (decide): no models requested; there is nothing to decide over.');
|
|
613
|
+
process.exit(2);
|
|
614
|
+
}
|
|
615
|
+
let acted = false;
|
|
616
|
+
if (flags['github-output']) { console.log(decision.githubOutputLines(d)); acted = true; }
|
|
617
|
+
if (flags.badge) {
|
|
618
|
+
const out = path.resolve(flags.badge);
|
|
619
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
620
|
+
const badge = decision.badgeEndpointForSet(d);
|
|
621
|
+
fs.writeFileSync(out, JSON.stringify(badge, null, 2) + '\n');
|
|
622
|
+
console.log(`badge written to ${flags.badge} (${d.worst}: ${badge.message}, ${badge.color})`);
|
|
623
|
+
acted = true;
|
|
624
|
+
}
|
|
625
|
+
if (flags.summary) {
|
|
626
|
+
const out = path.resolve(flags.summary);
|
|
627
|
+
fs.mkdirSync(path.dirname(out), { recursive: true });
|
|
628
|
+
fs.appendFileSync(out, decision.summaryMarkdown(d) + '\n');
|
|
629
|
+
acted = true;
|
|
630
|
+
}
|
|
631
|
+
if (flags.enforce) {
|
|
632
|
+
for (const line of decision.enforcementLines(d, { failOnRegression })) console.log(line);
|
|
633
|
+
acted = true;
|
|
634
|
+
if (d.fails) process.exit(1);
|
|
635
|
+
}
|
|
636
|
+
if (!acted) console.log(JSON.stringify(d, null, 2));
|
|
637
|
+
}
|
|
638
|
+
|
|
375
639
|
// Convert another tool's results file into a valid DECLARED receipt (interop,
|
|
376
640
|
// Phase 7). The converted receipt is validated + self-hash-verified before it
|
|
377
641
|
// is written; a failed conversion writes nothing.
|
|
@@ -408,7 +672,7 @@ function cmdExport(positional, flags) {
|
|
|
408
672
|
if (to !== 'summary-json') { console.error(`unknown export target "${to}" — supported: summary-json`); process.exit(2); }
|
|
409
673
|
const { toSummaryJson } = require('../lib/export');
|
|
410
674
|
const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
|
|
411
|
-
|
|
675
|
+
refuseUnverified(receipt, p, 'export');
|
|
412
676
|
const summary = toSummaryJson(receipt, { reportUrl: typeof flags['report-url'] === 'string' ? flags['report-url'] : null });
|
|
413
677
|
const json = JSON.stringify(summary, null, 2);
|
|
414
678
|
if (flags.out) {
|
|
@@ -430,6 +694,7 @@ async function main() {
|
|
|
430
694
|
case 'diff': return cmdDiff(positional, flags);
|
|
431
695
|
case 'validate': return cmdValidate(positional);
|
|
432
696
|
case 'badge': return cmdBadge(positional, flags);
|
|
697
|
+
case 'decide': return cmdDecide(positional, flags);
|
|
433
698
|
case 'import': return cmdImport(positional, flags);
|
|
434
699
|
case 'export': return cmdExport(positional, flags);
|
|
435
700
|
case 'version': case '--version': case '-v': console.log(RUNNER_VERSION); return;
|
package/config.js
CHANGED
|
@@ -9,7 +9,7 @@ const PROJECT_NAME = 'driftproof';
|
|
|
9
9
|
// Bumped whenever the runner's behaviour or receipt-generation semantics change
|
|
10
10
|
// in a way that could affect results. Recorded into every receipt as
|
|
11
11
|
// run.runner_version so a receipt is reproducible against a known engine.
|
|
12
|
-
const RUNNER_VERSION = '0.
|
|
12
|
+
const RUNNER_VERSION = '0.10.0';
|
|
13
13
|
|
|
14
14
|
// The eval format we CONSUME (we deliberately do not invent our own).
|
|
15
15
|
const SUITE_FORMAT = 'agentskills.io/evals';
|
|
@@ -29,7 +29,30 @@ const SUITE_FORMAT = 'agentskills.io/evals';
|
|
|
29
29
|
// at run time so derived dollars stay reproducible), and the derived
|
|
30
30
|
// `economics` block. v0.3.1 is frozen as receipt.v0.3.1.schema.json;
|
|
31
31
|
// v0.1/v0.2/v0.3/v0.3.1 receipts all still load.
|
|
32
|
-
|
|
32
|
+
// v0.5 samples the GENERATION n times per arm (per-case `generation` draw list,
|
|
33
|
+
// the across-draw band, the variance ratio, the sampling policy applied)
|
|
34
|
+
// and adds the per-suite canary. v0.4 is frozen as receipt.v0.4.schema.json.
|
|
35
|
+
// v0.6 (spec 026, receipt integrity) says WHAT ANSWERED: run.answered_by, a
|
|
36
|
+
// `stub` surface, run.judge.model_id + prompt_template_hash, per-draw
|
|
37
|
+
// stop_reason/truncated and n_truncated, case_status failed_unmeasured,
|
|
38
|
+
// results.aggregates.band_rule, and bands that are null where the formula
|
|
39
|
+
// cannot form. NOT additive for the validator: a v0.5 receipt restamped 0.6
|
|
40
|
+
// is refused. v0.5 is frozen as receipt.v0.5.schema.json.
|
|
41
|
+
const RECEIPT_SCHEMA_VERSION = '0.6';
|
|
42
|
+
|
|
43
|
+
// Input bounds on the skill loader and the post-checks (spec 026 AC-13, audit
|
|
44
|
+
// A3/A7). lib/skill.js loaded every bundled file with no count, depth, byte,
|
|
45
|
+
// case, prompt or rubric bound, so a 2 MB prompt went to the model at the
|
|
46
|
+
// fixed 250-token projection; lib/checks.js compiled a suite-supplied regex
|
|
47
|
+
// with no bound, and JavaScript cannot time a regex out. Each is set well
|
|
48
|
+
// above every suite this repository holds: they bind pathological input, not
|
|
49
|
+
// the archive. Named in the error a bound raises, and in AUTHORING.md.
|
|
50
|
+
const SKILL_MAX_FILES = 2000; // bundled files under a skill directory
|
|
51
|
+
const SKILL_MAX_BYTES = 32 * 1024 * 1024; // bundled bytes, all files together
|
|
52
|
+
const SKILL_MAX_DEPTH = 16; // directory depth below the skill dir
|
|
53
|
+
const SUITE_MAX_CASES = 500; // cases in one evals.json
|
|
54
|
+
const CASE_MAX_CHARS = 65536; // characters in one prompt or rubric
|
|
55
|
+
const CHECK_MAX_PATTERN = 256; // characters in one checks[] regex
|
|
33
56
|
|
|
34
57
|
// Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
|
|
35
58
|
// these. The projection is refused before any call if it exceeds the cap, on
|
|
@@ -133,6 +156,7 @@ const REPORT_005_JUDGE_MODEL = 'claude-haiku-4-5';
|
|
|
133
156
|
|
|
134
157
|
module.exports = {
|
|
135
158
|
PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES,
|
|
159
|
+
SKILL_MAX_FILES, SKILL_MAX_BYTES, SKILL_MAX_DEPTH, SUITE_MAX_CASES, CASE_MAX_CHARS, CHECK_MAX_PATTERN,
|
|
136
160
|
EFFECT_FLOOR, DEV_MAX_USD, DEV_MAX_CALLS, REPORT_MAX_USD, TRIGGER_MAX_USD,
|
|
137
161
|
GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX, GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
|
|
138
162
|
REPORT_002_CLAUDE_MODEL, REPORT_002_GPT_MODEL, REPORT_002_JUDGE_MODEL,
|