driftproof 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/driftproof ADDED
@@ -0,0 +1,325 @@
1
+ #!/usr/bin/env node
2
+ // SPDX-License-Identifier: Apache-2.0
3
+ 'use strict';
4
+
5
+ const fs = require('fs');
6
+ const path = require('path');
7
+ const { PROJECT_NAME, RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES, DEV_MAX_USD } = require('../config');
8
+ const { loadSkill } = require('../lib/skill');
9
+ const { runSkillOnModel, summarizeReceipt, projectCalls } = require('../lib/run');
10
+ const { validateReceipt, verifyReceiptHash } = require('../lib/receipt');
11
+ const { buildDriftReport } = require('../lib/diff');
12
+ const { surfaceLabel, providerName, resolveModel } = require('../lib/provider');
13
+ const { estimateRunCostUSD, BudgetTracker } = require('../lib/cost');
14
+ const { registryStatus } = require('../lib/models');
15
+ const { verdictFromReceipt, badgeEndpoint, githubOutputLines } = require('../lib/verdict');
16
+ const { scaffoldInit } = require('../lib/init');
17
+
18
+ // Output dirs default to the USER's current directory, not the package dir, so a
19
+ // global/npx install writes receipts/transcripts into the project being tested
20
+ // (never into a read-only global install). --out overrides the receipts dir.
21
+ const TRANSCRIPTS_DIR = path.join(process.cwd(), 'transcripts');
22
+
23
+ // Write retained transcripts for one receipt to transcripts/<receipt_hash>/.
24
+ // One JSON file per (case, mode); gitignored by default. Called only when
25
+ // --keep-transcripts is set (run.transcripts === 'retained-local').
26
+ function writeTranscripts(receipt, transcripts) {
27
+ const dir = path.join(TRANSCRIPTS_DIR, receipt.receipt_hash);
28
+ fs.mkdirSync(dir, { recursive: true });
29
+ const manifest = { receipt_hash: receipt.receipt_hash, model_id: receipt.run.model_id, date_utc: receipt.run.date_utc, entries: [] };
30
+ for (const t of transcripts) {
31
+ if (!t) continue;
32
+ const base = `${slug(t.id)}-${t.mode}`;
33
+ fs.writeFileSync(path.join(dir, `${base}.json`), JSON.stringify({ id: t.id, mode: t.mode, generation: t.generation, judge_outputs: t.judge_outputs }, null, 2));
34
+ manifest.entries.push(`${base}.json`);
35
+ }
36
+ fs.writeFileSync(path.join(dir, 'index.json'), JSON.stringify(manifest, null, 2));
37
+ return dir;
38
+ }
39
+
40
+ const RECEIPTS_DIR = path.join(process.cwd(), 'receipts');
41
+
42
+ // Read an optional .driftproofrc (JSON) for per-project run defaults. Looked up
43
+ // in the skill dir first (where `driftproof init` writes it), then the CWD. CLI
44
+ // flags always win over the rc; the rc wins over built-in defaults.
45
+ function loadRc(skillDir) {
46
+ const merged = {};
47
+ for (const dir of [process.cwd(), skillDir].filter(Boolean)) {
48
+ const p = path.join(dir, '.driftproofrc');
49
+ try {
50
+ if (fs.existsSync(p)) Object.assign(merged, JSON.parse(fs.readFileSync(p, 'utf8')));
51
+ } catch (_e) { /* a malformed rc is ignored, not fatal */ }
52
+ }
53
+ return merged;
54
+ }
55
+
56
+ function parseArgs(argv) {
57
+ const positional = [];
58
+ const flags = {};
59
+ for (let i = 0; i < argv.length; i++) {
60
+ const a = argv[i];
61
+ if (a.startsWith('--')) {
62
+ const key = a.slice(2);
63
+ const next = argv[i + 1];
64
+ if (next === undefined || next.startsWith('--')) { flags[key] = true; }
65
+ else { flags[key] = next; i++; }
66
+ } else positional.push(a);
67
+ }
68
+ return { positional, flags };
69
+ }
70
+
71
+ function usage() {
72
+ console.log(`${PROJECT_NAME} v${RUNNER_VERSION} — continuous verification of agent skills
73
+
74
+ USAGE
75
+ ${PROJECT_NAME} init <dir> scaffold SKILL.md + evals/evals.json + .driftproofrc
76
+ ${PROJECT_NAME} run <skill-dir> [--models a,b] [--samples N] [--max-cases N] [--max-calls N]
77
+ [--judge-model M] [--concurrency N] [--max-usd N]
78
+ [--keep-transcripts] [--out DIR]
79
+ ${PROJECT_NAME} diff <receiptA.json> <receiptB.json> [--out FILE]
80
+ ${PROJECT_NAME} validate <receipt.json>
81
+ ${PROJECT_NAME} badge <receipt.json> [--out FILE] [--github-output]
82
+
83
+ ENV
84
+ CLAUDE_PROVIDER api | cli (default: cli — spawns \`claude -p\`, strips ANTHROPIC_API_KEY)
85
+ ANTHROPIC_API_KEY required when CLAUDE_PROVIDER=api
86
+ DRIFTPROOF_REGISTRY path to an alternate model registry (default: the packaged config/models.json)
87
+ DRIFTPROOF_STUB=1 offline stub surface — canned receipts, zero model calls (for CI)
88
+
89
+ NOTES
90
+ - Default models list is 'haiku' only (cheap dev default).
91
+ - Sampled judging: each case does 2 generations + 2×samples judge calls.
92
+ Default --samples 5 → 12 calls/case; --samples 1 disables bands.
93
+ - --max-calls is a hard per-run call cap (default 200). The whole run is
94
+ projected up front and REFUSED before any call if it would exceed the cap.
95
+ - --max-usd is a hard dollar budget (default ${DEV_MAX_USD} for dev runs). The projected
96
+ cost is printed up front and the run is REFUSED on BOTH surfaces if it would
97
+ exceed the budget; on claude-cli the metered spend is $0 but the estimated-
98
+ equivalent api cost is counted against the cap identically. The run also
99
+ hard-stops mid-flight at 1.25× the cap.
100
+ - --keep-transcripts writes the raw generations + judge outputs to
101
+ transcripts/<receipt-hash>/ (gitignored) and records transcripts:"retained-
102
+ local" in the receipt. Default is "hashes-only" (only the sha256 hashes).
103
+ - Prices and registry status come from config/models.json; an unregistered
104
+ model still runs but is marked registry:"unregistered" and costed at the
105
+ conservative default price.
106
+ - --concurrency runs that many (case,mode) tasks at once (default 1). Higher
107
+ values cut wall-clock on the cli surface (cold-start dominated).`);
108
+ }
109
+
110
+ function slug(s) { return String(s).toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); }
111
+ function dateStamp(iso) { return iso.slice(0, 10); }
112
+
113
+ async function cmdRun(positional, flags) {
114
+ const skillDir = positional[0];
115
+ if (!skillDir) { usage(); process.exit(2); }
116
+
117
+ // Per-project defaults from .driftproofrc (if any); CLI flags override these.
118
+ const rc = loadRc(skillDir);
119
+ const models = String(flags.models || rc.models || 'haiku').split(',').map((s) => s.trim()).filter(Boolean);
120
+ const maxCases = flags['max-cases'] ? parseInt(flags['max-cases'], 10) : (rc.max_cases != null ? parseInt(rc.max_cases, 10) : null);
121
+ const maxCalls = flags['max-calls'] ? parseInt(flags['max-calls'], 10) : (rc.max_calls != null ? parseInt(rc.max_calls, 10) : 200);
122
+ const samples = flags.samples ? parseInt(flags.samples, 10) : (rc.samples != null ? parseInt(rc.samples, 10) : DEFAULT_JUDGE_SAMPLES);
123
+ const judgeModel = flags['judge-model'] || rc.judge_model || null;
124
+ const concurrency = flags.concurrency ? parseInt(flags.concurrency, 10) : (rc.concurrency != null ? parseInt(rc.concurrency, 10) : 1);
125
+ const maxUsd = flags['max-usd'] ? parseFloat(flags['max-usd']) : (rc.max_usd != null ? parseFloat(rc.max_usd) : DEV_MAX_USD);
126
+ const keepTranscripts = !!flags['keep-transcripts'];
127
+ const outDir = flags.out ? path.resolve(flags.out) : RECEIPTS_DIR;
128
+ fs.mkdirSync(outDir, { recursive: true });
129
+
130
+ const skill = loadSkill(skillDir);
131
+ const nCases = maxCases ? Math.min(maxCases, skill.suite.caseCount) : skill.suite.caseCount;
132
+ const perModelCalls = projectCalls(nCases, samples);
133
+ const totalProjected = perModelCalls * models.length;
134
+
135
+ // Dollar cost guard (Week 3): project the METERED USD cost up front. On the
136
+ // `api` surface this is real spend, so we refuse if it would exceed --max-usd.
137
+ // On `claude-cli` the metered spend is $0 (subscription); the figure is the
138
+ // hypothetical "if run on the metered API" cost — printed, never blocks.
139
+ const cost = estimateRunCostUSD({ caseCount: nCases, samples, models: models.map((m) => require('../lib/provider').resolveModel(m)), judgeModel: judgeModel || 'haiku' });
140
+ const surface = surfaceLabel();
141
+
142
+ const regStatuses = models.map((m) => `${m}:${registryStatus(m)}`);
143
+ console.log(`\n${PROJECT_NAME} run — skill "${skill.name}" v${skill.version}`);
144
+ console.log(` content_hash: ${skill.contentHash.slice(0, 16)}… suite: ${skill.suite.caseCount} cases (${skill.suite.suiteHash.slice(0, 12)}…)`);
145
+ console.log(` surface: ${surface} models: ${models.join(', ')} samples/case: ${samples} concurrency: ${concurrency}`);
146
+ console.log(` registry: ${regStatuses.join(' ')}${keepTranscripts ? ' transcripts: retained-local' : ''}`);
147
+ console.log(` projected calls: ${perModelCalls}/model × ${models.length} model(s) = ${totalProjected} per-model cap: ${maxCalls}`);
148
+ console.log(` projected cost: ~$${cost.totalUSD.toFixed(2)} (rough upper bound; budget $${maxUsd.toFixed(2)}, hard-stop $${(maxUsd * 1.25).toFixed(2)})`);
149
+ if (surface !== 'api') console.log(` actual metered spend on ${surface}: $0.00 (subscription; the $ figure is the estimated-equivalent API cost, counted against the cap identically)`);
150
+
151
+ // Call-count guard: refuse before spending anything if a single model run
152
+ // would exceed the per-run call cap.
153
+ if (perModelCalls > maxCalls) {
154
+ console.error(`\n ✗ ABORT (cost guard): projected ${perModelCalls} calls/model exceeds --max-calls ${maxCalls}.`);
155
+ console.error(` Lower --samples/--max-cases or raise --max-calls.`);
156
+ process.exit(3);
157
+ }
158
+ // Dollar guard: the projection is refused on BOTH surfaces. Subscription (cli)
159
+ // attention is not free, so its estimated-equivalent cost is counted against
160
+ // --max-usd identically to metered api spend.
161
+ if (cost.totalUSD > maxUsd) {
162
+ console.error(`\n ✗ ABORT (cost guard): projected cost ~$${cost.totalUSD.toFixed(2)} exceeds --max-usd $${maxUsd.toFixed(2)}.`);
163
+ console.error(` Trim skills/cases or raise --max-usd (do NOT lower --samples for a published run).`);
164
+ process.exit(3);
165
+ }
166
+ console.log('');
167
+
168
+ // One live budget tracker for the whole run: accumulates estimated per-call
169
+ // cost across all models and hard-stops at 1.25× the cap.
170
+ const budget = new BudgetTracker(maxUsd);
171
+
172
+ const emitted = [];
173
+ for (const model of models) {
174
+ console.log(`── model: ${model} ──`);
175
+ let result;
176
+ try {
177
+ result = await runSkillOnModel({
178
+ skill,
179
+ model,
180
+ opts: {
181
+ maxCases, maxCalls, samples, judgeModel, concurrency, budget, keepTranscripts,
182
+ onProgress: (p) => {
183
+ if (p.phase === 'done') console.log(` ${p.case} / ${p.mode}: ${p.outcome} (${p.score.toFixed(2)} ± ${(p.stddev || 0).toFixed(2)})`);
184
+ },
185
+ },
186
+ });
187
+ } catch (e) {
188
+ if (e && e.code === 'BUDGET_HARDSTOP') {
189
+ console.error(`\n ✗ ABORT (budget hard-stop): ${e.message}`);
190
+ console.error(` ${emitted.length} receipt(s) already written to ${path.relative(process.cwd(), outDir)}/.`);
191
+ process.exit(3);
192
+ }
193
+ throw e;
194
+ }
195
+ const { receipt, calls, transcripts } = result;
196
+
197
+ const { valid, errors } = validateReceipt(receipt);
198
+ if (!valid) {
199
+ console.error(' ✗ emitted receipt FAILED schema validation:', JSON.stringify(errors, null, 2));
200
+ process.exitCode = 1;
201
+ }
202
+ if (!verifyReceiptHash(receipt)) {
203
+ console.error(' ✗ receipt_hash does not verify');
204
+ process.exitCode = 1;
205
+ }
206
+
207
+ const base = `${slug(skill.name)}-${slug(receipt.run.model_id)}-${dateStamp(receipt.run.date_utc)}`;
208
+ const jsonPath = path.join(outDir, `${base}.json`);
209
+ const mdPath = path.join(outDir, `${base}.summary.md`);
210
+ fs.writeFileSync(jsonPath, JSON.stringify(receipt, null, 2));
211
+ fs.writeFileSync(mdPath, summarizeReceipt(receipt));
212
+ emitted.push(jsonPath);
213
+
214
+ // v0.3: retained transcripts are written keyed by the sealed receipt_hash,
215
+ // so a reader can verify every generation_hash / judge_sample_hash.
216
+ if (keepTranscripts && transcripts) {
217
+ const tdir = writeTranscripts(receipt, transcripts);
218
+ console.log(` → transcripts (retained-local): ${path.relative(process.cwd(), tdir)}/`);
219
+ }
220
+
221
+ const cmp = receipt.comparison;
222
+ const aw = receipt.results.aggregates.with_skill;
223
+ const ab = receipt.results.aggregates.baseline;
224
+ console.log(` → ${path.relative(process.cwd(), jsonPath)} (${calls} calls)`);
225
+ console.log(` → skill lift ${cmp.delta >= 0 ? '+' : ''}${cmp.delta.toFixed(3)} ± ${cmp.delta_uncertainty.toFixed(3)} (with ${aw.mean_score.toFixed(3)} ± ${aw.stddev.toFixed(3)} vs base ${ab.mean_score.toFixed(3)} ± ${ab.stddev.toFixed(3)})\n`);
226
+ }
227
+
228
+ console.log(`Done. ${emitted.length} receipt(s) emitted to ${path.relative(process.cwd(), outDir)}/`);
229
+ }
230
+
231
+ function cmdDiff(positional, flags) {
232
+ const [aPath, bPath] = positional;
233
+ if (!aPath || !bPath) { usage(); process.exit(2); }
234
+ const a = JSON.parse(fs.readFileSync(aPath, 'utf8'));
235
+ const b = JSON.parse(fs.readFileSync(bPath, 'utf8'));
236
+
237
+ // Warn (do not block) if either receipt fails self-verification.
238
+ for (const [p, r] of [[aPath, a], [bPath, b]]) {
239
+ if (!verifyReceiptHash(r)) console.error(` ⚠ ${path.basename(p)}: receipt_hash does not verify (tampered or hand-edited)`);
240
+ }
241
+
242
+ const labelA = dateStamp(a.run.date_utc);
243
+ const labelB = dateStamp(b.run.date_utc);
244
+ const { markdown } = buildDriftReport(a, b, { labelA, labelB });
245
+
246
+ if (flags.out) {
247
+ fs.writeFileSync(path.resolve(flags.out), markdown);
248
+ console.log(`Drift report written to ${flags.out}`);
249
+ } else {
250
+ console.log(markdown);
251
+ }
252
+ }
253
+
254
+ function cmdValidate(positional) {
255
+ const p = positional[0];
256
+ if (!p) { usage(); process.exit(2); }
257
+ const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
258
+ const { valid, errors } = validateReceipt(receipt);
259
+ const hashOk = verifyReceiptHash(receipt);
260
+ console.log(`schema: ${valid ? 'VALID' : 'INVALID'} receipt_hash: ${hashOk ? 'VERIFIED' : 'MISMATCH'}`);
261
+ if (!valid) console.log(JSON.stringify(errors, null, 2));
262
+ process.exit(valid && hashOk ? 0 : 1);
263
+ }
264
+
265
+ function cmdInit(positional) {
266
+ const target = positional[0];
267
+ if (!target) {
268
+ console.error('usage: driftproof init <dir>\n');
269
+ console.error('Scaffolds SKILL.md (stub), evals/evals.json (3 example cases), and .driftproofrc.');
270
+ process.exit(2);
271
+ }
272
+ const { dir, created, skipped } = scaffoldInit(target);
273
+ const rel = (p) => path.relative(process.cwd(), p) || '.';
274
+ console.log(`driftproof init — ${rel(dir)}/`);
275
+ for (const f of created) console.log(` + ${rel(f)}`);
276
+ for (const f of skipped) console.log(` · ${rel(f)} (exists — left untouched)`);
277
+ if (!created.length) {
278
+ console.log('\nNothing to create; every file already existed. init never overwrites.');
279
+ return;
280
+ }
281
+ console.log(`\nNext:
282
+ 1. Edit ${rel(path.join(dir, 'SKILL.md'))} with your skill's instructions.
283
+ 2. Edit ${rel(path.join(dir, 'evals', 'evals.json'))} — replace the 3 example cases (rubrics anchored at 0.80).
284
+ 3. driftproof run ${rel(dir)} --models claude-haiku-4-5 --max-usd 2
285
+ See AUTHORING.md for how to write a fair suite.`);
286
+ }
287
+
288
+ // Emit a shields.io endpoint badge for a single receipt, or GitHub Actions
289
+ // output lines. Default: print the shields JSON to stdout (commit it and point
290
+ // the shields endpoint URL at it). --out FILE writes the JSON to a file.
291
+ // --github-output prints verdict/delta/message/color as key=value lines.
292
+ function cmdBadge(positional, flags) {
293
+ const p = positional[0];
294
+ if (!p) { console.error('usage: driftproof badge <receipt.json> [--out FILE] [--github-output]'); process.exit(2); }
295
+ const receipt = JSON.parse(fs.readFileSync(p, 'utf8'));
296
+ if (flags['github-output']) { console.log(githubOutputLines(receipt)); return; }
297
+ const badge = badgeEndpoint(receipt);
298
+ const json = JSON.stringify(badge, null, 2);
299
+ if (flags.out) {
300
+ const out = path.resolve(flags.out);
301
+ fs.mkdirSync(path.dirname(out), { recursive: true });
302
+ fs.writeFileSync(out, json + '\n');
303
+ const v = verdictFromReceipt(receipt);
304
+ console.log(`badge written to ${flags.out} (${v.verdict}: ${badge.message}, ${badge.color})`);
305
+ } else {
306
+ console.log(json);
307
+ }
308
+ }
309
+
310
+ async function main() {
311
+ const [, , cmd, ...rest] = process.argv;
312
+ const { positional, flags } = parseArgs(rest);
313
+ switch (cmd) {
314
+ case 'init': return cmdInit(positional);
315
+ case 'run': return cmdRun(positional, flags);
316
+ case 'diff': return cmdDiff(positional, flags);
317
+ case 'validate': return cmdValidate(positional);
318
+ case 'badge': return cmdBadge(positional, flags);
319
+ case 'version': case '--version': case '-v': console.log(RUNNER_VERSION); return;
320
+ case 'help': case '--help': case '-h': case undefined: return usage();
321
+ default: console.error(`unknown command: ${cmd}\n`); usage(); process.exit(2);
322
+ }
323
+ }
324
+
325
+ main().catch((e) => { console.error('FATAL', e && (e.stack || e.message || e)); process.exit(2); });
@@ -0,0 +1,19 @@
1
+ {
2
+ "_comment": "Driftproof model registry. The runner resolves --models ids against this list; an unknown id still runs but its receipt is marked registry:\"unregistered\" and its cost is estimated with the conservative default price in lib/models.js. Prices are STANDARD first-party USD per 1,000,000 tokens (input/output) from platform.claude.com (July 2026); Sonnet 5's introductory rate is deliberately NOT used so projections stay an upper bound. `released` is best-effort (spec RECEIPT.md open question #4): dates derived from a dated model id or Anthropic launch posts, else null. `tier` drives nothing but reporting; `judge_eligible` encodes the fixed-judge policy (see docs/judge-policy.html) — only the cheap Haiku judge is eligible. The release trigger (scripts/release-watch.js) appends auto-discovered ids here with auto_added:true.",
3
+ "registry_version": "1.0",
4
+ "provider": "anthropic",
5
+ "default_price_note": "Unregistered ids are costed at the most expensive known tier (Fable, $10/$50) — a budget guard must never under-estimate. See lib/models.js DEFAULT_PRICE.",
6
+ "models": [
7
+ { "id": "claude-fable-5", "family": "fable", "provider": "anthropic", "released": null, "input_price": 10.0, "output_price": 50.0, "tier": "frontier", "judge_eligible": false },
8
+ { "id": "claude-opus-5", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
9
+ { "id": "claude-opus-4-8", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
10
+ { "id": "claude-opus-4-7", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
11
+ { "id": "claude-opus-4-6", "family": "opus", "provider": "anthropic", "released": null, "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
12
+ { "id": "claude-opus-4-5-20251101", "family": "opus", "provider": "anthropic", "released": "2025-11-01", "input_price": 5.0, "output_price": 25.0, "tier": "frontier", "judge_eligible": false },
13
+ { "id": "claude-sonnet-5", "family": "sonnet", "provider": "anthropic", "released": "2026-06-30", "input_price": 3.0, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
14
+ { "id": "claude-sonnet-4-6", "family": "sonnet", "provider": "anthropic", "released": "2026-02-17", "input_price": 3.0, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
15
+ { "id": "claude-sonnet-4-5-20250929", "family": "sonnet", "provider": "anthropic", "released": "2025-09-29", "input_price": 3.0, "output_price": 15.0, "tier": "standard", "judge_eligible": false },
16
+ { "id": "claude-haiku-4-5", "family": "haiku", "provider": "anthropic", "released": "2025-10-01", "input_price": 1.0, "output_price": 5.0, "tier": "cheap", "judge_eligible": true },
17
+ { "id": "claude-haiku-4-5-20251001", "family": "haiku", "provider": "anthropic", "released": "2025-10-01", "input_price": 1.0, "output_price": 5.0, "tier": "cheap", "judge_eligible": true }
18
+ ]
19
+ }
package/config.js ADDED
@@ -0,0 +1,48 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // The project name lives in exactly ONE place so a future rename touches one
5
+ // constant. Everything user-facing (CLI banner, receipt runner id, file names)
6
+ // derives from PROJECT_NAME.
7
+ const PROJECT_NAME = 'driftproof';
8
+
9
+ // Bumped whenever the runner's behaviour or receipt-generation semantics change
10
+ // in a way that could affect results. Recorded into every receipt as
11
+ // run.runner_version so a receipt is reproducible against a known engine.
12
+ const RUNNER_VERSION = '0.3.0';
13
+
14
+ // The eval format we CONSUME (we deliberately do not invent our own).
15
+ const SUITE_FORMAT = 'agentskills.io/evals';
16
+
17
+ // Receipt schema version this runner emits. Loader/validator accept older
18
+ // versions too (see lib/receipt.js), but new receipts are stamped current.
19
+ // v0.3 adds transcript-auditability: per-case generation_hash + judge_sample_
20
+ // hashes[], plus run.registry and run.transcripts. Additive; v0.1/v0.2 still load.
21
+ const RECEIPT_SCHEMA_VERSION = '0.3';
22
+
23
+ // Hard USD budget defaults per entry point (Week 4). --max-usd overrides any of
24
+ // these. The projection is refused before any call if it exceeds the cap, on
25
+ // BOTH surfaces; a run also hard-stops mid-flight at 1.25× (see lib/cost.js).
26
+ // dev — an interactive `driftproof run`
27
+ // report — a full Report-#001-style suite run (scripts/run-report-001.js)
28
+ // trigger — a release-trigger-initiated prepare-report run
29
+ const DEV_MAX_USD = 2;
30
+ const REPORT_MAX_USD = 40;
31
+ const TRIGGER_MAX_USD = 25;
32
+
33
+ // Default number of judge samples per case in sampled mode. Sampling is what
34
+ // turns a single noisy grade into a mean ± band; --samples overrides it.
35
+ const DEFAULT_JUDGE_SAMPLES = 5;
36
+
37
+ // Minimum practical-effect floor for a drift verdict. Band separation alone is
38
+ // necessary but NOT sufficient to claim a regression/improvement: the mean must
39
+ // ALSO move by at least this much. Rationale — the Haiku judge quantizes scores
40
+ // to a coarse ~0.05–0.1 grid, so a confident grade often has stddev 0 (a "point
41
+ // band"). Two point bands that differ by a single quantum (e.g. 0.60 vs 0.64)
42
+ // are technically non-overlapping yet represent no meaningful behaviour change.
43
+ // The floor turns such statistically-separated-but-trivial moves into
44
+ // "within noise (below effect floor)". 0.05 = one judge quantum. Documented in
45
+ // spec/RECEIPT.md § "Drift verdict rule" and the report methodology.
46
+ const EFFECT_FLOOR = 0.05;
47
+
48
+ module.exports = { PROJECT_NAME, RUNNER_VERSION, SUITE_FORMAT, RECEIPT_SCHEMA_VERSION, DEFAULT_JUDGE_SAMPLES, EFFECT_FLOOR, DEV_MAX_USD, REPORT_MAX_USD, TRIGGER_MAX_USD };
@@ -0,0 +1,41 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const crypto = require('crypto');
5
+
6
+ // Deterministic JSON serialization: object keys sorted lexicographically at
7
+ // every level, arrays kept in order, no insignificant whitespace. Two
8
+ // structurally-equal values always produce the same string, which is what makes
9
+ // content/suite/receipt hashes stable and comparable across machines.
10
+ function canonicalize(value) {
11
+ if (value === null || typeof value !== 'object') return JSON.stringify(value);
12
+ if (Array.isArray(value)) return '[' + value.map(canonicalize).join(',') + ']';
13
+ const keys = Object.keys(value).filter((k) => value[k] !== undefined).sort();
14
+ return '{' + keys.map((k) => JSON.stringify(k) + ':' + canonicalize(value[k])).join(',') + '}';
15
+ }
16
+
17
+ function sha256(buf) {
18
+ return crypto.createHash('sha256').update(buf).digest('hex');
19
+ }
20
+
21
+ // sha256 over the canonical form of a JSON-able value.
22
+ function sha256Canonical(value) {
23
+ return sha256(canonicalize(value));
24
+ }
25
+
26
+ // Hash an ordered list of { path, bytes } file entries. The list is sorted by
27
+ // path so ordering on disk never changes the digest. Each entry contributes its
28
+ // path and a hash of its bytes, giving a single content hash over a file set.
29
+ function sha256Files(files) {
30
+ const sorted = [...files].sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
31
+ const h = crypto.createHash('sha256');
32
+ for (const f of sorted) {
33
+ h.update(f.path, 'utf8');
34
+ h.update('\0');
35
+ h.update(sha256(f.bytes));
36
+ h.update('\n');
37
+ }
38
+ return h.digest('hex');
39
+ }
40
+
41
+ module.exports = { canonicalize, sha256, sha256Canonical, sha256Files };
package/lib/cost.js ADDED
@@ -0,0 +1,118 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { priceForModel } = require('./models');
5
+
6
+ // Dollar cost estimation and live budget tracking for a run.
7
+ //
8
+ // Week 3 added a HARD USD budget on top of the Week-2 call-count cap. Week 4
9
+ // moves prices into the model registry (config/models.json, via lib/models.js)
10
+ // and makes the budget guard apply on EVERY entry point and BOTH surfaces:
11
+ // - the projection is printed and, if it exceeds --max-usd, the run is refused
12
+ // BEFORE any model call (on cli too — subscription attention is not free);
13
+ // - a BudgetTracker accumulates the estimated spend AS THE RUN PROCEEDS and
14
+ // hard-stops at 1.25× the cap.
15
+ //
16
+ // IMPORTANT (honesty): the numbers below are a DELIBERATELY ROUGH upper bound.
17
+ // Token counts per call are estimated from a small fixed table (not measured
18
+ // with count_tokens), and prices are the standard first-party per-MTok rates
19
+ // from the registry. The estimate keeps a run bounded and states the order of
20
+ // magnitude — it is not an invoice. On the `claude-cli` surface the ACTUAL
21
+ // metered spend is $0 (subscription); the dollar figure is the hypothetical
22
+ // "if this had run on the metered API" cost, and it is counted against the caps
23
+ // identically so subscription usage is never treated as free.
24
+
25
+ // Rough per-call token estimates. Chosen to over- rather than under-estimate.
26
+ // - a generation call with the skill prepended carries the SKILL.md as a
27
+ // system prompt (our skills run ~0.4k–8k tokens; 3000 is a generous mean),
28
+ // - a baseline generation carries only the short case prompt,
29
+ // - a judge call carries task + response + rubric + judge system, and emits a
30
+ // short JSON grade.
31
+ const TOKENS = {
32
+ gen_with_skill: { input: 3000, output: 900 },
33
+ gen_baseline: { input: 250, output: 900 },
34
+ judge: { input: 1300, output: 200 },
35
+ };
36
+
37
+ // Price per MTok for a model, from the registry (conservative default for an
38
+ // unregistered id — see lib/models.js).
39
+ function priceFor(modelId) {
40
+ const p = priceForModel(modelId);
41
+ return { input: p.input, output: p.output };
42
+ }
43
+
44
+ function callCostUSD(price, tokens) {
45
+ return (tokens.input / 1e6) * price.input + (tokens.output / 1e6) * price.output;
46
+ }
47
+
48
+ // Estimated USD cost of ONE call of a given kind on a given model. Used by the
49
+ // live BudgetTracker so the running spend estimate is built from the same
50
+ // per-call pieces as the up-front projection.
51
+ function perCallCostUSD(modelId, kind) {
52
+ const tokens = TOKENS[kind];
53
+ if (!tokens) throw new Error(`unknown call kind: ${kind}`);
54
+ return callCostUSD(priceFor(modelId), tokens);
55
+ }
56
+
57
+ // Estimate the metered USD cost of a full run: `caseCount` cases × 2 modes,
58
+ // generated on each target model, judged `samples` times each on `judgeModel`.
59
+ //
60
+ // returns { totalUSD, perModel: [{ model, usd }], judgeUSD, genUSD, assumptions }
61
+ function estimateRunCostUSD({ caseCount, samples, models, judgeModel }) {
62
+ const judgePrice = priceFor(judgeModel);
63
+ const perModel = [];
64
+ let judgeUSD = 0;
65
+ let genUSD = 0;
66
+
67
+ for (const model of models) {
68
+ const price = priceFor(model);
69
+ // Two generations per case (with_skill + baseline).
70
+ const gen = caseCount * (callCostUSD(price, TOKENS.gen_with_skill) + callCostUSD(price, TOKENS.gen_baseline));
71
+ // Judge every generation `samples` times → caseCount × 2 modes × samples.
72
+ const judge = caseCount * 2 * samples * callCostUSD(judgePrice, TOKENS.judge);
73
+ perModel.push({ model, usd: round4(gen + judge) });
74
+ genUSD += gen;
75
+ judgeUSD += judge;
76
+ }
77
+
78
+ return {
79
+ totalUSD: round4(genUSD + judgeUSD),
80
+ perModel,
81
+ judgeUSD: round4(judgeUSD),
82
+ genUSD: round4(genUSD),
83
+ assumptions: { tokens: TOKENS, judgeModel, note: 'rough upper-bound estimate; registry per-MTok pricing; not measured with count_tokens' },
84
+ };
85
+ }
86
+
87
+ // Live budget tracker. The runner adds the estimated cost of each call as it is
88
+ // made; if the accumulated estimate crosses 1.25× the cap the tracker throws a
89
+ // BUDGET_HARDSTOP, aborting the run mid-flight. The 1.25× headroom exists because
90
+ // the up-front projection already guards the cap; the hard-stop is a backstop
91
+ // against a projection that turned out low (e.g. an unregistered model, longer
92
+ // outputs). On both surfaces the estimate is counted identically.
93
+ class BudgetTracker {
94
+ constructor(capUsd) {
95
+ this.capUsd = capUsd;
96
+ this.hardCap = round4(capUsd * 1.25);
97
+ this.spent = 0;
98
+ }
99
+
100
+ // Add estimated USD; throws BUDGET_HARDSTOP if the running total crosses 1.25×.
101
+ add(usd) {
102
+ this.spent = round4(this.spent + usd);
103
+ if (this.spent > this.hardCap) {
104
+ const e = new Error(
105
+ `budget hard-stop: estimated spend $${this.spent.toFixed(4)} exceeded 1.25× cap `
106
+ + `($${this.hardCap.toFixed(4)}; --max-usd $${this.capUsd.toFixed(2)})`);
107
+ e.code = 'BUDGET_HARDSTOP';
108
+ throw e;
109
+ }
110
+ return this.spent;
111
+ }
112
+
113
+ remaining() { return round4(this.hardCap - this.spent); }
114
+ }
115
+
116
+ function round4(n) { return Math.round(n * 1e4) / 1e4; }
117
+
118
+ module.exports = { estimateRunCostUSD, perCallCostUSD, priceFor, BudgetTracker, TOKENS };