driftproof 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +274 -0
- package/bin/driftproof +325 -0
- package/config/models.json +19 -0
- package/config.js +48 -0
- package/lib/canonical.js +41 -0
- package/lib/cost.js +118 -0
- package/lib/diff.js +152 -0
- package/lib/init.js +128 -0
- package/lib/json.js +141 -0
- package/lib/judge.js +126 -0
- package/lib/models.js +125 -0
- package/lib/provider.js +118 -0
- package/lib/receipt.js +133 -0
- package/lib/run.js +222 -0
- package/lib/runner.js +100 -0
- package/lib/skill.js +117 -0
- package/lib/stats.js +66 -0
- package/lib/stub.js +54 -0
- package/lib/verdict.js +76 -0
- package/package.json +45 -0
- package/spec/RECEIPT.md +246 -0
- package/spec/receipt.schema.json +211 -0
- package/spec/receipt.v0.1.schema.json +138 -0
- package/spec/receipt.v0.2.schema.json +190 -0
package/lib/provider.js
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const { spawn } = require('child_process');
|
|
5
|
+
const { withRetry, withTimeout } = require('./json');
|
|
6
|
+
const { stubComplete, stubEnabled } = require('./stub');
|
|
7
|
+
|
|
8
|
+
// Provider abstraction: one `complete()` call, two surfaces.
|
|
9
|
+
//
|
|
10
|
+
// CLAUDE_PROVIDER=api → Anthropic Messages API (needs ANTHROPIC_API_KEY).
|
|
11
|
+
// CLAUDE_PROVIDER=cli → spawn `claude -p` with ANTHROPIC_API_KEY STRIPPED
|
|
12
|
+
// from the child env, so dev runs draw on the local
|
|
13
|
+
// subscription session credit rather than metered API
|
|
14
|
+
// billing. This is the DEFAULT for dev.
|
|
15
|
+
//
|
|
16
|
+
// The surface actually used is returned so the runner can stamp it into the
|
|
17
|
+
// receipt (run.surface = "api" | "claude-cli").
|
|
18
|
+
|
|
19
|
+
// Short model aliases → canonical ids. Kept tiny and explicit; unknown values
|
|
20
|
+
// are passed through verbatim so a full model id always works.
|
|
21
|
+
const MODEL_ALIASES = {
|
|
22
|
+
haiku: 'claude-haiku-4-5-20251001',
|
|
23
|
+
sonnet: 'claude-sonnet-5', // current Sonnet
|
|
24
|
+
'sonnet-5': 'claude-sonnet-5',
|
|
25
|
+
'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
|
|
26
|
+
opus: 'claude-opus-4-8',
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
function resolveModel(m) {
|
|
30
|
+
return MODEL_ALIASES[m] || m;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function providerName() {
|
|
34
|
+
return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// The receipt-facing label for the surface in use.
|
|
38
|
+
function surfaceLabel() {
|
|
39
|
+
return providerName() === 'api' ? 'api' : 'claude-cli';
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Send a single-turn prompt and return { text, usage, surface }.
|
|
43
|
+
// usage is { input_tokens, output_tokens } when the surface reports it (api),
|
|
44
|
+
// else null (cli does not expose token counts to us).
|
|
45
|
+
// `temperature` is honoured ONLY on the api surface (the Messages API exposes
|
|
46
|
+
// it). On the cli surface sampling params are surface-controlled — we cannot set
|
|
47
|
+
// them, so temperature is ignored there and the receipt records that fact.
|
|
48
|
+
async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = 120000, temperature = undefined }) {
|
|
49
|
+
const surface = surfaceLabel();
|
|
50
|
+
// Offline stub surface: return a canned completion with zero model calls. The
|
|
51
|
+
// receipt still records the real surface label so a stub run is not mistaken
|
|
52
|
+
// for a genuine one at read time — only run.judge/generation TEXT is canned.
|
|
53
|
+
if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface };
|
|
54
|
+
const runner = () => (surface === 'api'
|
|
55
|
+
? completeApi({ system, prompt, model, maxTokens, temperature })
|
|
56
|
+
: completeCli({ system, prompt, model, timeoutMs }));
|
|
57
|
+
// A published run makes ~1000s of calls; be patient with transient
|
|
58
|
+
// throttling/cold-starts so one blip doesn't abort a multi-hour grind.
|
|
59
|
+
const out = await withRetry(() => withTimeout(runner, timeoutMs, `provider(${surface})`), { tries: 4, baseDelayMs: 3000 });
|
|
60
|
+
return { ...out, surface };
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
async function completeApi({ system, prompt, model, maxTokens, temperature }) {
|
|
64
|
+
let Anthropic;
|
|
65
|
+
try { Anthropic = require('@anthropic-ai/sdk'); }
|
|
66
|
+
catch (_e) { throw new Error('CLAUDE_PROVIDER=api requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
|
|
67
|
+
if (!process.env.ANTHROPIC_API_KEY) throw new Error('CLAUDE_PROVIDER=api requires ANTHROPIC_API_KEY');
|
|
68
|
+
const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
|
|
69
|
+
const params = {
|
|
70
|
+
model: resolveModel(model),
|
|
71
|
+
max_tokens: maxTokens,
|
|
72
|
+
messages: [{ role: 'user', content: prompt }],
|
|
73
|
+
};
|
|
74
|
+
if (system) params.system = system;
|
|
75
|
+
if (temperature !== undefined) params.temperature = temperature;
|
|
76
|
+
const resp = await client.messages.create(params);
|
|
77
|
+
const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
|
|
78
|
+
return {
|
|
79
|
+
text,
|
|
80
|
+
usage: {
|
|
81
|
+
input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
|
|
82
|
+
output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
|
|
83
|
+
},
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
function completeCli({ system, prompt, model, timeoutMs }) {
|
|
88
|
+
return new Promise((resolve, reject) => {
|
|
89
|
+
// Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
|
|
90
|
+
// metered API key. Everything else in the env is preserved.
|
|
91
|
+
const env = { ...process.env };
|
|
92
|
+
delete env.ANTHROPIC_API_KEY;
|
|
93
|
+
|
|
94
|
+
const args = ['-p', '--model', resolveModel(model)];
|
|
95
|
+
if (system) args.push('--append-system-prompt', system);
|
|
96
|
+
|
|
97
|
+
const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
98
|
+
let out = '';
|
|
99
|
+
let err = '';
|
|
100
|
+
const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
|
|
101
|
+
child.stdout.on('data', (d) => { out += d; });
|
|
102
|
+
child.stderr.on('data', (d) => { err += d; });
|
|
103
|
+
child.on('error', (e) => {
|
|
104
|
+
clearTimeout(killer);
|
|
105
|
+
if (e.code === 'ENOENT') reject(new Error("CLAUDE_PROVIDER=cli requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
|
|
106
|
+
else reject(e);
|
|
107
|
+
});
|
|
108
|
+
child.on('close', (code) => {
|
|
109
|
+
clearTimeout(killer);
|
|
110
|
+
if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
|
|
111
|
+
resolve({ text: out.trim(), usage: null });
|
|
112
|
+
});
|
|
113
|
+
child.stdin.write(prompt);
|
|
114
|
+
child.stdin.end();
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
module.exports = { complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES };
|
package/lib/receipt.js
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const path = require('path');
|
|
6
|
+
const { canonicalize, sha256 } = require('./canonical');
|
|
7
|
+
const { RECEIPT_SCHEMA_VERSION } = require('../config');
|
|
8
|
+
const { aggregateBands, combineUncertainty, round } = require('./stats');
|
|
9
|
+
|
|
10
|
+
// Schema file per receipt version. The current schema is receipt.schema.json;
|
|
11
|
+
// older versions live alongside it so v0.1 receipts still validate. The
|
|
12
|
+
// validator picks the schema by the receipt's own schema_version field.
|
|
13
|
+
const SCHEMA_FILES = {
|
|
14
|
+
'0.1': 'receipt.v0.1.schema.json',
|
|
15
|
+
'0.2': 'receipt.v0.2.schema.json',
|
|
16
|
+
'0.3': 'receipt.schema.json',
|
|
17
|
+
};
|
|
18
|
+
|
|
19
|
+
const _validators = {};
|
|
20
|
+
// Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
|
|
21
|
+
// so the library can be required without ajv present (pure hashing utilities).
|
|
22
|
+
function getValidator(version) {
|
|
23
|
+
const v = SCHEMA_FILES[version] ? version : RECEIPT_SCHEMA_VERSION;
|
|
24
|
+
if (_validators[v]) return _validators[v];
|
|
25
|
+
let Ajv;
|
|
26
|
+
// The schema is JSON Schema draft 2020-12, so use ajv's 2020 build.
|
|
27
|
+
try { Ajv = require('ajv/dist/2020'); }
|
|
28
|
+
catch (_e) { throw new Error('receipt validation requires the `ajv` package (npm install)'); }
|
|
29
|
+
const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[v]), 'utf8'));
|
|
30
|
+
const ajv = new Ajv({ allErrors: true, strict: false });
|
|
31
|
+
_validators[v] = ajv.compile(schema);
|
|
32
|
+
return _validators[v];
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// Compute the receipt_hash: sha256 over the canonical receipt JSON with the
|
|
36
|
+
// receipt_hash field itself removed. Deterministic and reproducible.
|
|
37
|
+
function computeReceiptHash(receipt) {
|
|
38
|
+
const { receipt_hash, ...rest } = receipt; // eslint-disable-line no-unused-vars
|
|
39
|
+
return sha256(canonicalize(rest));
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Stamp (or re-stamp) the receipt_hash in place and return the receipt.
|
|
43
|
+
function sealReceipt(receipt) {
|
|
44
|
+
receipt.receipt_hash = computeReceiptHash(receipt);
|
|
45
|
+
return receipt;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// Verify a receipt's self-hash matches its contents (tamper-evidence lite).
|
|
49
|
+
function verifyReceiptHash(receipt) {
|
|
50
|
+
return receipt.receipt_hash === computeReceiptHash(receipt);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// Validate against the schema matching the receipt's own schema_version (so both
|
|
54
|
+
// v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
|
|
55
|
+
function validateReceipt(receipt) {
|
|
56
|
+
const version = (receipt && receipt.schema_version) || RECEIPT_SCHEMA_VERSION;
|
|
57
|
+
const validate = getValidator(version);
|
|
58
|
+
const valid = validate(receipt);
|
|
59
|
+
return { valid, errors: valid ? [] : (validate.errors || []), version };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function mean(nums) {
|
|
63
|
+
if (!nums.length) return 0;
|
|
64
|
+
return round(nums.reduce((a, b) => a + b, 0) / nums.length);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Aggregate one mode's cases: mean of case means, band (suite dispersion — the
|
|
68
|
+
// stddev of the per-case means; see lib/stats.aggregateBands), and pass count.
|
|
69
|
+
function aggregate(caseResults) {
|
|
70
|
+
const band = aggregateBands(caseResults.map((r) => ({ mean: r.mean != null ? r.mean : r.score, stddev: r.stddev || 0, n: Array.isArray(r.samples) ? r.samples.length : 1 })));
|
|
71
|
+
const passes = caseResults.filter((r) => r.outcome === 'pass').length;
|
|
72
|
+
const borderline = caseResults.filter((r) => r.outcome === 'borderline').length;
|
|
73
|
+
return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// Assemble a full receipt from the runner's raw pieces, seal it, and return it.
|
|
77
|
+
// skill: { name, version, contentHash }
|
|
78
|
+
// suite: { format, suiteHash, caseCount }
|
|
79
|
+
// run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
|
|
80
|
+
// cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
|
|
81
|
+
// editorialReviews: optional [ { url, source, date } ]
|
|
82
|
+
function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
|
|
83
|
+
const withSkill = cases.filter((c) => c.mode === 'with_skill');
|
|
84
|
+
const baseline = cases.filter((c) => c.mode === 'baseline');
|
|
85
|
+
const aggWith = aggregate(withSkill);
|
|
86
|
+
const aggBase = aggregate(baseline);
|
|
87
|
+
|
|
88
|
+
const receipt = {
|
|
89
|
+
schema_version: RECEIPT_SCHEMA_VERSION,
|
|
90
|
+
skill: {
|
|
91
|
+
name: skill.name,
|
|
92
|
+
version: skill.version,
|
|
93
|
+
content_hash: skill.contentHash,
|
|
94
|
+
},
|
|
95
|
+
suite: {
|
|
96
|
+
format: suite.format,
|
|
97
|
+
suite_hash: suite.suiteHash,
|
|
98
|
+
case_count: suite.caseCount,
|
|
99
|
+
},
|
|
100
|
+
run: {
|
|
101
|
+
model_id: run.model_id,
|
|
102
|
+
model_release_date: run.model_release_date == null ? null : run.model_release_date,
|
|
103
|
+
surface: run.surface,
|
|
104
|
+
runner_version: run.runner_version,
|
|
105
|
+
date_utc: run.date_utc,
|
|
106
|
+
// v0.3: registry provenance + transcript-retention mode. Defaults keep the
|
|
107
|
+
// honest, cheapest interpretation when a caller omits them.
|
|
108
|
+
registry: run.registry || 'unregistered',
|
|
109
|
+
transcripts: run.transcripts || 'hashes-only',
|
|
110
|
+
judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
|
|
111
|
+
},
|
|
112
|
+
results: {
|
|
113
|
+
cases,
|
|
114
|
+
aggregates: { with_skill: aggWith, baseline: aggBase },
|
|
115
|
+
},
|
|
116
|
+
comparison: {
|
|
117
|
+
with_skill_score: aggWith.mean_score,
|
|
118
|
+
baseline_score: aggBase.mean_score,
|
|
119
|
+
delta: round(aggWith.mean_score - aggBase.mean_score),
|
|
120
|
+
// Combined uncertainty of the delta: quadrature sum of the two aggregate
|
|
121
|
+
// bands. Diff uses this for the headline "within noise" vs real-move rule.
|
|
122
|
+
delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
|
|
123
|
+
},
|
|
124
|
+
verification_level: verificationLevel,
|
|
125
|
+
receipt_hash: '',
|
|
126
|
+
};
|
|
127
|
+
if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
|
|
128
|
+
return sealReceipt(receipt);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
module.exports = {
|
|
132
|
+
buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
|
|
133
|
+
};
|
package/lib/run.js
ADDED
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
const { complete, resolveModel, surfaceLabel } = require('./provider');
|
|
5
|
+
const { gradeSamples, judgeSettings } = require('./judge');
|
|
6
|
+
const { buildReceipt } = require('./receipt');
|
|
7
|
+
const { sha256 } = require('./canonical');
|
|
8
|
+
const { registryStatus } = require('./models');
|
|
9
|
+
const { perCallCostUSD } = require('./cost');
|
|
10
|
+
const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
|
|
11
|
+
|
|
12
|
+
// Known model release dates (best-effort; null when unknown). Recorded into the
|
|
13
|
+
// receipt so drift reports can order runs by model age. Dateless model ids
|
|
14
|
+
// (the 4.6 generation onward) carry no date, so the announcement date is
|
|
15
|
+
// recorded explicitly here from Anthropic's public launch posts (provenance
|
|
16
|
+
// noted in the report; consistent with spec open question #4 — dates are
|
|
17
|
+
// best-effort, not verified against the Models API in this run).
|
|
18
|
+
const MODEL_RELEASE_DATES = {
|
|
19
|
+
'claude-haiku-4-5-20251001': '2025-10-01',
|
|
20
|
+
'claude-haiku-4-5': '2025-10-01',
|
|
21
|
+
'claude-sonnet-5': '2026-06-30', // anthropic.com/news/claude-sonnet-5
|
|
22
|
+
'claude-sonnet-4-6': '2026-02-17', // anthropic.com/news/claude-sonnet-4-6
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
function releaseDateFor(modelId) {
|
|
26
|
+
if (MODEL_RELEASE_DATES[modelId]) return MODEL_RELEASE_DATES[modelId];
|
|
27
|
+
// Derive from a trailing YYYYMMDD in the id if present.
|
|
28
|
+
const m = String(modelId).match(/(\d{4})(\d{2})(\d{2})$/);
|
|
29
|
+
return m ? `${m[1]}-${m[2]}-${m[3]}` : null;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Calls for one model run: each case does 2 generations (with_skill + baseline)
|
|
33
|
+
// and 2×samples judge calls (each generation judged `samples` times).
|
|
34
|
+
function projectCalls(caseCount, samples) {
|
|
35
|
+
return caseCount * (2 + 2 * samples);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// Ask the target model to perform one eval case. `withSkill` decides whether the
|
|
39
|
+
// SKILL.md is prepended as a system prompt (the whole point: measure the skill's
|
|
40
|
+
// marginal effect vs a bare baseline).
|
|
41
|
+
async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
|
|
42
|
+
const system = withSkill ? skillMd : undefined;
|
|
43
|
+
const { text, usage } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
|
|
44
|
+
return { text, usage };
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// Determine a case outcome from its sampled band and threshold.
|
|
48
|
+
// borderline : threshold lies within [mean - stddev, mean + stddev]
|
|
49
|
+
// pass/fail : mean clears / misses the threshold with the band clear of it
|
|
50
|
+
// score : un-thresholded case (report the number, no pass/fail)
|
|
51
|
+
function outcomeFor(mean, stddev, threshold) {
|
|
52
|
+
if (typeof threshold !== 'number') return 'score';
|
|
53
|
+
if (mean - stddev <= threshold && threshold <= mean + stddev) return 'borderline';
|
|
54
|
+
return mean >= threshold ? 'pass' : 'fail';
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// Judge one generated response `samples` times → sampled case result.
|
|
58
|
+
// `generationHash` binds the graded case to the exact generation text (v0.3).
|
|
59
|
+
// Returns { caseResult, sampleTexts } — sampleTexts is transient (retained only
|
|
60
|
+
// under --keep-transcripts; never part of the receipt).
|
|
61
|
+
async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples }) {
|
|
62
|
+
const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs });
|
|
63
|
+
const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
|
|
64
|
+
const caseResult = {
|
|
65
|
+
id: caseObj.id,
|
|
66
|
+
mode,
|
|
67
|
+
outcome,
|
|
68
|
+
score: g.mean, // `score` == sampled mean (v0.1 readers still work)
|
|
69
|
+
mean: g.mean,
|
|
70
|
+
stddev: g.stddev,
|
|
71
|
+
samples: g.samples,
|
|
72
|
+
// v0.3 transcript auditability: hash of the graded generation + one hash per
|
|
73
|
+
// judge sample. These make a receipt checkable against retained transcripts.
|
|
74
|
+
generation_hash: generationHash,
|
|
75
|
+
judge_sample_hashes: g.sample_hashes,
|
|
76
|
+
threshold: typeof caseObj.pass_threshold === 'number' ? caseObj.pass_threshold : null,
|
|
77
|
+
reason: g.reason,
|
|
78
|
+
judge: { model_id: g.model_id, rubric_hash: g.rubric_hash },
|
|
79
|
+
};
|
|
80
|
+
return { caseResult, sampleTexts: g.sample_texts };
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
// Run up to `concurrency` async tasks at a time, preserving input order in the
|
|
84
|
+
// results. Keeps the CLI grind tractable (each `claude -p` cold-start dominates
|
|
85
|
+
// wall-clock, so a handful of concurrent spawns is a large speedup) without
|
|
86
|
+
// unbounded fan-out.
|
|
87
|
+
async function mapPool(items, concurrency, fn) {
|
|
88
|
+
const results = new Array(items.length);
|
|
89
|
+
let next = 0;
|
|
90
|
+
const workers = new Array(Math.max(1, Math.min(concurrency, items.length))).fill(0).map(async () => {
|
|
91
|
+
while (true) {
|
|
92
|
+
const i = next++;
|
|
93
|
+
if (i >= items.length) return;
|
|
94
|
+
results[i] = await fn(items[i], i);
|
|
95
|
+
}
|
|
96
|
+
});
|
|
97
|
+
await Promise.all(workers);
|
|
98
|
+
return results;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// Run the full suite for ONE model, both modes, sampled judge-grading, and build
|
|
102
|
+
// a sealed receipt. Enforces a hard call cap; every model+judge call counts.
|
|
103
|
+
//
|
|
104
|
+
// opts: { maxCases, maxCalls, samples, judgeModel, timeoutMs, concurrency,
|
|
105
|
+
// onProgress, budget, keepTranscripts, nowIso }
|
|
106
|
+
// budget — optional BudgetTracker; accumulates estimated per-call
|
|
107
|
+
// USD as the run proceeds and hard-stops at 1.25× the cap.
|
|
108
|
+
// keepTranscripts — when true, the run records transcripts:"retained-local"
|
|
109
|
+
// and returns the raw generations + judge outputs so the
|
|
110
|
+
// caller can write them to transcripts/<receipt-id>/.
|
|
111
|
+
async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
112
|
+
const modelId = resolveModel(model);
|
|
113
|
+
const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
|
|
114
|
+
const timeoutMs = opts.timeoutMs || 120000;
|
|
115
|
+
const maxCalls = opts.maxCalls || 200;
|
|
116
|
+
const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
|
|
117
|
+
const concurrency = Math.max(1, opts.concurrency || 1);
|
|
118
|
+
const onProgress = opts.onProgress || (() => {});
|
|
119
|
+
const budget = opts.budget || null;
|
|
120
|
+
const keepTranscripts = !!opts.keepTranscripts;
|
|
121
|
+
|
|
122
|
+
let cases = skill.suite.cases;
|
|
123
|
+
if (opts.maxCases && cases.length > opts.maxCases) cases = cases.slice(0, opts.maxCases);
|
|
124
|
+
|
|
125
|
+
// Cost guard: project the whole run up front and refuse before spending a
|
|
126
|
+
// single call if it would blow the cap.
|
|
127
|
+
const projected = projectCalls(cases.length, samples);
|
|
128
|
+
if (projected > maxCalls) {
|
|
129
|
+
const e = new Error(`cost guard: projected ${projected} calls exceeds cap ${maxCalls} (${cases.length} cases × (2 + 2×${samples} samples)). Raise --max-calls or lower --max-cases/--samples.`);
|
|
130
|
+
e.code = 'CALL_CAP';
|
|
131
|
+
throw e;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// One task per (case, mode). Order is preserved in the receipt regardless of
|
|
135
|
+
// completion order, so receipts are deterministic under concurrency.
|
|
136
|
+
const tasks = [];
|
|
137
|
+
for (const c of cases) for (const withSkill of [true, false]) tasks.push({ c, withSkill });
|
|
138
|
+
|
|
139
|
+
let calls = 0;
|
|
140
|
+
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
141
|
+
const mode = withSkill ? 'with_skill' : 'baseline';
|
|
142
|
+
onProgress({ case: c.id, mode, phase: 'generate' });
|
|
143
|
+
const { text } = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs });
|
|
144
|
+
calls += 1;
|
|
145
|
+
// Live budget: count the generation call, then hard-stop if over 1.25× cap.
|
|
146
|
+
if (budget) budget.add(perCallCostUSD(modelId, withSkill ? 'gen_with_skill' : 'gen_baseline'));
|
|
147
|
+
const generationHash = sha256(String(text || ''));
|
|
148
|
+
onProgress({ case: c.id, mode, phase: 'judge', samples });
|
|
149
|
+
const { caseResult, sampleTexts } = await judgeCase({ caseObj: c, response: text, generationHash, judgeModel, mode, timeoutMs, samples });
|
|
150
|
+
calls += samples;
|
|
151
|
+
// Live budget: count all `samples` judge calls for this (case, mode).
|
|
152
|
+
if (budget) budget.add(samples * perCallCostUSD(judgeModel, 'judge'));
|
|
153
|
+
onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
|
|
154
|
+
const transcript = keepTranscripts
|
|
155
|
+
? { id: c.id, mode, generation: String(text || ''), judge_outputs: sampleTexts }
|
|
156
|
+
: null;
|
|
157
|
+
return { caseResult, transcript };
|
|
158
|
+
});
|
|
159
|
+
const caseResults = pairs.map((p) => p.caseResult);
|
|
160
|
+
const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
|
|
161
|
+
|
|
162
|
+
const receipt = buildReceipt({
|
|
163
|
+
skill: { name: skill.name, version: skill.version, contentHash: skill.contentHash },
|
|
164
|
+
suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
|
|
165
|
+
run: {
|
|
166
|
+
model_id: modelId,
|
|
167
|
+
model_release_date: releaseDateFor(modelId),
|
|
168
|
+
surface: surfaceLabel(),
|
|
169
|
+
runner_version: RUNNER_VERSION,
|
|
170
|
+
date_utc: opts.nowIso || new Date().toISOString(),
|
|
171
|
+
registry: registryStatus(modelId),
|
|
172
|
+
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
173
|
+
judge: judgeSettings(samples),
|
|
174
|
+
},
|
|
175
|
+
cases: caseResults,
|
|
176
|
+
verificationLevel: 'TESTED',
|
|
177
|
+
});
|
|
178
|
+
|
|
179
|
+
return { receipt, calls, transcripts };
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
|
|
183
|
+
|
|
184
|
+
// Render a short human-readable markdown summary of a receipt.
|
|
185
|
+
function summarizeReceipt(receipt) {
|
|
186
|
+
const L = [];
|
|
187
|
+
L.push(`# ${receipt.skill.name} — receipt summary`);
|
|
188
|
+
L.push('');
|
|
189
|
+
L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
|
|
190
|
+
L.push(`- **surface:** ${receipt.run.surface}`);
|
|
191
|
+
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
192
|
+
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
193
|
+
const j = receipt.run.judge || {};
|
|
194
|
+
L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a (surface-controlled)' : j.temperature} (${j.sampling || 'single'})`);
|
|
195
|
+
if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
|
|
196
|
+
L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
|
|
197
|
+
L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
|
|
198
|
+
L.push(`- **verification:** ${receipt.verification_level}`);
|
|
199
|
+
L.push(`- **receipt_hash:** \`${receipt.receipt_hash.slice(0, 16)}…\``);
|
|
200
|
+
L.push('');
|
|
201
|
+
L.push(`## Headline`);
|
|
202
|
+
L.push('');
|
|
203
|
+
const cmp = receipt.comparison;
|
|
204
|
+
const aggs = receipt.results.aggregates;
|
|
205
|
+
const sign = cmp.delta >= 0 ? '+' : '';
|
|
206
|
+
L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
|
|
207
|
+
L.push('');
|
|
208
|
+
L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
|
|
209
|
+
L.push('');
|
|
210
|
+
L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
|
|
211
|
+
L.push('');
|
|
212
|
+
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
213
|
+
L.push(`|---|---|---|---|---|`);
|
|
214
|
+
for (const c of receipt.results.cases) {
|
|
215
|
+
const flag = c.outcome === 'borderline' ? ' ⚠' : '';
|
|
216
|
+
L.push(`| \`${c.id}\` | ${c.mode} | ${c.outcome}${flag} | ${band(c.mean, c.stddev || 0)} | ${c.reason || ''} |`);
|
|
217
|
+
}
|
|
218
|
+
L.push('');
|
|
219
|
+
return L.join('\n');
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor };
|
package/lib/runner.js
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// The run / compare / report-delta / exit-nonzero engine.
|
|
5
|
+
//
|
|
6
|
+
// PORTED (then fully genericized) from a private behavioral-gate spec. The
|
|
7
|
+
// original drove a browser against a product and asserted product-specific
|
|
8
|
+
// journeys; ALL of that (endpoints, tenant probes, chrome tokens, journeys) was
|
|
9
|
+
// stripped. What remains is the reusable core:
|
|
10
|
+
// - a Gate collects named checks grouped by "section"
|
|
11
|
+
// - each check records { section, name, pass, detail }
|
|
12
|
+
// - summarize() prints a PASS/FAIL line per check and a headline count
|
|
13
|
+
// - toExitCode() gives 0 on all-pass, 1 on any failure
|
|
14
|
+
// Scanned clean against the confidentiality deny-list.
|
|
15
|
+
|
|
16
|
+
// A single collector for one gate run.
|
|
17
|
+
class Gate {
|
|
18
|
+
constructor(title = 'gate') {
|
|
19
|
+
this.title = title;
|
|
20
|
+
this.results = [];
|
|
21
|
+
this._section = '?';
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// Start a new named section; subsequent checks are grouped under it.
|
|
25
|
+
section(name) {
|
|
26
|
+
this._section = name;
|
|
27
|
+
return this;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// Record an assertion. `pass` is coerced to boolean; `detail` is any JSON-able
|
|
31
|
+
// context printed only when the check fails.
|
|
32
|
+
check(name, pass, detail = null) {
|
|
33
|
+
const rec = { section: this._section, name, pass: !!pass, detail: pass ? null : detail };
|
|
34
|
+
this.results.push(rec);
|
|
35
|
+
const tag = rec.pass ? 'PASS' : 'FAIL';
|
|
36
|
+
const suffix = rec.pass ? '' : ` -- ${safeJson(detail)}`;
|
|
37
|
+
// eslint-disable-next-line no-console
|
|
38
|
+
console.log(` [${tag}] ${name}${suffix}`);
|
|
39
|
+
return rec.pass;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Convenience: assert deep equality of two JSON-able values.
|
|
43
|
+
checkEqual(name, actual, expected) {
|
|
44
|
+
const pass = safeJson(actual) === safeJson(expected);
|
|
45
|
+
return this.check(name, pass, { actual, expected });
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
get passed() { return this.results.filter((r) => r.pass).length; }
|
|
49
|
+
get failed() { return this.results.filter((r) => !r.pass); }
|
|
50
|
+
get total() { return this.results.length; }
|
|
51
|
+
|
|
52
|
+
// Print the headline and per-failure detail. Returns the summary object.
|
|
53
|
+
summarize() {
|
|
54
|
+
const failed = this.failed;
|
|
55
|
+
// eslint-disable-next-line no-console
|
|
56
|
+
console.log(`\n=== ${this.title.toUpperCase()} RESULT: ${this.passed}/${this.total} passed, ${failed.length} failed ===`);
|
|
57
|
+
for (const f of failed) {
|
|
58
|
+
// eslint-disable-next-line no-console
|
|
59
|
+
console.log(` FAIL [${f.section}] ${f.name}: ${safeJson(f.detail)}`);
|
|
60
|
+
}
|
|
61
|
+
return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, results: this.results };
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
toExitCode() {
|
|
65
|
+
return this.failed.length === 0 ? 0 : 1;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// Compute the per-item delta between a "before" and "after" map of numeric
|
|
70
|
+
// scores, keyed by id. Returns entries with { id, before, after, delta,
|
|
71
|
+
// regressed } sorted worst-regression-first. This is the generic report-delta
|
|
72
|
+
// primitive the drift report builds on.
|
|
73
|
+
function reportDelta(beforeById, afterById, { regressionEpsilon = 0.0 } = {}) {
|
|
74
|
+
const ids = new Set([...Object.keys(beforeById || {}), ...Object.keys(afterById || {})]);
|
|
75
|
+
const rows = [];
|
|
76
|
+
for (const id of ids) {
|
|
77
|
+
const before = num(beforeById && beforeById[id]);
|
|
78
|
+
const after = num(afterById && afterById[id]);
|
|
79
|
+
const delta = (after == null || before == null) ? null : round(after - before);
|
|
80
|
+
const regressed = delta != null && delta < -regressionEpsilon;
|
|
81
|
+
rows.push({ id, before, after, delta, regressed });
|
|
82
|
+
}
|
|
83
|
+
rows.sort((a, b) => {
|
|
84
|
+
// Regressions first, then by magnitude of drop.
|
|
85
|
+
const da = a.delta == null ? 0 : a.delta;
|
|
86
|
+
const db = b.delta == null ? 0 : b.delta;
|
|
87
|
+
return da - db;
|
|
88
|
+
});
|
|
89
|
+
return rows;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function num(v) {
|
|
93
|
+
return typeof v === 'number' && Number.isFinite(v) ? v : (v == null ? null : Number(v));
|
|
94
|
+
}
|
|
95
|
+
function round(n) { return Math.round(n * 1e6) / 1e6; }
|
|
96
|
+
function safeJson(v) {
|
|
97
|
+
try { return JSON.stringify(v); } catch (_e) { return String(v); }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
module.exports = { Gate, reportDelta };
|