driftproof 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,118 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { spawn } = require('child_process');
5
+ const { withRetry, withTimeout } = require('./json');
6
+ const { stubComplete, stubEnabled } = require('./stub');
7
+
8
+ // Provider abstraction: one `complete()` call, two surfaces.
9
+ //
10
+ // CLAUDE_PROVIDER=api → Anthropic Messages API (needs ANTHROPIC_API_KEY).
11
+ // CLAUDE_PROVIDER=cli → spawn `claude -p` with ANTHROPIC_API_KEY STRIPPED
12
+ // from the child env, so dev runs draw on the local
13
+ // subscription session credit rather than metered API
14
+ // billing. This is the DEFAULT for dev.
15
+ //
16
+ // The surface actually used is returned so the runner can stamp it into the
17
+ // receipt (run.surface = "api" | "claude-cli").
18
+
19
+ // Short model aliases → canonical ids. Kept tiny and explicit; unknown values
20
+ // are passed through verbatim so a full model id always works.
21
+ const MODEL_ALIASES = {
22
+ haiku: 'claude-haiku-4-5-20251001',
23
+ sonnet: 'claude-sonnet-5', // current Sonnet
24
+ 'sonnet-5': 'claude-sonnet-5',
25
+ 'sonnet-4-6': 'claude-sonnet-4-6', // previous Sonnet point release
26
+ opus: 'claude-opus-4-8',
27
+ };
28
+
29
+ function resolveModel(m) {
30
+ return MODEL_ALIASES[m] || m;
31
+ }
32
+
33
+ function providerName() {
34
+ return (process.env.CLAUDE_PROVIDER || 'cli').toLowerCase();
35
+ }
36
+
37
+ // The receipt-facing label for the surface in use.
38
+ function surfaceLabel() {
39
+ return providerName() === 'api' ? 'api' : 'claude-cli';
40
+ }
41
+
42
+ // Send a single-turn prompt and return { text, usage, surface }.
43
+ // usage is { input_tokens, output_tokens } when the surface reports it (api),
44
+ // else null (cli does not expose token counts to us).
45
+ // `temperature` is honoured ONLY on the api surface (the Messages API exposes
46
+ // it). On the cli surface sampling params are surface-controlled — we cannot set
47
+ // them, so temperature is ignored there and the receipt records that fact.
48
+ async function complete({ system, prompt, model, maxTokens = 1024, timeoutMs = 120000, temperature = undefined }) {
49
+ const surface = surfaceLabel();
50
+ // Offline stub surface: return a canned completion with zero model calls. The
51
+ // receipt still records the real surface label so a stub run is not mistaken
52
+ // for a genuine one at read time — only run.judge/generation TEXT is canned.
53
+ if (stubEnabled()) return { ...stubComplete({ system, prompt }), surface };
54
+ const runner = () => (surface === 'api'
55
+ ? completeApi({ system, prompt, model, maxTokens, temperature })
56
+ : completeCli({ system, prompt, model, timeoutMs }));
57
+ // A published run makes ~1000s of calls; be patient with transient
58
+ // throttling/cold-starts so one blip doesn't abort a multi-hour grind.
59
+ const out = await withRetry(() => withTimeout(runner, timeoutMs, `provider(${surface})`), { tries: 4, baseDelayMs: 3000 });
60
+ return { ...out, surface };
61
+ }
62
+
63
+ async function completeApi({ system, prompt, model, maxTokens, temperature }) {
64
+ let Anthropic;
65
+ try { Anthropic = require('@anthropic-ai/sdk'); }
66
+ catch (_e) { throw new Error('CLAUDE_PROVIDER=api requires the @anthropic-ai/sdk package (npm i @anthropic-ai/sdk)'); }
67
+ if (!process.env.ANTHROPIC_API_KEY) throw new Error('CLAUDE_PROVIDER=api requires ANTHROPIC_API_KEY');
68
+ const client = new Anthropic({ apiKey: process.env.ANTHROPIC_API_KEY });
69
+ const params = {
70
+ model: resolveModel(model),
71
+ max_tokens: maxTokens,
72
+ messages: [{ role: 'user', content: prompt }],
73
+ };
74
+ if (system) params.system = system;
75
+ if (temperature !== undefined) params.temperature = temperature;
76
+ const resp = await client.messages.create(params);
77
+ const text = (resp.content || []).filter((b) => b.type === 'text').map((b) => b.text).join('');
78
+ return {
79
+ text,
80
+ usage: {
81
+ input_tokens: (resp.usage && resp.usage.input_tokens) || 0,
82
+ output_tokens: (resp.usage && resp.usage.output_tokens) || 0,
83
+ },
84
+ };
85
+ }
86
+
87
+ function completeCli({ system, prompt, model, timeoutMs }) {
88
+ return new Promise((resolve, reject) => {
89
+ // Strip ANTHROPIC_API_KEY so the CLI uses the subscription session, not the
90
+ // metered API key. Everything else in the env is preserved.
91
+ const env = { ...process.env };
92
+ delete env.ANTHROPIC_API_KEY;
93
+
94
+ const args = ['-p', '--model', resolveModel(model)];
95
+ if (system) args.push('--append-system-prompt', system);
96
+
97
+ const child = spawn('claude', args, { env, stdio: ['pipe', 'pipe', 'pipe'] });
98
+ let out = '';
99
+ let err = '';
100
+ const killer = setTimeout(() => { child.kill('SIGKILL'); }, timeoutMs + 5000);
101
+ child.stdout.on('data', (d) => { out += d; });
102
+ child.stderr.on('data', (d) => { err += d; });
103
+ child.on('error', (e) => {
104
+ clearTimeout(killer);
105
+ if (e.code === 'ENOENT') reject(new Error("CLAUDE_PROVIDER=cli requires the `claude` CLI on PATH. Install it or set CLAUDE_PROVIDER=api."));
106
+ else reject(e);
107
+ });
108
+ child.on('close', (code) => {
109
+ clearTimeout(killer);
110
+ if (code !== 0) return reject(new Error(`claude CLI exited ${code}: ${err.slice(0, 400)}`));
111
+ resolve({ text: out.trim(), usage: null });
112
+ });
113
+ child.stdin.write(prompt);
114
+ child.stdin.end();
115
+ });
116
+ }
117
+
118
+ module.exports = { complete, resolveModel, surfaceLabel, providerName, MODEL_ALIASES };
package/lib/receipt.js ADDED
@@ -0,0 +1,133 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const fs = require('fs');
5
+ const path = require('path');
6
+ const { canonicalize, sha256 } = require('./canonical');
7
+ const { RECEIPT_SCHEMA_VERSION } = require('../config');
8
+ const { aggregateBands, combineUncertainty, round } = require('./stats');
9
+
10
+ // Schema file per receipt version. The current schema is receipt.schema.json;
11
+ // older versions live alongside it so v0.1 receipts still validate. The
12
+ // validator picks the schema by the receipt's own schema_version field.
13
+ const SCHEMA_FILES = {
14
+ '0.1': 'receipt.v0.1.schema.json',
15
+ '0.2': 'receipt.v0.2.schema.json',
16
+ '0.3': 'receipt.schema.json',
17
+ };
18
+
19
+ const _validators = {};
20
+ // Lazily compile the JSON Schema validator (ajv) for a given version. Kept lazy
21
+ // so the library can be required without ajv present (pure hashing utilities).
22
+ function getValidator(version) {
23
+ const v = SCHEMA_FILES[version] ? version : RECEIPT_SCHEMA_VERSION;
24
+ if (_validators[v]) return _validators[v];
25
+ let Ajv;
26
+ // The schema is JSON Schema draft 2020-12, so use ajv's 2020 build.
27
+ try { Ajv = require('ajv/dist/2020'); }
28
+ catch (_e) { throw new Error('receipt validation requires the `ajv` package (npm install)'); }
29
+ const schema = JSON.parse(fs.readFileSync(path.join(__dirname, '..', 'spec', SCHEMA_FILES[v]), 'utf8'));
30
+ const ajv = new Ajv({ allErrors: true, strict: false });
31
+ _validators[v] = ajv.compile(schema);
32
+ return _validators[v];
33
+ }
34
+
35
+ // Compute the receipt_hash: sha256 over the canonical receipt JSON with the
36
+ // receipt_hash field itself removed. Deterministic and reproducible.
37
+ function computeReceiptHash(receipt) {
38
+ const { receipt_hash, ...rest } = receipt; // eslint-disable-line no-unused-vars
39
+ return sha256(canonicalize(rest));
40
+ }
41
+
42
+ // Stamp (or re-stamp) the receipt_hash in place and return the receipt.
43
+ function sealReceipt(receipt) {
44
+ receipt.receipt_hash = computeReceiptHash(receipt);
45
+ return receipt;
46
+ }
47
+
48
+ // Verify a receipt's self-hash matches its contents (tamper-evidence lite).
49
+ function verifyReceiptHash(receipt) {
50
+ return receipt.receipt_hash === computeReceiptHash(receipt);
51
+ }
52
+
53
+ // Validate against the schema matching the receipt's own schema_version (so both
54
+ // v0.1 and v0.2 receipts validate). Returns { valid, errors, version }.
55
+ function validateReceipt(receipt) {
56
+ const version = (receipt && receipt.schema_version) || RECEIPT_SCHEMA_VERSION;
57
+ const validate = getValidator(version);
58
+ const valid = validate(receipt);
59
+ return { valid, errors: valid ? [] : (validate.errors || []), version };
60
+ }
61
+
62
+ function mean(nums) {
63
+ if (!nums.length) return 0;
64
+ return round(nums.reduce((a, b) => a + b, 0) / nums.length);
65
+ }
66
+
67
+ // Aggregate one mode's cases: mean of case means, band (suite dispersion — the
68
+ // stddev of the per-case means; see lib/stats.aggregateBands), and pass count.
69
+ function aggregate(caseResults) {
70
+ const band = aggregateBands(caseResults.map((r) => ({ mean: r.mean != null ? r.mean : r.score, stddev: r.stddev || 0, n: Array.isArray(r.samples) ? r.samples.length : 1 })));
71
+ const passes = caseResults.filter((r) => r.outcome === 'pass').length;
72
+ const borderline = caseResults.filter((r) => r.outcome === 'borderline').length;
73
+ return { case_count: caseResults.length, pass_count: passes, borderline_count: borderline, mean_score: band.mean, stddev: band.stddev };
74
+ }
75
+
76
+ // Assemble a full receipt from the runner's raw pieces, seal it, and return it.
77
+ // skill: { name, version, contentHash }
78
+ // suite: { format, suiteHash, caseCount }
79
+ // run: { model_id, model_release_date, surface, runner_version, date_utc, judge, registry, transcripts }
80
+ // cases: [ { id, mode, outcome, score, mean, stddev, samples, generation_hash, judge_sample_hashes, threshold, reason, judge } ]
81
+ // editorialReviews: optional [ { url, source, date } ]
82
+ function buildReceipt({ skill, suite, run, cases, verificationLevel = 'TESTED', editorialReviews = null }) {
83
+ const withSkill = cases.filter((c) => c.mode === 'with_skill');
84
+ const baseline = cases.filter((c) => c.mode === 'baseline');
85
+ const aggWith = aggregate(withSkill);
86
+ const aggBase = aggregate(baseline);
87
+
88
+ const receipt = {
89
+ schema_version: RECEIPT_SCHEMA_VERSION,
90
+ skill: {
91
+ name: skill.name,
92
+ version: skill.version,
93
+ content_hash: skill.contentHash,
94
+ },
95
+ suite: {
96
+ format: suite.format,
97
+ suite_hash: suite.suiteHash,
98
+ case_count: suite.caseCount,
99
+ },
100
+ run: {
101
+ model_id: run.model_id,
102
+ model_release_date: run.model_release_date == null ? null : run.model_release_date,
103
+ surface: run.surface,
104
+ runner_version: run.runner_version,
105
+ date_utc: run.date_utc,
106
+ // v0.3: registry provenance + transcript-retention mode. Defaults keep the
107
+ // honest, cheapest interpretation when a caller omits them.
108
+ registry: run.registry || 'unregistered',
109
+ transcripts: run.transcripts || 'hashes-only',
110
+ judge: run.judge || { samples: 1, temperature: null, sampling: 'single', surface: run.surface },
111
+ },
112
+ results: {
113
+ cases,
114
+ aggregates: { with_skill: aggWith, baseline: aggBase },
115
+ },
116
+ comparison: {
117
+ with_skill_score: aggWith.mean_score,
118
+ baseline_score: aggBase.mean_score,
119
+ delta: round(aggWith.mean_score - aggBase.mean_score),
120
+ // Combined uncertainty of the delta: quadrature sum of the two aggregate
121
+ // bands. Diff uses this for the headline "within noise" vs real-move rule.
122
+ delta_uncertainty: combineUncertainty(aggWith.stddev, aggBase.stddev),
123
+ },
124
+ verification_level: verificationLevel,
125
+ receipt_hash: '',
126
+ };
127
+ if (editorialReviews && editorialReviews.length) receipt.editorial_reviews = editorialReviews;
128
+ return sealReceipt(receipt);
129
+ }
130
+
131
+ module.exports = {
132
+ buildReceipt, sealReceipt, computeReceiptHash, verifyReceiptHash, validateReceipt,
133
+ };
package/lib/run.js ADDED
@@ -0,0 +1,222 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ const { complete, resolveModel, surfaceLabel } = require('./provider');
5
+ const { gradeSamples, judgeSettings } = require('./judge');
6
+ const { buildReceipt } = require('./receipt');
7
+ const { sha256 } = require('./canonical');
8
+ const { registryStatus } = require('./models');
9
+ const { perCallCostUSD } = require('./cost');
10
+ const { RUNNER_VERSION, DEFAULT_JUDGE_SAMPLES } = require('../config');
11
+
12
+ // Known model release dates (best-effort; null when unknown). Recorded into the
13
+ // receipt so drift reports can order runs by model age. Dateless model ids
14
+ // (the 4.6 generation onward) carry no date, so the announcement date is
15
+ // recorded explicitly here from Anthropic's public launch posts (provenance
16
+ // noted in the report; consistent with spec open question #4 — dates are
17
+ // best-effort, not verified against the Models API in this run).
18
+ const MODEL_RELEASE_DATES = {
19
+ 'claude-haiku-4-5-20251001': '2025-10-01',
20
+ 'claude-haiku-4-5': '2025-10-01',
21
+ 'claude-sonnet-5': '2026-06-30', // anthropic.com/news/claude-sonnet-5
22
+ 'claude-sonnet-4-6': '2026-02-17', // anthropic.com/news/claude-sonnet-4-6
23
+ };
24
+
25
+ function releaseDateFor(modelId) {
26
+ if (MODEL_RELEASE_DATES[modelId]) return MODEL_RELEASE_DATES[modelId];
27
+ // Derive from a trailing YYYYMMDD in the id if present.
28
+ const m = String(modelId).match(/(\d{4})(\d{2})(\d{2})$/);
29
+ return m ? `${m[1]}-${m[2]}-${m[3]}` : null;
30
+ }
31
+
32
+ // Calls for one model run: each case does 2 generations (with_skill + baseline)
33
+ // and 2×samples judge calls (each generation judged `samples` times).
34
+ function projectCalls(caseCount, samples) {
35
+ return caseCount * (2 + 2 * samples);
36
+ }
37
+
38
+ // Ask the target model to perform one eval case. `withSkill` decides whether the
39
+ // SKILL.md is prepended as a system prompt (the whole point: measure the skill's
40
+ // marginal effect vs a bare baseline).
41
+ async function runCase({ skillMd, caseObj, model, withSkill, timeoutMs }) {
42
+ const system = withSkill ? skillMd : undefined;
43
+ const { text, usage } = await complete({ system, prompt: caseObj.prompt, model, maxTokens: 1024, timeoutMs });
44
+ return { text, usage };
45
+ }
46
+
47
+ // Determine a case outcome from its sampled band and threshold.
48
+ // borderline : threshold lies within [mean - stddev, mean + stddev]
49
+ // pass/fail : mean clears / misses the threshold with the band clear of it
50
+ // score : un-thresholded case (report the number, no pass/fail)
51
+ function outcomeFor(mean, stddev, threshold) {
52
+ if (typeof threshold !== 'number') return 'score';
53
+ if (mean - stddev <= threshold && threshold <= mean + stddev) return 'borderline';
54
+ return mean >= threshold ? 'pass' : 'fail';
55
+ }
56
+
57
+ // Judge one generated response `samples` times → sampled case result.
58
+ // `generationHash` binds the graded case to the exact generation text (v0.3).
59
+ // Returns { caseResult, sampleTexts } — sampleTexts is transient (retained only
60
+ // under --keep-transcripts; never part of the receipt).
61
+ async function judgeCase({ caseObj, response, generationHash, judgeModel, mode, timeoutMs, samples }) {
62
+ const g = await gradeSamples({ task: caseObj.prompt, response, rubric: caseObj.rubric, model: judgeModel, samples, timeoutMs });
63
+ const outcome = outcomeFor(g.mean, g.stddev, caseObj.pass_threshold);
64
+ const caseResult = {
65
+ id: caseObj.id,
66
+ mode,
67
+ outcome,
68
+ score: g.mean, // `score` == sampled mean (v0.1 readers still work)
69
+ mean: g.mean,
70
+ stddev: g.stddev,
71
+ samples: g.samples,
72
+ // v0.3 transcript auditability: hash of the graded generation + one hash per
73
+ // judge sample. These make a receipt checkable against retained transcripts.
74
+ generation_hash: generationHash,
75
+ judge_sample_hashes: g.sample_hashes,
76
+ threshold: typeof caseObj.pass_threshold === 'number' ? caseObj.pass_threshold : null,
77
+ reason: g.reason,
78
+ judge: { model_id: g.model_id, rubric_hash: g.rubric_hash },
79
+ };
80
+ return { caseResult, sampleTexts: g.sample_texts };
81
+ }
82
+
83
+ // Run up to `concurrency` async tasks at a time, preserving input order in the
84
+ // results. Keeps the CLI grind tractable (each `claude -p` cold-start dominates
85
+ // wall-clock, so a handful of concurrent spawns is a large speedup) without
86
+ // unbounded fan-out.
87
+ async function mapPool(items, concurrency, fn) {
88
+ const results = new Array(items.length);
89
+ let next = 0;
90
+ const workers = new Array(Math.max(1, Math.min(concurrency, items.length))).fill(0).map(async () => {
91
+ while (true) {
92
+ const i = next++;
93
+ if (i >= items.length) return;
94
+ results[i] = await fn(items[i], i);
95
+ }
96
+ });
97
+ await Promise.all(workers);
98
+ return results;
99
+ }
100
+
101
+ // Run the full suite for ONE model, both modes, sampled judge-grading, and build
102
+ // a sealed receipt. Enforces a hard call cap; every model+judge call counts.
103
+ //
104
+ // opts: { maxCases, maxCalls, samples, judgeModel, timeoutMs, concurrency,
105
+ // onProgress, budget, keepTranscripts, nowIso }
106
+ // budget — optional BudgetTracker; accumulates estimated per-call
107
+ // USD as the run proceeds and hard-stops at 1.25× the cap.
108
+ // keepTranscripts — when true, the run records transcripts:"retained-local"
109
+ // and returns the raw generations + judge outputs so the
110
+ // caller can write them to transcripts/<receipt-id>/.
111
+ async function runSkillOnModel({ skill, model, opts = {} }) {
112
+ const modelId = resolveModel(model);
113
+ const judgeModel = opts.judgeModel ? resolveModel(opts.judgeModel) : modelId;
114
+ const timeoutMs = opts.timeoutMs || 120000;
115
+ const maxCalls = opts.maxCalls || 200;
116
+ const samples = opts.samples || DEFAULT_JUDGE_SAMPLES;
117
+ const concurrency = Math.max(1, opts.concurrency || 1);
118
+ const onProgress = opts.onProgress || (() => {});
119
+ const budget = opts.budget || null;
120
+ const keepTranscripts = !!opts.keepTranscripts;
121
+
122
+ let cases = skill.suite.cases;
123
+ if (opts.maxCases && cases.length > opts.maxCases) cases = cases.slice(0, opts.maxCases);
124
+
125
+ // Cost guard: project the whole run up front and refuse before spending a
126
+ // single call if it would blow the cap.
127
+ const projected = projectCalls(cases.length, samples);
128
+ if (projected > maxCalls) {
129
+ const e = new Error(`cost guard: projected ${projected} calls exceeds cap ${maxCalls} (${cases.length} cases × (2 + 2×${samples} samples)). Raise --max-calls or lower --max-cases/--samples.`);
130
+ e.code = 'CALL_CAP';
131
+ throw e;
132
+ }
133
+
134
+ // One task per (case, mode). Order is preserved in the receipt regardless of
135
+ // completion order, so receipts are deterministic under concurrency.
136
+ const tasks = [];
137
+ for (const c of cases) for (const withSkill of [true, false]) tasks.push({ c, withSkill });
138
+
139
+ let calls = 0;
140
+ const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
141
+ const mode = withSkill ? 'with_skill' : 'baseline';
142
+ onProgress({ case: c.id, mode, phase: 'generate' });
143
+ const { text } = await runCase({ skillMd: skill.skillMd, caseObj: c, model: modelId, withSkill, timeoutMs });
144
+ calls += 1;
145
+ // Live budget: count the generation call, then hard-stop if over 1.25× cap.
146
+ if (budget) budget.add(perCallCostUSD(modelId, withSkill ? 'gen_with_skill' : 'gen_baseline'));
147
+ const generationHash = sha256(String(text || ''));
148
+ onProgress({ case: c.id, mode, phase: 'judge', samples });
149
+ const { caseResult, sampleTexts } = await judgeCase({ caseObj: c, response: text, generationHash, judgeModel, mode, timeoutMs, samples });
150
+ calls += samples;
151
+ // Live budget: count all `samples` judge calls for this (case, mode).
152
+ if (budget) budget.add(samples * perCallCostUSD(judgeModel, 'judge'));
153
+ onProgress({ case: c.id, mode, phase: 'done', outcome: caseResult.outcome, score: caseResult.mean, stddev: caseResult.stddev });
154
+ const transcript = keepTranscripts
155
+ ? { id: c.id, mode, generation: String(text || ''), judge_outputs: sampleTexts }
156
+ : null;
157
+ return { caseResult, transcript };
158
+ });
159
+ const caseResults = pairs.map((p) => p.caseResult);
160
+ const transcripts = keepTranscripts ? pairs.map((p) => p.transcript) : null;
161
+
162
+ const receipt = buildReceipt({
163
+ skill: { name: skill.name, version: skill.version, contentHash: skill.contentHash },
164
+ suite: { format: skill.suite.format, suiteHash: skill.suite.suiteHash, caseCount: skill.suite.caseCount },
165
+ run: {
166
+ model_id: modelId,
167
+ model_release_date: releaseDateFor(modelId),
168
+ surface: surfaceLabel(),
169
+ runner_version: RUNNER_VERSION,
170
+ date_utc: opts.nowIso || new Date().toISOString(),
171
+ registry: registryStatus(modelId),
172
+ transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
173
+ judge: judgeSettings(samples),
174
+ },
175
+ cases: caseResults,
176
+ verificationLevel: 'TESTED',
177
+ });
178
+
179
+ return { receipt, calls, transcripts };
180
+ }
181
+
182
+ function band(mean, sd) { return `${mean.toFixed(3)} ± ${sd.toFixed(3)}`; }
183
+
184
+ // Render a short human-readable markdown summary of a receipt.
185
+ function summarizeReceipt(receipt) {
186
+ const L = [];
187
+ L.push(`# ${receipt.skill.name} — receipt summary`);
188
+ L.push('');
189
+ L.push(`- **model:** \`${receipt.run.model_id}\`${receipt.run.model_release_date ? ` (released ${receipt.run.model_release_date})` : ''}`);
190
+ L.push(`- **surface:** ${receipt.run.surface}`);
191
+ L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
192
+ L.push(`- **runner:** v${receipt.run.runner_version}`);
193
+ const j = receipt.run.judge || {};
194
+ L.push(`- **judge:** ${j.samples || 1} samples/case, temperature ${j.temperature == null ? 'n/a (surface-controlled)' : j.temperature} (${j.sampling || 'single'})`);
195
+ if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
196
+ L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
197
+ L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
198
+ L.push(`- **verification:** ${receipt.verification_level}`);
199
+ L.push(`- **receipt_hash:** \`${receipt.receipt_hash.slice(0, 16)}…\``);
200
+ L.push('');
201
+ L.push(`## Headline`);
202
+ L.push('');
203
+ const cmp = receipt.comparison;
204
+ const aggs = receipt.results.aggregates;
205
+ const sign = cmp.delta >= 0 ? '+' : '';
206
+ L.push(`with_skill **${band(aggs.with_skill.mean_score, aggs.with_skill.stddev)}** vs baseline **${band(aggs.baseline.mean_score, aggs.baseline.stddev)}**`);
207
+ L.push('');
208
+ L.push(`skill lift **${sign}${cmp.delta.toFixed(3)}** (combined uncertainty ± ${cmp.delta_uncertainty.toFixed(3)})`);
209
+ L.push('');
210
+ L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples || 1} judge samples)`);
211
+ L.push('');
212
+ L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
213
+ L.push(`|---|---|---|---|---|`);
214
+ for (const c of receipt.results.cases) {
215
+ const flag = c.outcome === 'borderline' ? ' ⚠' : '';
216
+ L.push(`| \`${c.id}\` | ${c.mode} | ${c.outcome}${flag} | ${band(c.mean, c.stddev || 0)} | ${c.reason || ''} |`);
217
+ }
218
+ L.push('');
219
+ return L.join('\n');
220
+ }
221
+
222
+ module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor };
package/lib/runner.js ADDED
@@ -0,0 +1,100 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // The run / compare / report-delta / exit-nonzero engine.
5
+ //
6
+ // PORTED (then fully genericized) from a private behavioral-gate spec. The
7
+ // original drove a browser against a product and asserted product-specific
8
+ // journeys; ALL of that (endpoints, tenant probes, chrome tokens, journeys) was
9
+ // stripped. What remains is the reusable core:
10
+ // - a Gate collects named checks grouped by "section"
11
+ // - each check records { section, name, pass, detail }
12
+ // - summarize() prints a PASS/FAIL line per check and a headline count
13
+ // - toExitCode() gives 0 on all-pass, 1 on any failure
14
+ // Scanned clean against the confidentiality deny-list.
15
+
16
+ // A single collector for one gate run.
17
+ class Gate {
18
+ constructor(title = 'gate') {
19
+ this.title = title;
20
+ this.results = [];
21
+ this._section = '?';
22
+ }
23
+
24
+ // Start a new named section; subsequent checks are grouped under it.
25
+ section(name) {
26
+ this._section = name;
27
+ return this;
28
+ }
29
+
30
+ // Record an assertion. `pass` is coerced to boolean; `detail` is any JSON-able
31
+ // context printed only when the check fails.
32
+ check(name, pass, detail = null) {
33
+ const rec = { section: this._section, name, pass: !!pass, detail: pass ? null : detail };
34
+ this.results.push(rec);
35
+ const tag = rec.pass ? 'PASS' : 'FAIL';
36
+ const suffix = rec.pass ? '' : ` -- ${safeJson(detail)}`;
37
+ // eslint-disable-next-line no-console
38
+ console.log(` [${tag}] ${name}${suffix}`);
39
+ return rec.pass;
40
+ }
41
+
42
+ // Convenience: assert deep equality of two JSON-able values.
43
+ checkEqual(name, actual, expected) {
44
+ const pass = safeJson(actual) === safeJson(expected);
45
+ return this.check(name, pass, { actual, expected });
46
+ }
47
+
48
+ get passed() { return this.results.filter((r) => r.pass).length; }
49
+ get failed() { return this.results.filter((r) => !r.pass); }
50
+ get total() { return this.results.length; }
51
+
52
+ // Print the headline and per-failure detail. Returns the summary object.
53
+ summarize() {
54
+ const failed = this.failed;
55
+ // eslint-disable-next-line no-console
56
+ console.log(`\n=== ${this.title.toUpperCase()} RESULT: ${this.passed}/${this.total} passed, ${failed.length} failed ===`);
57
+ for (const f of failed) {
58
+ // eslint-disable-next-line no-console
59
+ console.log(` FAIL [${f.section}] ${f.name}: ${safeJson(f.detail)}`);
60
+ }
61
+ return { title: this.title, total: this.total, passed: this.passed, failed: failed.length, results: this.results };
62
+ }
63
+
64
+ toExitCode() {
65
+ return this.failed.length === 0 ? 0 : 1;
66
+ }
67
+ }
68
+
69
+ // Compute the per-item delta between a "before" and "after" map of numeric
70
+ // scores, keyed by id. Returns entries with { id, before, after, delta,
71
+ // regressed } sorted worst-regression-first. This is the generic report-delta
72
+ // primitive the drift report builds on.
73
+ function reportDelta(beforeById, afterById, { regressionEpsilon = 0.0 } = {}) {
74
+ const ids = new Set([...Object.keys(beforeById || {}), ...Object.keys(afterById || {})]);
75
+ const rows = [];
76
+ for (const id of ids) {
77
+ const before = num(beforeById && beforeById[id]);
78
+ const after = num(afterById && afterById[id]);
79
+ const delta = (after == null || before == null) ? null : round(after - before);
80
+ const regressed = delta != null && delta < -regressionEpsilon;
81
+ rows.push({ id, before, after, delta, regressed });
82
+ }
83
+ rows.sort((a, b) => {
84
+ // Regressions first, then by magnitude of drop.
85
+ const da = a.delta == null ? 0 : a.delta;
86
+ const db = b.delta == null ? 0 : b.delta;
87
+ return da - db;
88
+ });
89
+ return rows;
90
+ }
91
+
92
+ function num(v) {
93
+ return typeof v === 'number' && Number.isFinite(v) ? v : (v == null ? null : Number(v));
94
+ }
95
+ function round(n) { return Math.round(n * 1e6) / 1e6; }
96
+ function safeJson(v) {
97
+ try { return JSON.stringify(v); } catch (_e) { return String(v); }
98
+ }
99
+
100
+ module.exports = { Gate, reportDelta };