driftproof 0.11.2 → 0.11.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -11
- package/bin/driftproof +194 -8
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +40 -7
- package/lib/diff.js +12 -3
- package/lib/export.js +4 -1
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/receipt.js +72 -3
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +26 -9
- package/lib/stale.js +194 -0
- package/lib/verdict.js +18 -0
- package/package.json +1 -1
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/lib/reuse.js
CHANGED
|
@@ -21,6 +21,9 @@
|
|
|
21
21
|
// receipt's own recorded provenance — model, suite and skill content force
|
|
22
22
|
// a rerun because the generation would differ; judge and rubric force a
|
|
23
23
|
// regrade because only the scoring would; neither changing permits reuse.
|
|
24
|
+
// Spec 053 made it one comparison of two provenance records, shared with
|
|
25
|
+
// `driftproof stale` (lib/stale.js), and added the harness and the judge's
|
|
26
|
+
// template; see PROVENANCE below.
|
|
24
27
|
//
|
|
25
28
|
// PURE: two receipts in, a decision out. No I/O, no provider. F-009-K's lesson
|
|
26
29
|
// applied — the decision derives from the artifact, never from a filename.
|
|
@@ -120,15 +123,147 @@ function compare(older, newer) {
|
|
|
120
123
|
return { verdict: 'MEASURED', delta: round(b - a), baseline_reproduced: true };
|
|
121
124
|
}
|
|
122
125
|
|
|
126
|
+
// ── PROVENANCE: ONE COMPARISON FOR TRIAGE AND STALE (spec 053 AC-1, AC-2) ─────
|
|
127
|
+
//
|
|
128
|
+
// What a receipt records about how its numbers were made, and one pure function
|
|
129
|
+
// that compares two such records: two receipts (`triage`), or a receipt and what
|
|
130
|
+
// would run today (`driftproof stale`, lib/stale.js). Before spec 053 `triage`
|
|
131
|
+
// read `run.rubric_hash`, which no receipt carries (rubric hashes are per case),
|
|
132
|
+
// and never looked at the judge's template or the harness, so a rubric or
|
|
133
|
+
// template change was reused and a harness change went unseen.
|
|
134
|
+
//
|
|
135
|
+
// UNKNOWN IS NEVER CURRENT. An axis either side does not record is `unknown`,
|
|
136
|
+
// with the reason, and an arm with an unknown axis is not reused.
|
|
137
|
+
//
|
|
138
|
+
// THE SUITE HASH COVERS PROMPTS AND RUBRICS TOGETHER ({id, prompt, rubric,
|
|
139
|
+
// pass_threshold}), and no receipt records a per-case prompt hash. So a
|
|
140
|
+
// different suite reruns both arms even when only a rubric moved: the record
|
|
141
|
+
// cannot show the prompts did not. A case's rubric hash moving under an equal
|
|
142
|
+
// suite hash (the rubric hash also covers the judge's system prompt) regrades
|
|
143
|
+
// that case (spec 053 R-3).
|
|
144
|
+
|
|
145
|
+
// The axes, in the order a reader is shown them, with what a difference does
|
|
146
|
+
// and to which arms.
|
|
147
|
+
const AXES = [
|
|
148
|
+
{ axis: 'model', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the model differs, so the generation would differ' },
|
|
149
|
+
{ axis: 'harness', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the harness differs, so the generation would differ' },
|
|
150
|
+
{ axis: 'skill', effect: 'rerun', arms: ['with_skill'], why: 'the skill content differs, so the with-skill generation would differ; the baseline arm contains no skill text and still stands' },
|
|
151
|
+
{ axis: 'suite', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the suite differs, so the prompts may differ' },
|
|
152
|
+
{ axis: 'judge_model', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'the judge model differs, so the existing generations can be graded again' },
|
|
153
|
+
{ axis: 'judge_template', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'the judge prompt template differs, so the existing generations can be graded again' },
|
|
154
|
+
{ axis: 'rubric', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'a case\'s rubric hash differs under the same suite, so that case can be graded again' },
|
|
155
|
+
];
|
|
156
|
+
const UNRECORDED = { model: 'no model id', harness: 'no harness (receipts before v0.9 carry none)', skill: 'no skill content hash', suite: 'no suite hash', judge_model: 'no judge model', judge_template: 'no judge prompt template hash (receipts before v0.6 carry none)', rubric: 'no per-case rubric hash' };
|
|
157
|
+
|
|
158
|
+
const canonicalId = (id) => (id == null ? null : String(id).replace(/-\d{8}$/, ''));
|
|
159
|
+
|
|
160
|
+
// THE HARNESS BY SEMVER (spec 053 A-053-1, the operator's instruction of 24 Sep 2026, a departure
|
|
161
|
+
// from the brief's exact match). A major or minor change (2.1 to 2.2) reruns both arms; a patch-only
|
|
162
|
+
// change (2.1.281 to 2.1.282) is an advisory. Claude Code ships a patch release every few days, and
|
|
163
|
+
// an exact match would mark every receipt stale within days. A version that is not semver, or a
|
|
164
|
+
// different harness, must match exactly.
|
|
165
|
+
const semver = (v) => { const m = /^(\d+)\.(\d+)\.(\d+)(.*)$/.exec(String(v)); return m ? [m[1], m[2], m[3], m[4]] : null; };
|
|
166
|
+
function harnessDrift(a, b) {
|
|
167
|
+
if (a.name !== b.name) return 'moved';
|
|
168
|
+
if (a.version === b.version) return 'same';
|
|
169
|
+
const x = semver(a.version); const y = semver(b.version);
|
|
170
|
+
if (!x || !y) return 'moved';
|
|
171
|
+
if (x[0] === y[0] && x[1] === y[1]) return x[2] === y[2] && x[3] === y[3] ? 'same' : 'patch';
|
|
172
|
+
return 'moved';
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// The provenance a receipt records. Every field is null when the receipt does
|
|
176
|
+
// not record it; an imported receipt's `unknown` model is null.
|
|
177
|
+
function provenanceOf(r) {
|
|
178
|
+
const run = (r && r.run) || {};
|
|
179
|
+
const judge = run.judge || {};
|
|
180
|
+
const cases = (r && r.results && Array.isArray(r.results.cases)) ? r.results.cases : [];
|
|
181
|
+
const rubrics = {};
|
|
182
|
+
for (const c of cases) if (c && c.judge && typeof c.judge.rubric_hash === 'string' && !(c.id in rubrics)) rubrics[c.id] = c.judge.rubric_hash;
|
|
183
|
+
const firstCaseJudge = (cases.find((c) => c && c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
184
|
+
return {
|
|
185
|
+
model_id: run.model_id && run.model_id !== 'unknown' ? run.model_id : null,
|
|
186
|
+
harness: run.harness && typeof run.harness.name === 'string' ? { name: run.harness.name, version: run.harness.version == null ? null : String(run.harness.version) } : null,
|
|
187
|
+
skill_hash: (r && r.skill && r.skill.content_hash) || null,
|
|
188
|
+
suite_hash: (r && r.suite && r.suite.suite_hash) || null,
|
|
189
|
+
case_ids: cases.length ? [...new Set(cases.map((c) => c.id))].sort() : null,
|
|
190
|
+
judge_model_id: judge.model_id || firstCaseJudge || null,
|
|
191
|
+
judge_template_hash: judge.prompt_template_hash || null,
|
|
192
|
+
rubrics: Object.keys(rubrics).length ? rubrics : null,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const show = (axis, v) => {
|
|
197
|
+
if (v == null) return null;
|
|
198
|
+
if (axis === 'harness') return v.version == null ? v.name : `${v.name} ${v.version}`;
|
|
199
|
+
if (axis === 'rubric' && typeof v === 'object') return `${Object.keys(v).length} case(s)`;
|
|
200
|
+
return v;
|
|
201
|
+
};
|
|
202
|
+
|
|
203
|
+
// One entry per axis: { axis, recorded, current, effect, arms, reason[, cases, cases_added, cases_removed] }.
|
|
204
|
+
// `why` optionally carries a reason for a current value that could not be found
|
|
205
|
+
// (lib/stale.js: "no --skill given", "claude --version failed: ...").
|
|
206
|
+
function compareProvenance(rec, cur, why = {}) {
|
|
207
|
+
const out = [];
|
|
208
|
+
const entry = (a, recorded, current, same, extra = {}) => {
|
|
209
|
+
const unknownSide = recorded == null ? `the receipt records ${UNRECORDED[a.axis]}` : current == null ? (why[a.axis] || `the other side records ${UNRECORDED[a.axis]}`) : null;
|
|
210
|
+
if (unknownSide) return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: 'unknown', arms: a.arms, reason: unknownSide, ...extra });
|
|
211
|
+
if (same) return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: 'current', arms: a.arms, reason: 'unchanged', ...extra });
|
|
212
|
+
return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: a.effect, arms: a.arms, reason: a.why, ...extra });
|
|
213
|
+
};
|
|
214
|
+
const A = Object.fromEntries(AXES.map((a) => [a.axis, a]));
|
|
215
|
+
entry(A.model, rec.model_id, cur.model_id, canonicalId(rec.model_id) === canonicalId(cur.model_id));
|
|
216
|
+
// A harness with no version recorded is unknown, except an API surface, which has no harness to drift.
|
|
217
|
+
const h = (x) => (x && (x.version != null || x.name === 'api') ? x : null);
|
|
218
|
+
const drift = h(rec.harness) && h(cur.harness) ? harnessDrift(rec.harness, cur.harness) : null;
|
|
219
|
+
if (drift === 'patch') out.push({ axis: 'harness', recorded: show('harness', rec.harness), current: show('harness', cur.harness), effect: 'advisory', arms: A.harness.arms, reason: 'a patch release of the same harness: an advisory, not a rerun (spec 053 A-053-1)' });
|
|
220
|
+
else entry(A.harness, h(rec.harness), h(cur.harness), drift === 'same');
|
|
221
|
+
entry(A.skill, rec.skill_hash, cur.skill_hash, rec.skill_hash === cur.skill_hash);
|
|
222
|
+
const added = rec.case_ids && cur.case_ids ? cur.case_ids.filter((id) => !rec.case_ids.includes(id)) : [];
|
|
223
|
+
const removed = rec.case_ids && cur.case_ids ? rec.case_ids.filter((id) => !cur.case_ids.includes(id)) : [];
|
|
224
|
+
entry(A.suite, rec.suite_hash, cur.suite_hash, rec.suite_hash === cur.suite_hash, added.length || removed.length ? { cases_added: added, cases_removed: removed } : {});
|
|
225
|
+
entry(A.judge_model, rec.judge_model_id, cur.judge_model_id, canonicalId(rec.judge_model_id) === canonicalId(cur.judge_model_id));
|
|
226
|
+
entry(A.judge_template, rec.judge_template_hash, cur.judge_template_hash, rec.judge_template_hash === cur.judge_template_hash);
|
|
227
|
+
// The rubric, per case, over the cases both sides record.
|
|
228
|
+
if (rec.rubrics && cur.rubrics) {
|
|
229
|
+
const moved = Object.keys(rec.rubrics).filter((id) => id in cur.rubrics && rec.rubrics[id] !== cur.rubrics[id]).sort();
|
|
230
|
+
const suiteMoved = out.find((e) => e.axis === 'suite').effect === 'rerun';
|
|
231
|
+
if (!moved.length) entry(A.rubric, 'recorded', 'recorded', true);
|
|
232
|
+
else if (suiteMoved) out.push({ axis: 'rubric', recorded: `${moved.length} case(s)`, current: `${moved.length} case(s)`, effect: 'rerun', arms: A.rubric.arms, cases: moved, reason: 'the rubric moved on these cases, and the suite differs, so the rerun the suite needs grades them again; the receipt records no per-case prompt hash, so a rubric edit cannot be told from a prompt edit' });
|
|
233
|
+
else out.push({ axis: 'rubric', recorded: `${moved.length} case(s)`, current: `${moved.length} case(s)`, effect: 'regrade', arms: A.rubric.arms, cases: moved, reason: A.rubric.why });
|
|
234
|
+
} else entry(A.rubric, rec.rubrics, cur.rubrics, false);
|
|
235
|
+
return out;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
// Each arm's decision: rerun, then regrade, then unknown, then reuse.
|
|
239
|
+
const RANK = { rerun: 3, regrade: 2, unknown: 1, current: 0, advisory: 0 };
|
|
240
|
+
function armDecisions(axes) {
|
|
241
|
+
const arms = {};
|
|
242
|
+
for (const arm of ['with_skill', 'baseline']) {
|
|
243
|
+
const on = axes.filter((e) => e.arms.includes(arm));
|
|
244
|
+
const top = on.reduce((m, e) => (RANK[e.effect] > RANK[m] ? e.effect : m), 'current');
|
|
245
|
+
const decision = RANK[top] === 0 ? 'reuse' : top;
|
|
246
|
+
const because = on.filter((e) => e.effect === top && decision !== 'reuse');
|
|
247
|
+
const cases = decision === 'regrade' && because.every((e) => Array.isArray(e.cases)) ? [...new Set(because.flatMap((e) => e.cases))].sort() : null;
|
|
248
|
+
arms[arm] = { decision, axes: because.map((e) => e.axis), ...(cases ? { cases } : {}) };
|
|
249
|
+
}
|
|
250
|
+
return arms;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Two receipts' provenance compared (spec 053 AC-2). The decision is the most
|
|
254
|
+
// consequential arm's: rerun, then regrade, then unknown, then reuse. `cases`
|
|
255
|
+
// names the cases a rubric regrade reaches.
|
|
123
256
|
function triage(a, b) {
|
|
124
|
-
const
|
|
125
|
-
const
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
257
|
+
const axes = compareProvenance(provenanceOf(a), provenanceOf(b));
|
|
258
|
+
const arms = armDecisions(axes);
|
|
259
|
+
const decision = ['rerun', 'regrade', 'unknown'].find((d) => arms.with_skill.decision === d || arms.baseline.decision === d) || 'reuse';
|
|
260
|
+
const first = axes.find((e) => e.effect === decision);
|
|
261
|
+
const advised = axes.filter((e) => e.effect === 'advisory').map((e) => e.axis);
|
|
262
|
+
const reason = decision === 'reuse'
|
|
263
|
+
? `model, harness, skill, suite, judge, template and rubric all match${advised.length ? ` but for a patch release of the ${advised.join(', ')} (an advisory)` : ''}, so the earlier result stands`
|
|
264
|
+
: decision === 'unknown' ? `${first.axis} is unknown: ${first.reason}` : first.reason;
|
|
265
|
+
const cases = decision === 'regrade' ? (arms.with_skill.cases || arms.baseline.cases || null) : null;
|
|
266
|
+
return { decision, reason, arms, axes, ...(cases ? { cases } : {}) };
|
|
132
267
|
}
|
|
133
268
|
|
|
134
|
-
module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf };
|
|
269
|
+
module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf, provenanceOf, compareProvenance, armDecisions, AXES };
|
package/lib/run.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE, buildSpawnPlan } = require('./provider');
|
|
5
|
+
const { spawnSync } = require('child_process');
|
|
5
6
|
const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
|
|
6
7
|
const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
|
|
7
8
|
const { sha256 } = require('./canonical');
|
|
@@ -32,6 +33,30 @@ const MODEL_RELEASE_DATES = {
|
|
|
32
33
|
'claude-opus-4-8': '2026-05-28', // anthropic.com/news/claude-opus-4-8
|
|
33
34
|
};
|
|
34
35
|
|
|
36
|
+
// The harness that will answer, recorded when a run starts (spec 053 AC-8, R-4): claude-code on
|
|
37
|
+
// claude-cli, codex on openai-cli, each version read from `<bin> --version` through the same spawn
|
|
38
|
+
// plan the run's calls use, so it is the binary that answers; "api" with no version on an API
|
|
39
|
+
// surface. A version that cannot be read is null and the reason is returned for the caller to print;
|
|
40
|
+
// it never stops a run. The receipt has no field for the reason (receipt v0.9's run.harness is
|
|
41
|
+
// {name, version}).
|
|
42
|
+
const HARNESS_OF = { 'claude-cli': ['claude-code', 'claude'], 'openai-cli': ['codex', 'codex'] };
|
|
43
|
+
function harnessFor(surface, trusted) {
|
|
44
|
+
if (surface === 'api' || surface === 'openai-api') return { harness: { name: 'api', version: null } };
|
|
45
|
+
const h = HARNESS_OF[surface];
|
|
46
|
+
if (!h) return { harness: null };
|
|
47
|
+
const [name, bin] = h;
|
|
48
|
+
try {
|
|
49
|
+
const plan = buildSpawnPlan({ bin, args: ['--version'], trusted });
|
|
50
|
+
const r = spawnSync(plan.file, plan.args, { env: plan.env, encoding: 'utf8', timeout: 30000 });
|
|
51
|
+
if (r.error) return { harness: { name, version: null }, reason: `${bin} --version could not run: ${r.error.code || r.error.message}` };
|
|
52
|
+
if (r.status !== 0) return { harness: { name, version: null }, reason: `${bin} --version exited ${r.status}` };
|
|
53
|
+
const m = /\d+\.\d+\.\d+[0-9A-Za-z.+-]*/.exec(r.stdout || '');
|
|
54
|
+
return m ? { harness: { name, version: m[0] } } : { harness: { name, version: null }, reason: `${bin} --version printed no version` };
|
|
55
|
+
} catch (e) {
|
|
56
|
+
return { harness: { name, version: null }, reason: `${bin} --version could not be planned: ${e.message}` };
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
35
60
|
function releaseDateFor(modelId) {
|
|
36
61
|
if (MODEL_RELEASE_DATES[modelId]) return MODEL_RELEASE_DATES[modelId];
|
|
37
62
|
// Derive from a trailing YYYYMMDD in the id if present.
|
|
@@ -263,6 +288,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
263
288
|
const reportedAll = new Set();
|
|
264
289
|
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
265
290
|
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
291
|
+
// v0.8 (spec 043 AC-4): when generation began, per receipt, before the first call.
|
|
292
|
+
// date_utc (nowIso, below) is stamped after the last call, or passed by a batch caller.
|
|
293
|
+
const generatedAt = new Date().toISOString();
|
|
294
|
+
// spec 053: the harness, read once as generation begins; a stub run has none.
|
|
295
|
+
const harnessRead = stubEnabled() ? { harness: null } : harnessFor(surfaceForModel(modelId), trusted);
|
|
296
|
+
if (harnessRead.reason) console.error(` ! harness version not recorded: ${harnessRead.reason}; the run carries on with run.harness.version null`);
|
|
266
297
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
267
298
|
const mode = withSkill ? 'with_skill' : 'baseline';
|
|
268
299
|
// A per-case override is an operator's explicit input and still wins, for
|
|
@@ -517,8 +548,11 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
517
548
|
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
518
549
|
// preamble. Absent on a stub run: it describes a harness that did not run.
|
|
519
550
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
551
|
+
// v0.9, spec 053: the harness that answered; absent on a stub run.
|
|
552
|
+
harness: surface !== 'stub' && harnessRead.harness ? harnessRead.harness : undefined,
|
|
520
553
|
runner_version: RUNNER_VERSION,
|
|
521
554
|
date_utc: nowIso,
|
|
555
|
+
generated_at: generatedAt,
|
|
522
556
|
registry: registryStatus(modelId),
|
|
523
557
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
524
558
|
judge: judgeBlock,
|
|
@@ -572,7 +606,8 @@ function summarizeReceipt(receipt) {
|
|
|
572
606
|
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
573
607
|
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
574
608
|
const j = receipt.run.judge || {};
|
|
575
|
-
|
|
609
|
+
// v0.8 (spec 043 AC-1): an absent count is unknown, never 1.
|
|
610
|
+
L.push(`- **judge:** ${Number.isInteger(j.samples) ? j.samples : 'unknown'} samples/case, temperature ${j.temperature == null ? 'n/a' : j.temperature} (${j.sampling || 'single'})`);
|
|
576
611
|
if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
|
|
577
612
|
L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
|
|
578
613
|
L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
|
|
@@ -609,7 +644,7 @@ function summarizeReceipt(receipt) {
|
|
|
609
644
|
// The rule the bands above are derived by, said beside them (spec 026 AC-6).
|
|
610
645
|
L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
|
|
611
646
|
L.push('');
|
|
612
|
-
L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples
|
|
647
|
+
L.push(`## Per-case (mean ± stddev over ${Number.isInteger((receipt.run.judge || {}).samples) ? receipt.run.judge.samples : 'an unknown number of'} judge samples)`);
|
|
613
648
|
L.push('');
|
|
614
649
|
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
615
650
|
L.push(`|---|---|---|---|---|`);
|
|
@@ -625,4 +660,4 @@ function summarizeReceipt(receipt) {
|
|
|
625
660
|
return L.join('\n');
|
|
626
661
|
}
|
|
627
662
|
|
|
628
|
-
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
|
663
|
+
module.exports = { runSkillOnModel, judgeCase, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
package/lib/skill.js
CHANGED
|
@@ -76,14 +76,7 @@ function loadSkill(skillDir) {
|
|
|
76
76
|
throw new Error(`no evals/evals.json found in ${dir}`);
|
|
77
77
|
}
|
|
78
78
|
const suiteRaw = JSON.parse(fs.readFileSync(suitePath, 'utf8'));
|
|
79
|
-
const cases =
|
|
80
|
-
// suite_hash is over the CORE case fields only ({id, prompt, rubric,
|
|
81
|
-
// pass_threshold}); optional annotations (checks, and the claim/grounding the
|
|
82
|
-
// gate reads straight from the raw suite) are excluded so adding them never
|
|
83
|
-
// disturbs a receipt's suite_hash. See spec/RECEIPT.md § suite.
|
|
84
|
-
const suiteHash = sha256Canonical(cases.map((c) => ({
|
|
85
|
-
id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
|
|
86
|
-
})));
|
|
79
|
+
const { cases, suiteHash } = suiteIdentity(suiteRaw);
|
|
87
80
|
|
|
88
81
|
return {
|
|
89
82
|
dir,
|
|
@@ -100,6 +93,20 @@ function loadSkill(skillDir) {
|
|
|
100
93
|
};
|
|
101
94
|
}
|
|
102
95
|
|
|
96
|
+
// A suite's normalised cases and its hash, as a receipt records them. suite_hash
|
|
97
|
+
// is over the CORE case fields only ({id, prompt, rubric, pass_threshold});
|
|
98
|
+
// optional annotations (checks, and the claim/grounding the gate reads straight
|
|
99
|
+
// from the raw suite) are excluded so adding them never disturbs a receipt's
|
|
100
|
+
// suite_hash. See spec/RECEIPT.md § suite. Exported for `driftproof stale
|
|
101
|
+
// --suite` (spec 053), so a suite file is hashed by the same code as a run.
|
|
102
|
+
function suiteIdentity(suiteRaw) {
|
|
103
|
+
const cases = normalizeCases(suiteRaw);
|
|
104
|
+
const suiteHash = sha256Canonical(cases.map((c) => ({
|
|
105
|
+
id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
|
|
106
|
+
})));
|
|
107
|
+
return { cases, suiteHash };
|
|
108
|
+
}
|
|
109
|
+
|
|
103
110
|
// Pull name/version from a YAML-ish front-matter block, else from the H1 line.
|
|
104
111
|
function parseSkillMeta(md) {
|
|
105
112
|
const meta = {};
|
|
@@ -128,8 +135,18 @@ function normalizeCases(raw) {
|
|
|
128
135
|
if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
|
|
129
136
|
// Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
|
|
130
137
|
if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
|
|
138
|
+
// Spec 050 AC-1: normalized ids are unique. The receipt and every reader of it key a
|
|
139
|
+
// case by this string, so two cases that share it are measured, paid for, and then
|
|
140
|
+
// one is silently dropped from the verdict. Refused here, before any call. The
|
|
141
|
+
// comparison is on the id as assigned, including a `case-<n>` given to an unnamed
|
|
142
|
+
// case, which an explicit id can collide with.
|
|
143
|
+
const firstAt = new Map();
|
|
131
144
|
return list.map((c, i) => {
|
|
132
145
|
const id = String(c.id || c.name || `case-${i + 1}`);
|
|
146
|
+
if (firstAt.has(id)) {
|
|
147
|
+
throw Object.assign(new Error(`case ids must be unique: cases ${firstAt.get(id)} and ${i + 1} both have the id ${JSON.stringify(id)}; rename one and run again`), { code: 'DUPLICATE_CASE_ID' });
|
|
148
|
+
}
|
|
149
|
+
firstAt.set(id, i + 1);
|
|
133
150
|
const prompt = c.prompt || c.input || c.task;
|
|
134
151
|
const rubric = c.rubric || c.criteria || c.expected;
|
|
135
152
|
if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
|
|
@@ -146,4 +163,4 @@ function normalizeCases(raw) {
|
|
|
146
163
|
});
|
|
147
164
|
}
|
|
148
165
|
|
|
149
|
-
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
|
|
166
|
+
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound, suiteIdentity };
|
package/lib/stale.js
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// `driftproof stale` (spec 053): for each receipt, does its conclusion still stand under what would
|
|
5
|
+
// run today, or does it need a rerun of one or both arms, or only a regrade?
|
|
6
|
+
//
|
|
7
|
+
// What a receipt recorded is read by lib/reuse.js provenanceOf; what would run today is built here,
|
|
8
|
+
// from the same places `driftproof run` reads (flags, then .driftproofrc; a value nobody set is
|
|
9
|
+
// unknown, never a default); the two are compared by lib/reuse.js compareProvenance, the function
|
|
10
|
+
// `triage()` uses, so the command and the triage cannot disagree about an axis.
|
|
11
|
+
//
|
|
12
|
+
// No model call and no network. The one thing spawned is `claude --version` or `codex --version`,
|
|
13
|
+
// and only when neither --harness-version nor --no-harness-check is given.
|
|
14
|
+
//
|
|
15
|
+
// THE OUTPUT IS A PUBLIC CONTRACT (spec/stale.v1.schema.json, `driftproof.stale/1`): fields are
|
|
16
|
+
// only ever added. It carries no clock, so the same inputs print the same bytes.
|
|
17
|
+
|
|
18
|
+
const fs = require('fs');
|
|
19
|
+
const path = require('path');
|
|
20
|
+
const { spawnSync } = require('child_process');
|
|
21
|
+
const { provenanceOf, compareProvenance, armDecisions } = require('./reuse');
|
|
22
|
+
const { loadSkill, suiteIdentity } = require('./skill');
|
|
23
|
+
const { rubricHash, promptTemplateHash } = require('./judge');
|
|
24
|
+
const { resolveModel } = require('./provider');
|
|
25
|
+
const { loadRegistry } = require('./models');
|
|
26
|
+
const { validateReceipt, verifyReceiptHash } = require('./receipt');
|
|
27
|
+
const { RUNNER_VERSION, PROJECT_NAME } = require('../config');
|
|
28
|
+
|
|
29
|
+
const SCHEMA_ID = 'driftproof.stale/1';
|
|
30
|
+
const HARNESS_BIN = { 'claude-code': 'claude', codex: 'codex' };
|
|
31
|
+
const SURFACE_HARNESS = { 'claude-cli': 'claude-code', 'openai-cli': 'codex', api: 'api', 'openai-api': 'api' };
|
|
32
|
+
|
|
33
|
+
// Today's version of a harness, read from its binary on PATH (spec 053 R-5).
|
|
34
|
+
function harnessVersionNow(name) {
|
|
35
|
+
const bin = HARNESS_BIN[name];
|
|
36
|
+
if (!bin) return { version: null, why: `no way to read today's ${name} version` };
|
|
37
|
+
const r = spawnSync(bin, ['--version'], { encoding: 'utf8', timeout: 10000 });
|
|
38
|
+
if (r.error) return { version: null, why: `${bin} --version could not run: ${r.error.code || r.error.message}` };
|
|
39
|
+
if (r.status !== 0) return { version: null, why: `${bin} --version exited ${r.status}` };
|
|
40
|
+
const m = /\d+\.\d+\.\d+[0-9A-Za-z.+-]*/.exec(r.stdout || '');
|
|
41
|
+
return m ? { version: m[0] } : { version: null, why: `${bin} --version printed no version` };
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// What would run today, for one receipt. `why` carries the reason for each value that could not be found.
|
|
45
|
+
function currentProvenance(receipt, opts, cache) {
|
|
46
|
+
const why = {};
|
|
47
|
+
const rec = provenanceOf(receipt);
|
|
48
|
+
const rc = opts.rc || {};
|
|
49
|
+
// Model: --model, else the rc's models; the receipt's model is current when the rc names it.
|
|
50
|
+
let model = null;
|
|
51
|
+
if (opts.model) model = resolveModel(opts.model);
|
|
52
|
+
else if (rc.models !== undefined && rc.models !== null) {
|
|
53
|
+
const list = String(rc.models).split(',').map((x) => x.trim()).filter(Boolean).map((x) => resolveModel(x));
|
|
54
|
+
model = rec.model_id && list.includes(resolveModel(rec.model_id)) ? resolveModel(rec.model_id) : list.join(',') || null;
|
|
55
|
+
}
|
|
56
|
+
if (!model) why.model = 'no --model given and no models in .driftproofrc';
|
|
57
|
+
// Judge: --judge, else the rc's judge_model.
|
|
58
|
+
const judgeRaw = opts.judge || rc.judge_model || null;
|
|
59
|
+
const judge = judgeRaw ? resolveModel(judgeRaw) : null;
|
|
60
|
+
if (!judge) why.judge_model = 'no --judge given and no judge_model in .driftproofrc';
|
|
61
|
+
// Skill and suite.
|
|
62
|
+
let skillHash = null; let suite = null;
|
|
63
|
+
if (opts.skill) {
|
|
64
|
+
const s = cache.skill || (cache.skill = loadSkill(opts.skill));
|
|
65
|
+
skillHash = s.contentHash;
|
|
66
|
+
if (!opts.suite) suite = { suiteHash: s.suite.suiteHash, cases: s.suite.cases };
|
|
67
|
+
} else why.skill = 'no --skill given';
|
|
68
|
+
if (opts.suite) suite = cache.suite || (cache.suite = suiteIdentity(JSON.parse(fs.readFileSync(opts.suite, 'utf8'))));
|
|
69
|
+
if (!suite) { why.suite = 'no --suite or --skill given'; why.rubric = why.suite; }
|
|
70
|
+
// Harness: the receipt's harness name, else its surface's.
|
|
71
|
+
const name = (rec.harness && rec.harness.name) || SURFACE_HARNESS[(receipt.run || {}).surface] || null;
|
|
72
|
+
let harness = null;
|
|
73
|
+
if (!name) why.harness = 'the receipt names no harness and no surface with one';
|
|
74
|
+
else if (name === 'api') harness = { name: 'api', version: null };
|
|
75
|
+
else if (opts.noHarnessCheck) why.harness = 'not checked (--no-harness-check)';
|
|
76
|
+
else if (opts.harnessVersion) harness = { name, version: String(opts.harnessVersion) };
|
|
77
|
+
else {
|
|
78
|
+
const v = cache[`h:${name}`] || (cache[`h:${name}`] = harnessVersionNow(name));
|
|
79
|
+
harness = { name, version: v.version };
|
|
80
|
+
if (v.version == null) why.harness = v.why;
|
|
81
|
+
}
|
|
82
|
+
const rubrics = suite ? Object.fromEntries(suite.cases.map((c) => [c.id, rubricHash(c.rubric)])) : null;
|
|
83
|
+
return {
|
|
84
|
+
prov: {
|
|
85
|
+
model_id: model, harness, skill_hash: skillHash, suite_hash: suite ? suite.suiteHash : null,
|
|
86
|
+
case_ids: suite ? [...new Set(suite.cases.map((c) => c.id))].sort() : null,
|
|
87
|
+
judge_model_id: judge, judge_template_hash: promptTemplateHash(), rubrics,
|
|
88
|
+
},
|
|
89
|
+
why,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// The newer-model advisory (spec 053 R-9): same family, registered, released after both the run's
|
|
94
|
+
// date and the recorded model's release date. No advisory when either date is missing.
|
|
95
|
+
const dated = (m) => !!m && typeof m.released === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(m.released);
|
|
96
|
+
function advisories(receipt) {
|
|
97
|
+
const reg = loadRegistry();
|
|
98
|
+
const id = (receipt.run || {}).model_id;
|
|
99
|
+
const row = id ? reg.byId[resolveModel(id)] || reg.byId[id] : null;
|
|
100
|
+
const runDay = typeof (receipt.run || {}).date_utc === 'string' ? receipt.run.date_utc.slice(0, 10) : null;
|
|
101
|
+
if (!dated(row) || !runDay) return [];
|
|
102
|
+
return reg.models
|
|
103
|
+
.filter((m) => m.id !== row.id && m.family === row.family && !m.auto_added && dated(m) && String(m.released) > runDay && String(m.released) > String(row.released))
|
|
104
|
+
.sort((a, b) => (a.released === b.released ? (a.id < b.id ? -1 : 1) : (a.released < b.released ? 1 : -1)))
|
|
105
|
+
.map((m) => ({ model_id: m.id, released: m.released }));
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// The command to run next (spec 053 R-7).
|
|
109
|
+
function nextCommand(file, receipt, arms, opts, current) {
|
|
110
|
+
const skillArg = opts.skill || '<skill dir>';
|
|
111
|
+
const d = [arms.with_skill.decision, arms.baseline.decision];
|
|
112
|
+
if (d.includes('rerun')) return `${PROJECT_NAME} run ${skillArg} --model ${current.model_id && !current.model_id.includes(',') ? current.model_id : receipt.run.model_id}`;
|
|
113
|
+
if (d.includes('regrade')) return `${PROJECT_NAME} regrade ${file} --skill ${skillArg} --answers <answers.json> --judge-model ${current.judge_model_id || receipt.run.judge.model_id}`;
|
|
114
|
+
return null;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function assess(file, receipt, opts, cache) {
|
|
118
|
+
const { prov, why } = currentProvenance(receipt, opts, cache);
|
|
119
|
+
const axes = compareProvenance(provenanceOf(receipt), prov, why);
|
|
120
|
+
const arms = armDecisions(axes);
|
|
121
|
+
const d = [arms.with_skill.decision, arms.baseline.decision];
|
|
122
|
+
const status = d.includes('rerun') || d.includes('regrade') ? 'stale' : d.includes('unknown') ? 'unknown' : 'current';
|
|
123
|
+
const run = receipt.run || {};
|
|
124
|
+
return {
|
|
125
|
+
path: file,
|
|
126
|
+
receipt_hash: receipt.receipt_hash || null,
|
|
127
|
+
skill: (receipt.skill && receipt.skill.name) || null,
|
|
128
|
+
model_id: run.model_id || null,
|
|
129
|
+
date_utc: run.date_utc || null,
|
|
130
|
+
verification_level: receipt.verification_level || null,
|
|
131
|
+
status,
|
|
132
|
+
arms,
|
|
133
|
+
axes,
|
|
134
|
+
advisories: advisories(receipt),
|
|
135
|
+
runner_version: { recorded: run.runner_version || null, current: RUNNER_VERSION },
|
|
136
|
+
next: nextCommand(file, receipt, arms, opts, prov),
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// The whole report: every receipt read, validated and assessed, and the exit code (spec 053 R-8).
|
|
141
|
+
function staleReport(files, opts = {}) {
|
|
142
|
+
const cache = {};
|
|
143
|
+
const receipts = files.map((file) => {
|
|
144
|
+
let r;
|
|
145
|
+
try { r = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { return { path: file, error: e.code === 'ENOENT' ? 'the file does not exist' : `the file is not a readable JSON document: ${e.message}` }; }
|
|
146
|
+
const v = validateReceipt(r);
|
|
147
|
+
if (!v.valid) return { path: file, error: `the receipt does not validate: ${(v.errors || []).slice(0, 2).map((x) => (typeof x === 'string' ? x : `${x.instancePath || ''} ${x.message || ''}`.trim())).join('; ')}` };
|
|
148
|
+
// As `driftproof validate` reads a receipt: the schema, and the hash over its content.
|
|
149
|
+
if (!verifyReceiptHash(r)) return { path: file, error: 'the receipt_hash does not match the receipt\'s content' };
|
|
150
|
+
try { return assess(file, r, opts, cache); } catch (e) { return { path: file, error: `the receipt could not be assessed: ${e.message}` }; }
|
|
151
|
+
});
|
|
152
|
+
const ok = receipts.filter((x) => !x.error);
|
|
153
|
+
// Advisories: a newer model, and an axis that moved by a patch release only (A-053-1).
|
|
154
|
+
const summary = { receipts: receipts.length, current: ok.filter((x) => x.status === 'current').length, stale: ok.filter((x) => x.status === 'stale').length, unknown: ok.filter((x) => x.status === 'unknown').length, errors: receipts.length - ok.length, advisories: ok.reduce((n, x) => n + x.advisories.length + x.axes.filter((e) => e.effect === 'advisory').length, 0) };
|
|
155
|
+
let exit = summary.errors ? 2 : summary.stale ? 1 : summary.unknown || summary.advisories ? 3 : 0;
|
|
156
|
+
if (opts.strict && exit === 3) exit = 1;
|
|
157
|
+
return { schema: SCHEMA_ID, runner_version: RUNNER_VERSION, strict: !!opts.strict, receipts, summary, exit_code: exit };
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// ── the human output ─────────────────────────────────────────────────────────
|
|
161
|
+
const LABEL = { model: 'model', harness: 'harness', skill: 'skill', suite: 'suite', judge_model: 'judge model', judge_template: 'judge template', rubric: 'rubric' };
|
|
162
|
+
const short = (v) => (typeof v === 'string' && /^[0-9a-f]{64}$/.test(v) ? v.slice(0, 8) : v == null ? 'unrecorded' : String(v));
|
|
163
|
+
function effectText(e) {
|
|
164
|
+
if (e.effect === 'unknown') return `unknown: ${e.reason}`;
|
|
165
|
+
if (e.effect === 'advisory') return 'a patch release only (advisory)';
|
|
166
|
+
if (e.effect === 'regrade') return e.cases ? `regrade case(s) ${e.cases.join(', ')}` : 'regrade both arms';
|
|
167
|
+
if (e.effect === 'rerun') {
|
|
168
|
+
if (e.arms.length === 1) return 'rerun the with-skill arm; baseline still valid';
|
|
169
|
+
const extra = [e.cases_added && e.cases_added.length ? `added ${e.cases_added.join(', ')}` : '', e.cases_removed && e.cases_removed.length ? `removed ${e.cases_removed.join(', ')}` : '', e.cases ? `rubric moved on ${e.cases.join(', ')}` : ''].filter(Boolean).join('; ');
|
|
170
|
+
return `rerun both arms${extra ? ` (${extra})` : ''}`;
|
|
171
|
+
}
|
|
172
|
+
return 'unchanged';
|
|
173
|
+
}
|
|
174
|
+
function renderText(doc) {
|
|
175
|
+
const L = [];
|
|
176
|
+
for (const x of doc.receipts) {
|
|
177
|
+
if (x.error) { L.push(`${x.path} ERROR`, ` ${x.error}`, ''); continue; }
|
|
178
|
+
L.push(`${x.skill || '(no skill)'} ${x.model_id || '(no model)'} ${x.date_utc ? String(x.date_utc).slice(0, 10) : '(no date)'} ${x.status.toUpperCase()}`);
|
|
179
|
+
for (const e of x.axes) {
|
|
180
|
+
if (e.effect === 'current' && e.axis !== 'judge_model') continue;
|
|
181
|
+
const now = e.current == null ? 'not known' : short(e.current);
|
|
182
|
+
const move = e.effect === 'current' ? `${now}, unchanged` : `${short(e.recorded)} to ${now}`;
|
|
183
|
+
// Two spaces at least between the move and its effect, however long the move (approval F-1).
|
|
184
|
+
L.push(` ${LABEL[e.axis].padEnd(14)}${(move + ' ').padEnd(38)}${e.effect === 'current' ? '' : effectText(e)}`.trimEnd());
|
|
185
|
+
}
|
|
186
|
+
for (const a of x.advisories) L.push(` ${'newer model'.padEnd(14)}${a.model_id}, released ${a.released} (advisory)`);
|
|
187
|
+
L.push(` ${'next'.padEnd(14)}${x.next || (x.status === 'current' ? 'nothing to run' : 'nothing can be decided until the unknown axes are known')}`, '');
|
|
188
|
+
}
|
|
189
|
+
const s = doc.summary;
|
|
190
|
+
L.push(`${s.receipts} receipt(s): ${s.current} current, ${s.stale} stale, ${s.unknown} unknown, ${s.errors} error(s); exit ${doc.exit_code}${doc.strict ? ' (--strict)' : ''}`);
|
|
191
|
+
return L.join('\n') + '\n';
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
module.exports = { staleReport, renderText, currentProvenance, advisories, SCHEMA_ID };
|
package/lib/verdict.js
CHANGED
|
@@ -4,6 +4,18 @@
|
|
|
4
4
|
const crypto = require('crypto');
|
|
5
5
|
const { EFFECT_FLOOR, POWER_Z } = require('../config');
|
|
6
6
|
const { bandOf } = require('./reuse');
|
|
7
|
+
const { duplicateCaseRows, ambiguityLine } = require('./receipt');
|
|
8
|
+
|
|
9
|
+
// Spec 050 AC-3. A receipt with two rows for one (id, mode) has no verdict: which row
|
|
10
|
+
// the byId map below kept would decide it. The error carries a code so a caller that
|
|
11
|
+
// renders refusals can tell this one from a defect.
|
|
12
|
+
class AmbiguousReceiptError extends Error {
|
|
13
|
+
constructor(dups) {
|
|
14
|
+
super(`ambiguous receipt: ${ambiguityLine(dups)}; a verdict over it would depend on which row was read`);
|
|
15
|
+
this.code = 'AMBIGUOUS_RECEIPT';
|
|
16
|
+
this.duplicates = dups;
|
|
17
|
+
}
|
|
18
|
+
}
|
|
7
19
|
|
|
8
20
|
// Single-receipt verdict + shields.io badge.
|
|
9
21
|
//
|
|
@@ -156,6 +168,11 @@ function notMeasuredRoutes(level, kind, delta, incomplete) {
|
|
|
156
168
|
// lib/decision.js states. An absent field satisfies "not TESTED" and "not model":
|
|
157
169
|
// a receipt that does not say is not taken to have said the reassuring thing.
|
|
158
170
|
function receiptVerdict(receipt) {
|
|
171
|
+
// Spec 050 AC-3: refused before anything is read, whatever the level. An existing
|
|
172
|
+
// receipt written before the loader refused duplicate ids reaches here unchanged,
|
|
173
|
+
// and a surface that caught nothing must not render a verdict for it.
|
|
174
|
+
const dups = duplicateCaseRows(receipt);
|
|
175
|
+
if (dups.length) throw new AmbiguousReceiptError(dups);
|
|
159
176
|
const cmp = (receipt && receipt.comparison) || {};
|
|
160
177
|
// Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
|
|
161
178
|
// verdicted — we did not run the suite, so we do not certify the outcome.
|
|
@@ -284,4 +301,5 @@ function githubOutputLines(receipt) {
|
|
|
284
301
|
module.exports = {
|
|
285
302
|
verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS,
|
|
286
303
|
receiptVerdict, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
|
|
304
|
+
AmbiguousReceiptError,
|
|
287
305
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.11.
|
|
3
|
+
"version": "0.11.3",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|