driftproof 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -12
- package/bin/driftproof +343 -39
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +72 -31
- package/lib/diff.js +12 -3
- package/lib/export.js +6 -2
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/init.js +3 -2
- package/lib/receipt.js +182 -10
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +70 -15
- package/lib/stale.js +194 -0
- package/lib/verdict.js +50 -16
- package/package.json +1 -1
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/lib/reuse.js
CHANGED
|
@@ -21,6 +21,9 @@
|
|
|
21
21
|
// receipt's own recorded provenance — model, suite and skill content force
|
|
22
22
|
// a rerun because the generation would differ; judge and rubric force a
|
|
23
23
|
// regrade because only the scoring would; neither changing permits reuse.
|
|
24
|
+
// Spec 053 made it one comparison of two provenance records, shared with
|
|
25
|
+
// `driftproof stale` (lib/stale.js), and added the harness and the judge's
|
|
26
|
+
// template; see PROVENANCE below.
|
|
24
27
|
//
|
|
25
28
|
// PURE: two receipts in, a decision out. No I/O, no provider. F-009-K's lesson
|
|
26
29
|
// applied — the decision derives from the artifact, never from a filename.
|
|
@@ -120,15 +123,147 @@ function compare(older, newer) {
|
|
|
120
123
|
return { verdict: 'MEASURED', delta: round(b - a), baseline_reproduced: true };
|
|
121
124
|
}
|
|
122
125
|
|
|
126
|
+
// ── PROVENANCE: ONE COMPARISON FOR TRIAGE AND STALE (spec 053 AC-1, AC-2) ─────
|
|
127
|
+
//
|
|
128
|
+
// What a receipt records about how its numbers were made, and one pure function
|
|
129
|
+
// that compares two such records: two receipts (`triage`), or a receipt and what
|
|
130
|
+
// would run today (`driftproof stale`, lib/stale.js). Before spec 053 `triage`
|
|
131
|
+
// read `run.rubric_hash`, which no receipt carries (rubric hashes are per case),
|
|
132
|
+
// and never looked at the judge's template or the harness, so a rubric or
|
|
133
|
+
// template change was reused and a harness change went unseen.
|
|
134
|
+
//
|
|
135
|
+
// UNKNOWN IS NEVER CURRENT. An axis either side does not record is `unknown`,
|
|
136
|
+
// with the reason, and an arm with an unknown axis is not reused.
|
|
137
|
+
//
|
|
138
|
+
// THE SUITE HASH COVERS PROMPTS AND RUBRICS TOGETHER ({id, prompt, rubric,
|
|
139
|
+
// pass_threshold}), and no receipt records a per-case prompt hash. So a
|
|
140
|
+
// different suite reruns both arms even when only a rubric moved: the record
|
|
141
|
+
// cannot show the prompts did not. A case's rubric hash moving under an equal
|
|
142
|
+
// suite hash (the rubric hash also covers the judge's system prompt) regrades
|
|
143
|
+
// that case (spec 053 R-3).
|
|
144
|
+
|
|
145
|
+
// The axes, in the order a reader is shown them, with what a difference does
|
|
146
|
+
// and to which arms.
|
|
147
|
+
const AXES = [
|
|
148
|
+
{ axis: 'model', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the model differs, so the generation would differ' },
|
|
149
|
+
{ axis: 'harness', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the harness differs, so the generation would differ' },
|
|
150
|
+
{ axis: 'skill', effect: 'rerun', arms: ['with_skill'], why: 'the skill content differs, so the with-skill generation would differ; the baseline arm contains no skill text and still stands' },
|
|
151
|
+
{ axis: 'suite', effect: 'rerun', arms: ['with_skill', 'baseline'], why: 'the suite differs, so the prompts may differ' },
|
|
152
|
+
{ axis: 'judge_model', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'the judge model differs, so the existing generations can be graded again' },
|
|
153
|
+
{ axis: 'judge_template', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'the judge prompt template differs, so the existing generations can be graded again' },
|
|
154
|
+
{ axis: 'rubric', effect: 'regrade', arms: ['with_skill', 'baseline'], why: 'a case\'s rubric hash differs under the same suite, so that case can be graded again' },
|
|
155
|
+
];
|
|
156
|
+
const UNRECORDED = { model: 'no model id', harness: 'no harness (receipts before v0.9 carry none)', skill: 'no skill content hash', suite: 'no suite hash', judge_model: 'no judge model', judge_template: 'no judge prompt template hash (receipts before v0.6 carry none)', rubric: 'no per-case rubric hash' };
|
|
157
|
+
|
|
158
|
+
const canonicalId = (id) => (id == null ? null : String(id).replace(/-\d{8}$/, ''));
|
|
159
|
+
|
|
160
|
+
// THE HARNESS BY SEMVER (spec 053 A-053-1, the operator's instruction of 24 Sep 2026, a departure
|
|
161
|
+
// from the brief's exact match). A major or minor change (2.1 to 2.2) reruns both arms; a patch-only
|
|
162
|
+
// change (2.1.281 to 2.1.282) is an advisory. Claude Code ships a patch release every few days, and
|
|
163
|
+
// an exact match would mark every receipt stale within days. A version that is not semver, or a
|
|
164
|
+
// different harness, must match exactly.
|
|
165
|
+
const semver = (v) => { const m = /^(\d+)\.(\d+)\.(\d+)(.*)$/.exec(String(v)); return m ? [m[1], m[2], m[3], m[4]] : null; };
|
|
166
|
+
function harnessDrift(a, b) {
|
|
167
|
+
if (a.name !== b.name) return 'moved';
|
|
168
|
+
if (a.version === b.version) return 'same';
|
|
169
|
+
const x = semver(a.version); const y = semver(b.version);
|
|
170
|
+
if (!x || !y) return 'moved';
|
|
171
|
+
if (x[0] === y[0] && x[1] === y[1]) return x[2] === y[2] && x[3] === y[3] ? 'same' : 'patch';
|
|
172
|
+
return 'moved';
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// The provenance a receipt records. Every field is null when the receipt does
|
|
176
|
+
// not record it; an imported receipt's `unknown` model is null.
|
|
177
|
+
function provenanceOf(r) {
|
|
178
|
+
const run = (r && r.run) || {};
|
|
179
|
+
const judge = run.judge || {};
|
|
180
|
+
const cases = (r && r.results && Array.isArray(r.results.cases)) ? r.results.cases : [];
|
|
181
|
+
const rubrics = {};
|
|
182
|
+
for (const c of cases) if (c && c.judge && typeof c.judge.rubric_hash === 'string' && !(c.id in rubrics)) rubrics[c.id] = c.judge.rubric_hash;
|
|
183
|
+
const firstCaseJudge = (cases.find((c) => c && c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
184
|
+
return {
|
|
185
|
+
model_id: run.model_id && run.model_id !== 'unknown' ? run.model_id : null,
|
|
186
|
+
harness: run.harness && typeof run.harness.name === 'string' ? { name: run.harness.name, version: run.harness.version == null ? null : String(run.harness.version) } : null,
|
|
187
|
+
skill_hash: (r && r.skill && r.skill.content_hash) || null,
|
|
188
|
+
suite_hash: (r && r.suite && r.suite.suite_hash) || null,
|
|
189
|
+
case_ids: cases.length ? [...new Set(cases.map((c) => c.id))].sort() : null,
|
|
190
|
+
judge_model_id: judge.model_id || firstCaseJudge || null,
|
|
191
|
+
judge_template_hash: judge.prompt_template_hash || null,
|
|
192
|
+
rubrics: Object.keys(rubrics).length ? rubrics : null,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const show = (axis, v) => {
|
|
197
|
+
if (v == null) return null;
|
|
198
|
+
if (axis === 'harness') return v.version == null ? v.name : `${v.name} ${v.version}`;
|
|
199
|
+
if (axis === 'rubric' && typeof v === 'object') return `${Object.keys(v).length} case(s)`;
|
|
200
|
+
return v;
|
|
201
|
+
};
|
|
202
|
+
|
|
203
|
+
// One entry per axis: { axis, recorded, current, effect, arms, reason[, cases, cases_added, cases_removed] }.
|
|
204
|
+
// `why` optionally carries a reason for a current value that could not be found
|
|
205
|
+
// (lib/stale.js: "no --skill given", "claude --version failed: ...").
|
|
206
|
+
function compareProvenance(rec, cur, why = {}) {
|
|
207
|
+
const out = [];
|
|
208
|
+
const entry = (a, recorded, current, same, extra = {}) => {
|
|
209
|
+
const unknownSide = recorded == null ? `the receipt records ${UNRECORDED[a.axis]}` : current == null ? (why[a.axis] || `the other side records ${UNRECORDED[a.axis]}`) : null;
|
|
210
|
+
if (unknownSide) return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: 'unknown', arms: a.arms, reason: unknownSide, ...extra });
|
|
211
|
+
if (same) return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: 'current', arms: a.arms, reason: 'unchanged', ...extra });
|
|
212
|
+
return out.push({ axis: a.axis, recorded: show(a.axis, recorded), current: show(a.axis, current), effect: a.effect, arms: a.arms, reason: a.why, ...extra });
|
|
213
|
+
};
|
|
214
|
+
const A = Object.fromEntries(AXES.map((a) => [a.axis, a]));
|
|
215
|
+
entry(A.model, rec.model_id, cur.model_id, canonicalId(rec.model_id) === canonicalId(cur.model_id));
|
|
216
|
+
// A harness with no version recorded is unknown, except an API surface, which has no harness to drift.
|
|
217
|
+
const h = (x) => (x && (x.version != null || x.name === 'api') ? x : null);
|
|
218
|
+
const drift = h(rec.harness) && h(cur.harness) ? harnessDrift(rec.harness, cur.harness) : null;
|
|
219
|
+
if (drift === 'patch') out.push({ axis: 'harness', recorded: show('harness', rec.harness), current: show('harness', cur.harness), effect: 'advisory', arms: A.harness.arms, reason: 'a patch release of the same harness: an advisory, not a rerun (spec 053 A-053-1)' });
|
|
220
|
+
else entry(A.harness, h(rec.harness), h(cur.harness), drift === 'same');
|
|
221
|
+
entry(A.skill, rec.skill_hash, cur.skill_hash, rec.skill_hash === cur.skill_hash);
|
|
222
|
+
const added = rec.case_ids && cur.case_ids ? cur.case_ids.filter((id) => !rec.case_ids.includes(id)) : [];
|
|
223
|
+
const removed = rec.case_ids && cur.case_ids ? rec.case_ids.filter((id) => !cur.case_ids.includes(id)) : [];
|
|
224
|
+
entry(A.suite, rec.suite_hash, cur.suite_hash, rec.suite_hash === cur.suite_hash, added.length || removed.length ? { cases_added: added, cases_removed: removed } : {});
|
|
225
|
+
entry(A.judge_model, rec.judge_model_id, cur.judge_model_id, canonicalId(rec.judge_model_id) === canonicalId(cur.judge_model_id));
|
|
226
|
+
entry(A.judge_template, rec.judge_template_hash, cur.judge_template_hash, rec.judge_template_hash === cur.judge_template_hash);
|
|
227
|
+
// The rubric, per case, over the cases both sides record.
|
|
228
|
+
if (rec.rubrics && cur.rubrics) {
|
|
229
|
+
const moved = Object.keys(rec.rubrics).filter((id) => id in cur.rubrics && rec.rubrics[id] !== cur.rubrics[id]).sort();
|
|
230
|
+
const suiteMoved = out.find((e) => e.axis === 'suite').effect === 'rerun';
|
|
231
|
+
if (!moved.length) entry(A.rubric, 'recorded', 'recorded', true);
|
|
232
|
+
else if (suiteMoved) out.push({ axis: 'rubric', recorded: `${moved.length} case(s)`, current: `${moved.length} case(s)`, effect: 'rerun', arms: A.rubric.arms, cases: moved, reason: 'the rubric moved on these cases, and the suite differs, so the rerun the suite needs grades them again; the receipt records no per-case prompt hash, so a rubric edit cannot be told from a prompt edit' });
|
|
233
|
+
else out.push({ axis: 'rubric', recorded: `${moved.length} case(s)`, current: `${moved.length} case(s)`, effect: 'regrade', arms: A.rubric.arms, cases: moved, reason: A.rubric.why });
|
|
234
|
+
} else entry(A.rubric, rec.rubrics, cur.rubrics, false);
|
|
235
|
+
return out;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
// Each arm's decision: rerun, then regrade, then unknown, then reuse.
|
|
239
|
+
const RANK = { rerun: 3, regrade: 2, unknown: 1, current: 0, advisory: 0 };
|
|
240
|
+
function armDecisions(axes) {
|
|
241
|
+
const arms = {};
|
|
242
|
+
for (const arm of ['with_skill', 'baseline']) {
|
|
243
|
+
const on = axes.filter((e) => e.arms.includes(arm));
|
|
244
|
+
const top = on.reduce((m, e) => (RANK[e.effect] > RANK[m] ? e.effect : m), 'current');
|
|
245
|
+
const decision = RANK[top] === 0 ? 'reuse' : top;
|
|
246
|
+
const because = on.filter((e) => e.effect === top && decision !== 'reuse');
|
|
247
|
+
const cases = decision === 'regrade' && because.every((e) => Array.isArray(e.cases)) ? [...new Set(because.flatMap((e) => e.cases))].sort() : null;
|
|
248
|
+
arms[arm] = { decision, axes: because.map((e) => e.axis), ...(cases ? { cases } : {}) };
|
|
249
|
+
}
|
|
250
|
+
return arms;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Two receipts' provenance compared (spec 053 AC-2). The decision is the most
|
|
254
|
+
// consequential arm's: rerun, then regrade, then unknown, then reuse. `cases`
|
|
255
|
+
// names the cases a rubric regrade reaches.
|
|
123
256
|
function triage(a, b) {
|
|
124
|
-
const
|
|
125
|
-
const
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
257
|
+
const axes = compareProvenance(provenanceOf(a), provenanceOf(b));
|
|
258
|
+
const arms = armDecisions(axes);
|
|
259
|
+
const decision = ['rerun', 'regrade', 'unknown'].find((d) => arms.with_skill.decision === d || arms.baseline.decision === d) || 'reuse';
|
|
260
|
+
const first = axes.find((e) => e.effect === decision);
|
|
261
|
+
const advised = axes.filter((e) => e.effect === 'advisory').map((e) => e.axis);
|
|
262
|
+
const reason = decision === 'reuse'
|
|
263
|
+
? `model, harness, skill, suite, judge, template and rubric all match${advised.length ? ` but for a patch release of the ${advised.join(', ')} (an advisory)` : ''}, so the earlier result stands`
|
|
264
|
+
: decision === 'unknown' ? `${first.axis} is unknown: ${first.reason}` : first.reason;
|
|
265
|
+
const cases = decision === 'regrade' ? (arms.with_skill.cases || arms.baseline.cases || null) : null;
|
|
266
|
+
return { decision, reason, arms, axes, ...(cases ? { cases } : {}) };
|
|
132
267
|
}
|
|
133
268
|
|
|
134
|
-
module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf };
|
|
269
|
+
module.exports = { compare, baselineReproduces, triage, REFUSAL_REASONS, bandOf, provenanceOf, compareProvenance, armDecisions, AXES };
|
package/lib/run.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
-
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE } = require('./provider');
|
|
4
|
+
const { complete, resolveModel, surfaceForModel, isMeteredSurface, retryPolicyForSurface, CODEX_OVERHEAD_NOTE, buildSpawnPlan } = require('./provider');
|
|
5
|
+
const { spawnSync } = require('child_process');
|
|
5
6
|
const { gradeSamples, judgeSettings, promptTemplateHash, isTruncated } = require('./judge');
|
|
6
7
|
const { buildReceipt, caseFailed, FAILED_STATUSES } = require('./receipt');
|
|
7
8
|
const { sha256 } = require('./canonical');
|
|
@@ -32,6 +33,30 @@ const MODEL_RELEASE_DATES = {
|
|
|
32
33
|
'claude-opus-4-8': '2026-05-28', // anthropic.com/news/claude-opus-4-8
|
|
33
34
|
};
|
|
34
35
|
|
|
36
|
+
// The harness that will answer, recorded when a run starts (spec 053 AC-8, R-4): claude-code on
|
|
37
|
+
// claude-cli, codex on openai-cli, each version read from `<bin> --version` through the same spawn
|
|
38
|
+
// plan the run's calls use, so it is the binary that answers; "api" with no version on an API
|
|
39
|
+
// surface. A version that cannot be read is null and the reason is returned for the caller to print;
|
|
40
|
+
// it never stops a run. The receipt has no field for the reason (receipt v0.9's run.harness is
|
|
41
|
+
// {name, version}).
|
|
42
|
+
const HARNESS_OF = { 'claude-cli': ['claude-code', 'claude'], 'openai-cli': ['codex', 'codex'] };
|
|
43
|
+
function harnessFor(surface, trusted) {
|
|
44
|
+
if (surface === 'api' || surface === 'openai-api') return { harness: { name: 'api', version: null } };
|
|
45
|
+
const h = HARNESS_OF[surface];
|
|
46
|
+
if (!h) return { harness: null };
|
|
47
|
+
const [name, bin] = h;
|
|
48
|
+
try {
|
|
49
|
+
const plan = buildSpawnPlan({ bin, args: ['--version'], trusted });
|
|
50
|
+
const r = spawnSync(plan.file, plan.args, { env: plan.env, encoding: 'utf8', timeout: 30000 });
|
|
51
|
+
if (r.error) return { harness: { name, version: null }, reason: `${bin} --version could not run: ${r.error.code || r.error.message}` };
|
|
52
|
+
if (r.status !== 0) return { harness: { name, version: null }, reason: `${bin} --version exited ${r.status}` };
|
|
53
|
+
const m = /\d+\.\d+\.\d+[0-9A-Za-z.+-]*/.exec(r.stdout || '');
|
|
54
|
+
return m ? { harness: { name, version: m[0] } } : { harness: { name, version: null }, reason: `${bin} --version printed no version` };
|
|
55
|
+
} catch (e) {
|
|
56
|
+
return { harness: { name, version: null }, reason: `${bin} --version could not be planned: ${e.message}` };
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
35
60
|
function releaseDateFor(modelId) {
|
|
36
61
|
if (MODEL_RELEASE_DATES[modelId]) return MODEL_RELEASE_DATES[modelId];
|
|
37
62
|
// Derive from a trailing YYYYMMDD in the id if present.
|
|
@@ -263,6 +288,12 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
263
288
|
const reportedAll = new Set();
|
|
264
289
|
const isTimeout = (e) => !!(e && (e.code === 'TIMEOUT' || /tim(e|ed)\s*out/i.test(String((e && e.message) || ''))));
|
|
265
290
|
const genKind = (ws) => (ws ? 'gen_with_skill' : 'gen_baseline');
|
|
291
|
+
// v0.8 (spec 043 AC-4): when generation began, per receipt, before the first call.
|
|
292
|
+
// date_utc (nowIso, below) is stamped after the last call, or passed by a batch caller.
|
|
293
|
+
const generatedAt = new Date().toISOString();
|
|
294
|
+
// spec 053: the harness, read once as generation begins; a stub run has none.
|
|
295
|
+
const harnessRead = stubEnabled() ? { harness: null } : harnessFor(surfaceForModel(modelId), trusted);
|
|
296
|
+
if (harnessRead.reason) console.error(` ! harness version not recorded: ${harnessRead.reason}; the run carries on with run.harness.version null`);
|
|
266
297
|
const pairs = await mapPool(tasks, concurrency, async ({ c, withSkill }) => {
|
|
267
298
|
const mode = withSkill ? 'with_skill' : 'baseline';
|
|
268
299
|
// A per-case override is an operator's explicit input and still wins, for
|
|
@@ -517,8 +548,11 @@ async function runSkillOnModel({ skill, model, opts = {} }) {
|
|
|
517
548
|
// v0.3.1: on the openai/cli (codex) surface, record the fixed harness
|
|
518
549
|
// preamble. Absent on a stub run: it describes a harness that did not run.
|
|
519
550
|
surface_overhead_note: surface === 'openai-cli' ? CODEX_OVERHEAD_NOTE : undefined,
|
|
551
|
+
// v0.9, spec 053: the harness that answered; absent on a stub run.
|
|
552
|
+
harness: surface !== 'stub' && harnessRead.harness ? harnessRead.harness : undefined,
|
|
520
553
|
runner_version: RUNNER_VERSION,
|
|
521
554
|
date_utc: nowIso,
|
|
555
|
+
generated_at: generatedAt,
|
|
522
556
|
registry: registryStatus(modelId),
|
|
523
557
|
transcripts: keepTranscripts ? 'retained-local' : 'hashes-only',
|
|
524
558
|
judge: judgeBlock,
|
|
@@ -572,7 +606,8 @@ function summarizeReceipt(receipt) {
|
|
|
572
606
|
L.push(`- **run (UTC):** ${receipt.run.date_utc}`);
|
|
573
607
|
L.push(`- **runner:** v${receipt.run.runner_version}`);
|
|
574
608
|
const j = receipt.run.judge || {};
|
|
575
|
-
|
|
609
|
+
// v0.8 (spec 043 AC-1): an absent count is unknown, never 1.
|
|
610
|
+
L.push(`- **judge:** ${Number.isInteger(j.samples) ? j.samples : 'unknown'} samples/case, temperature ${j.temperature == null ? 'n/a' : j.temperature} (${j.sampling || 'single'})`);
|
|
576
611
|
if (receipt.run.registry) L.push(`- **registry:** ${receipt.run.registry} **transcripts:** ${receipt.run.transcripts || 'hashes-only'}`);
|
|
577
612
|
L.push(`- **skill content_hash:** \`${receipt.skill.content_hash.slice(0, 16)}…\``);
|
|
578
613
|
L.push(`- **suite:** ${receipt.suite.case_count} cases (${receipt.suite.format})`);
|
|
@@ -609,7 +644,7 @@ function summarizeReceipt(receipt) {
|
|
|
609
644
|
// The rule the bands above are derived by, said beside them (spec 026 AC-6).
|
|
610
645
|
L.push(`band rule: each arm's band is the sample stddev of its per-case means${aggs.band_rule ? ` — ${aggs.band_rule}` : ' (unstated on this pre-v0.6 receipt)'}`);
|
|
611
646
|
L.push('');
|
|
612
|
-
L.push(`## Per-case (mean ± stddev over ${(receipt.run.judge || {}).samples
|
|
647
|
+
L.push(`## Per-case (mean ± stddev over ${Number.isInteger((receipt.run.judge || {}).samples) ? receipt.run.judge.samples : 'an unknown number of'} judge samples)`);
|
|
613
648
|
L.push('');
|
|
614
649
|
L.push(`| case | mode | outcome | mean ± stddev | judge reason |`);
|
|
615
650
|
L.push(`|---|---|---|---|---|`);
|
|
@@ -625,4 +660,4 @@ function summarizeReceipt(receipt) {
|
|
|
625
660
|
return L.join('\n');
|
|
626
661
|
}
|
|
627
662
|
|
|
628
|
-
module.exports = { runSkillOnModel, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
|
663
|
+
module.exports = { runSkillOnModel, judgeCase, summarizeReceipt, releaseDateFor, projectCalls, outcomeFor, resolveCallTimeoutMs, answeredLine, answeredByOf, attest, band, uncertaintyStr };
|
package/lib/skill.js
CHANGED
|
@@ -30,23 +30,57 @@ function pastBound(bound, limit, value, what) {
|
|
|
30
30
|
const IGNORE_DIRS = new Set(['.git', 'node_modules', 'evals']);
|
|
31
31
|
const IGNORE_FILES = new Set(['.DS_Store']);
|
|
32
32
|
|
|
33
|
+
// Spec 062 (register row 5): A LINK IS REFUSED WHERE THE SKILL IS READ. Until this SKILL.md was
|
|
34
|
+
// read through a link, so a skill directory could make any file the operator can read into the
|
|
35
|
+
// system prompt, while the walk skipped links and content_hash covered none of the bytes used
|
|
36
|
+
// (the SHA-256 of empty input, unchanged when the target changed). A link is neither followed nor
|
|
37
|
+
// skipped: the skill is refused, naming it. The skill directory itself may be reached by a link;
|
|
38
|
+
// what is under it may not.
|
|
39
|
+
function linkRefusal(rel, base) {
|
|
40
|
+
const e = new Error(`${rel} under ${base} is a symbolic link; a skill is read only from its own directory, so a link is refused rather than followed or skipped`);
|
|
41
|
+
e.code = 'SKILL_LINK'; e.path = rel;
|
|
42
|
+
return e;
|
|
43
|
+
}
|
|
44
|
+
// Every component from the skill directory down to `rel` is a plain entry, not a link. An absent
|
|
45
|
+
// component is left to the caller, whose own message names what is missing.
|
|
46
|
+
function assertNoLink(dir, rel) {
|
|
47
|
+
let cur = dir;
|
|
48
|
+
for (const part of rel.split('/')) {
|
|
49
|
+
cur = path.join(cur, part);
|
|
50
|
+
let st;
|
|
51
|
+
try { st = fs.lstatSync(cur); } catch (_e) { return; }
|
|
52
|
+
if (st.isSymbolicLink()) throw linkRefusal(path.relative(dir, cur), dir);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
33
56
|
// Bounded (spec 026 AC-13): the walk stops at SKILL_MAX_DEPTH directories
|
|
34
57
|
// below the skill dir, SKILL_MAX_FILES bundled files, and SKILL_MAX_BYTES of
|
|
35
58
|
// them together, and throws naming the bound the moment one is passed, so a
|
|
36
59
|
// pathological tree is refused before its bytes are read into memory.
|
|
37
|
-
|
|
60
|
+
//
|
|
61
|
+
// `read` maps a path relative to the skill dir to bytes the caller has already read, so the hash
|
|
62
|
+
// covers those bytes and not a second read of the same path (spec 062: SKILL.md).
|
|
63
|
+
function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0, read = {}) {
|
|
38
64
|
if (depth > SKILL_MAX_DEPTH) throw pastBound('SKILL_MAX_DEPTH', SKILL_MAX_DEPTH, depth, `directory depth under ${base}`);
|
|
39
65
|
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
66
|
+
if (entry.isSymbolicLink()) {
|
|
67
|
+
// A link under an ignored name is never read, so it is left as the name is; `evals` is
|
|
68
|
+
// checked where the suite is read.
|
|
69
|
+
if (IGNORE_DIRS.has(entry.name) || IGNORE_FILES.has(entry.name)) continue;
|
|
70
|
+
throw linkRefusal(path.relative(base, path.join(dir, entry.name)), base);
|
|
71
|
+
}
|
|
40
72
|
if (entry.isDirectory()) {
|
|
41
73
|
if (IGNORE_DIRS.has(entry.name)) continue;
|
|
42
|
-
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1);
|
|
74
|
+
walkFiles(path.join(dir, entry.name), base, acc, state, depth + 1, read);
|
|
43
75
|
} else if (entry.isFile() && !IGNORE_FILES.has(entry.name)) {
|
|
44
76
|
const abs = path.join(dir, entry.name);
|
|
77
|
+
const rel = path.relative(base, abs);
|
|
45
78
|
if (acc.length + 1 > SKILL_MAX_FILES) throw pastBound('SKILL_MAX_FILES', SKILL_MAX_FILES, acc.length + 1, `bundled files under ${base}`);
|
|
46
|
-
const
|
|
79
|
+
const given = Object.hasOwn(read, rel) ? read[rel] : null;
|
|
80
|
+
const size = given ? given.length : fs.statSync(abs).size;
|
|
47
81
|
state.bytes += size;
|
|
48
82
|
if (state.bytes > SKILL_MAX_BYTES) throw pastBound('SKILL_MAX_BYTES', SKILL_MAX_BYTES, state.bytes, `bundled bytes under ${base}`);
|
|
49
|
-
acc.push({ path:
|
|
83
|
+
acc.push({ path: rel, bytes: given || fs.readFileSync(abs) });
|
|
50
84
|
}
|
|
51
85
|
}
|
|
52
86
|
return acc;
|
|
@@ -55,14 +89,17 @@ function walkFiles(dir, base = dir, acc = [], state = { bytes: 0 }, depth = 0) {
|
|
|
55
89
|
function loadSkill(skillDir) {
|
|
56
90
|
const dir = path.resolve(skillDir);
|
|
57
91
|
const skillMdPath = path.join(dir, 'SKILL.md');
|
|
92
|
+
assertNoLink(dir, 'SKILL.md');
|
|
58
93
|
if (!fs.existsSync(skillMdPath)) {
|
|
59
94
|
throw new Error(`no SKILL.md found in ${dir}`);
|
|
60
95
|
}
|
|
61
|
-
const
|
|
96
|
+
const skillMdBytes = fs.readFileSync(skillMdPath);
|
|
97
|
+
const skillMd = skillMdBytes.toString('utf8');
|
|
62
98
|
|
|
63
99
|
// Content hash over SKILL.md + every bundled file (evals/ excluded — the suite
|
|
64
100
|
// is hashed separately so a suite edit doesn't masquerade as a skill change).
|
|
65
|
-
|
|
101
|
+
// SKILL.md enters it as the bytes read above, the bytes the run is given.
|
|
102
|
+
const files = walkFiles(dir, dir, [], { bytes: 0 }, 0, { 'SKILL.md': skillMdBytes });
|
|
66
103
|
const contentHash = sha256Files(files);
|
|
67
104
|
|
|
68
105
|
// Parse skill name/version from front-matter or the first H1; fall back to dir.
|
|
@@ -72,18 +109,12 @@ function loadSkill(skillDir) {
|
|
|
72
109
|
|
|
73
110
|
// Load the eval suite.
|
|
74
111
|
const suitePath = path.join(dir, 'evals', 'evals.json');
|
|
112
|
+
assertNoLink(dir, 'evals/evals.json');
|
|
75
113
|
if (!fs.existsSync(suitePath)) {
|
|
76
114
|
throw new Error(`no evals/evals.json found in ${dir}`);
|
|
77
115
|
}
|
|
78
116
|
const suiteRaw = JSON.parse(fs.readFileSync(suitePath, 'utf8'));
|
|
79
|
-
const cases =
|
|
80
|
-
// suite_hash is over the CORE case fields only ({id, prompt, rubric,
|
|
81
|
-
// pass_threshold}); optional annotations (checks, and the claim/grounding the
|
|
82
|
-
// gate reads straight from the raw suite) are excluded so adding them never
|
|
83
|
-
// disturbs a receipt's suite_hash. See spec/RECEIPT.md § suite.
|
|
84
|
-
const suiteHash = sha256Canonical(cases.map((c) => ({
|
|
85
|
-
id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
|
|
86
|
-
})));
|
|
117
|
+
const { cases, suiteHash } = suiteIdentity(suiteRaw);
|
|
87
118
|
|
|
88
119
|
return {
|
|
89
120
|
dir,
|
|
@@ -100,6 +131,20 @@ function loadSkill(skillDir) {
|
|
|
100
131
|
};
|
|
101
132
|
}
|
|
102
133
|
|
|
134
|
+
// A suite's normalised cases and its hash, as a receipt records them. suite_hash
|
|
135
|
+
// is over the CORE case fields only ({id, prompt, rubric, pass_threshold});
|
|
136
|
+
// optional annotations (checks, and the claim/grounding the gate reads straight
|
|
137
|
+
// from the raw suite) are excluded so adding them never disturbs a receipt's
|
|
138
|
+
// suite_hash. See spec/RECEIPT.md § suite. Exported for `driftproof stale
|
|
139
|
+
// --suite` (spec 053), so a suite file is hashed by the same code as a run.
|
|
140
|
+
function suiteIdentity(suiteRaw) {
|
|
141
|
+
const cases = normalizeCases(suiteRaw);
|
|
142
|
+
const suiteHash = sha256Canonical(cases.map((c) => ({
|
|
143
|
+
id: c.id, prompt: c.prompt, rubric: c.rubric, pass_threshold: c.pass_threshold,
|
|
144
|
+
})));
|
|
145
|
+
return { cases, suiteHash };
|
|
146
|
+
}
|
|
147
|
+
|
|
103
148
|
// Pull name/version from a YAML-ish front-matter block, else from the H1 line.
|
|
104
149
|
function parseSkillMeta(md) {
|
|
105
150
|
const meta = {};
|
|
@@ -128,8 +173,18 @@ function normalizeCases(raw) {
|
|
|
128
173
|
if (!list) throw new Error('evals.json must be an array or have a `cases`/`evals` array');
|
|
129
174
|
// Bounded (spec 026 AC-13): the case count, and each prompt and rubric.
|
|
130
175
|
if (list.length > SUITE_MAX_CASES) throw pastBound('SUITE_MAX_CASES', SUITE_MAX_CASES, list.length, 'cases in the suite');
|
|
176
|
+
// Spec 050 AC-1: normalized ids are unique. The receipt and every reader of it key a
|
|
177
|
+
// case by this string, so two cases that share it are measured, paid for, and then
|
|
178
|
+
// one is silently dropped from the verdict. Refused here, before any call. The
|
|
179
|
+
// comparison is on the id as assigned, including a `case-<n>` given to an unnamed
|
|
180
|
+
// case, which an explicit id can collide with.
|
|
181
|
+
const firstAt = new Map();
|
|
131
182
|
return list.map((c, i) => {
|
|
132
183
|
const id = String(c.id || c.name || `case-${i + 1}`);
|
|
184
|
+
if (firstAt.has(id)) {
|
|
185
|
+
throw Object.assign(new Error(`case ids must be unique: cases ${firstAt.get(id)} and ${i + 1} both have the id ${JSON.stringify(id)}; rename one and run again`), { code: 'DUPLICATE_CASE_ID' });
|
|
186
|
+
}
|
|
187
|
+
firstAt.set(id, i + 1);
|
|
133
188
|
const prompt = c.prompt || c.input || c.task;
|
|
134
189
|
const rubric = c.rubric || c.criteria || c.expected;
|
|
135
190
|
if (!prompt) throw new Error(`case "${id}" is missing a prompt/input/task`);
|
|
@@ -146,4 +201,4 @@ function normalizeCases(raw) {
|
|
|
146
201
|
});
|
|
147
202
|
}
|
|
148
203
|
|
|
149
|
-
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound };
|
|
204
|
+
module.exports = { loadSkill, normalizeCases, parseSkillMeta, pastBound, suiteIdentity };
|
package/lib/stale.js
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// `driftproof stale` (spec 053): for each receipt, does its conclusion still stand under what would
|
|
5
|
+
// run today, or does it need a rerun of one or both arms, or only a regrade?
|
|
6
|
+
//
|
|
7
|
+
// What a receipt recorded is read by lib/reuse.js provenanceOf; what would run today is built here,
|
|
8
|
+
// from the same places `driftproof run` reads (flags, then .driftproofrc; a value nobody set is
|
|
9
|
+
// unknown, never a default); the two are compared by lib/reuse.js compareProvenance, the function
|
|
10
|
+
// `triage()` uses, so the command and the triage cannot disagree about an axis.
|
|
11
|
+
//
|
|
12
|
+
// No model call and no network. The one thing spawned is `claude --version` or `codex --version`,
|
|
13
|
+
// and only when neither --harness-version nor --no-harness-check is given.
|
|
14
|
+
//
|
|
15
|
+
// THE OUTPUT IS A PUBLIC CONTRACT (spec/stale.v1.schema.json, `driftproof.stale/1`): fields are
|
|
16
|
+
// only ever added. It carries no clock, so the same inputs print the same bytes.
|
|
17
|
+
|
|
18
|
+
const fs = require('fs');
|
|
19
|
+
const path = require('path');
|
|
20
|
+
const { spawnSync } = require('child_process');
|
|
21
|
+
const { provenanceOf, compareProvenance, armDecisions } = require('./reuse');
|
|
22
|
+
const { loadSkill, suiteIdentity } = require('./skill');
|
|
23
|
+
const { rubricHash, promptTemplateHash } = require('./judge');
|
|
24
|
+
const { resolveModel } = require('./provider');
|
|
25
|
+
const { loadRegistry } = require('./models');
|
|
26
|
+
const { validateReceipt, verifyReceiptHash } = require('./receipt');
|
|
27
|
+
const { RUNNER_VERSION, PROJECT_NAME } = require('../config');
|
|
28
|
+
|
|
29
|
+
const SCHEMA_ID = 'driftproof.stale/1';
|
|
30
|
+
const HARNESS_BIN = { 'claude-code': 'claude', codex: 'codex' };
|
|
31
|
+
const SURFACE_HARNESS = { 'claude-cli': 'claude-code', 'openai-cli': 'codex', api: 'api', 'openai-api': 'api' };
|
|
32
|
+
|
|
33
|
+
// Today's version of a harness, read from its binary on PATH (spec 053 R-5).
|
|
34
|
+
function harnessVersionNow(name) {
|
|
35
|
+
const bin = HARNESS_BIN[name];
|
|
36
|
+
if (!bin) return { version: null, why: `no way to read today's ${name} version` };
|
|
37
|
+
const r = spawnSync(bin, ['--version'], { encoding: 'utf8', timeout: 10000 });
|
|
38
|
+
if (r.error) return { version: null, why: `${bin} --version could not run: ${r.error.code || r.error.message}` };
|
|
39
|
+
if (r.status !== 0) return { version: null, why: `${bin} --version exited ${r.status}` };
|
|
40
|
+
const m = /\d+\.\d+\.\d+[0-9A-Za-z.+-]*/.exec(r.stdout || '');
|
|
41
|
+
return m ? { version: m[0] } : { version: null, why: `${bin} --version printed no version` };
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// What would run today, for one receipt. `why` carries the reason for each value that could not be found.
|
|
45
|
+
function currentProvenance(receipt, opts, cache) {
|
|
46
|
+
const why = {};
|
|
47
|
+
const rec = provenanceOf(receipt);
|
|
48
|
+
const rc = opts.rc || {};
|
|
49
|
+
// Model: --model, else the rc's models; the receipt's model is current when the rc names it.
|
|
50
|
+
let model = null;
|
|
51
|
+
if (opts.model) model = resolveModel(opts.model);
|
|
52
|
+
else if (rc.models !== undefined && rc.models !== null) {
|
|
53
|
+
const list = String(rc.models).split(',').map((x) => x.trim()).filter(Boolean).map((x) => resolveModel(x));
|
|
54
|
+
model = rec.model_id && list.includes(resolveModel(rec.model_id)) ? resolveModel(rec.model_id) : list.join(',') || null;
|
|
55
|
+
}
|
|
56
|
+
if (!model) why.model = 'no --model given and no models in .driftproofrc';
|
|
57
|
+
// Judge: --judge, else the rc's judge_model.
|
|
58
|
+
const judgeRaw = opts.judge || rc.judge_model || null;
|
|
59
|
+
const judge = judgeRaw ? resolveModel(judgeRaw) : null;
|
|
60
|
+
if (!judge) why.judge_model = 'no --judge given and no judge_model in .driftproofrc';
|
|
61
|
+
// Skill and suite.
|
|
62
|
+
let skillHash = null; let suite = null;
|
|
63
|
+
if (opts.skill) {
|
|
64
|
+
const s = cache.skill || (cache.skill = loadSkill(opts.skill));
|
|
65
|
+
skillHash = s.contentHash;
|
|
66
|
+
if (!opts.suite) suite = { suiteHash: s.suite.suiteHash, cases: s.suite.cases };
|
|
67
|
+
} else why.skill = 'no --skill given';
|
|
68
|
+
if (opts.suite) suite = cache.suite || (cache.suite = suiteIdentity(JSON.parse(fs.readFileSync(opts.suite, 'utf8'))));
|
|
69
|
+
if (!suite) { why.suite = 'no --suite or --skill given'; why.rubric = why.suite; }
|
|
70
|
+
// Harness: the receipt's harness name, else its surface's.
|
|
71
|
+
const name = (rec.harness && rec.harness.name) || SURFACE_HARNESS[(receipt.run || {}).surface] || null;
|
|
72
|
+
let harness = null;
|
|
73
|
+
if (!name) why.harness = 'the receipt names no harness and no surface with one';
|
|
74
|
+
else if (name === 'api') harness = { name: 'api', version: null };
|
|
75
|
+
else if (opts.noHarnessCheck) why.harness = 'not checked (--no-harness-check)';
|
|
76
|
+
else if (opts.harnessVersion) harness = { name, version: String(opts.harnessVersion) };
|
|
77
|
+
else {
|
|
78
|
+
const v = cache[`h:${name}`] || (cache[`h:${name}`] = harnessVersionNow(name));
|
|
79
|
+
harness = { name, version: v.version };
|
|
80
|
+
if (v.version == null) why.harness = v.why;
|
|
81
|
+
}
|
|
82
|
+
const rubrics = suite ? Object.fromEntries(suite.cases.map((c) => [c.id, rubricHash(c.rubric)])) : null;
|
|
83
|
+
return {
|
|
84
|
+
prov: {
|
|
85
|
+
model_id: model, harness, skill_hash: skillHash, suite_hash: suite ? suite.suiteHash : null,
|
|
86
|
+
case_ids: suite ? [...new Set(suite.cases.map((c) => c.id))].sort() : null,
|
|
87
|
+
judge_model_id: judge, judge_template_hash: promptTemplateHash(), rubrics,
|
|
88
|
+
},
|
|
89
|
+
why,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// The newer-model advisory (spec 053 R-9): same family, registered, released after both the run's
|
|
94
|
+
// date and the recorded model's release date. No advisory when either date is missing.
|
|
95
|
+
const dated = (m) => !!m && typeof m.released === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(m.released);
|
|
96
|
+
function advisories(receipt) {
|
|
97
|
+
const reg = loadRegistry();
|
|
98
|
+
const id = (receipt.run || {}).model_id;
|
|
99
|
+
const row = id ? reg.byId[resolveModel(id)] || reg.byId[id] : null;
|
|
100
|
+
const runDay = typeof (receipt.run || {}).date_utc === 'string' ? receipt.run.date_utc.slice(0, 10) : null;
|
|
101
|
+
if (!dated(row) || !runDay) return [];
|
|
102
|
+
return reg.models
|
|
103
|
+
.filter((m) => m.id !== row.id && m.family === row.family && !m.auto_added && dated(m) && String(m.released) > runDay && String(m.released) > String(row.released))
|
|
104
|
+
.sort((a, b) => (a.released === b.released ? (a.id < b.id ? -1 : 1) : (a.released < b.released ? 1 : -1)))
|
|
105
|
+
.map((m) => ({ model_id: m.id, released: m.released }));
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// The command to run next (spec 053 R-7).
|
|
109
|
+
function nextCommand(file, receipt, arms, opts, current) {
|
|
110
|
+
const skillArg = opts.skill || '<skill dir>';
|
|
111
|
+
const d = [arms.with_skill.decision, arms.baseline.decision];
|
|
112
|
+
if (d.includes('rerun')) return `${PROJECT_NAME} run ${skillArg} --model ${current.model_id && !current.model_id.includes(',') ? current.model_id : receipt.run.model_id}`;
|
|
113
|
+
if (d.includes('regrade')) return `${PROJECT_NAME} regrade ${file} --skill ${skillArg} --answers <answers.json> --judge-model ${current.judge_model_id || receipt.run.judge.model_id}`;
|
|
114
|
+
return null;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function assess(file, receipt, opts, cache) {
|
|
118
|
+
const { prov, why } = currentProvenance(receipt, opts, cache);
|
|
119
|
+
const axes = compareProvenance(provenanceOf(receipt), prov, why);
|
|
120
|
+
const arms = armDecisions(axes);
|
|
121
|
+
const d = [arms.with_skill.decision, arms.baseline.decision];
|
|
122
|
+
const status = d.includes('rerun') || d.includes('regrade') ? 'stale' : d.includes('unknown') ? 'unknown' : 'current';
|
|
123
|
+
const run = receipt.run || {};
|
|
124
|
+
return {
|
|
125
|
+
path: file,
|
|
126
|
+
receipt_hash: receipt.receipt_hash || null,
|
|
127
|
+
skill: (receipt.skill && receipt.skill.name) || null,
|
|
128
|
+
model_id: run.model_id || null,
|
|
129
|
+
date_utc: run.date_utc || null,
|
|
130
|
+
verification_level: receipt.verification_level || null,
|
|
131
|
+
status,
|
|
132
|
+
arms,
|
|
133
|
+
axes,
|
|
134
|
+
advisories: advisories(receipt),
|
|
135
|
+
runner_version: { recorded: run.runner_version || null, current: RUNNER_VERSION },
|
|
136
|
+
next: nextCommand(file, receipt, arms, opts, prov),
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// The whole report: every receipt read, validated and assessed, and the exit code (spec 053 R-8).
|
|
141
|
+
function staleReport(files, opts = {}) {
|
|
142
|
+
const cache = {};
|
|
143
|
+
const receipts = files.map((file) => {
|
|
144
|
+
let r;
|
|
145
|
+
try { r = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { return { path: file, error: e.code === 'ENOENT' ? 'the file does not exist' : `the file is not a readable JSON document: ${e.message}` }; }
|
|
146
|
+
const v = validateReceipt(r);
|
|
147
|
+
if (!v.valid) return { path: file, error: `the receipt does not validate: ${(v.errors || []).slice(0, 2).map((x) => (typeof x === 'string' ? x : `${x.instancePath || ''} ${x.message || ''}`.trim())).join('; ')}` };
|
|
148
|
+
// As `driftproof validate` reads a receipt: the schema, and the hash over its content.
|
|
149
|
+
if (!verifyReceiptHash(r)) return { path: file, error: 'the receipt_hash does not match the receipt\'s content' };
|
|
150
|
+
try { return assess(file, r, opts, cache); } catch (e) { return { path: file, error: `the receipt could not be assessed: ${e.message}` }; }
|
|
151
|
+
});
|
|
152
|
+
const ok = receipts.filter((x) => !x.error);
|
|
153
|
+
// Advisories: a newer model, and an axis that moved by a patch release only (A-053-1).
|
|
154
|
+
const summary = { receipts: receipts.length, current: ok.filter((x) => x.status === 'current').length, stale: ok.filter((x) => x.status === 'stale').length, unknown: ok.filter((x) => x.status === 'unknown').length, errors: receipts.length - ok.length, advisories: ok.reduce((n, x) => n + x.advisories.length + x.axes.filter((e) => e.effect === 'advisory').length, 0) };
|
|
155
|
+
let exit = summary.errors ? 2 : summary.stale ? 1 : summary.unknown || summary.advisories ? 3 : 0;
|
|
156
|
+
if (opts.strict && exit === 3) exit = 1;
|
|
157
|
+
return { schema: SCHEMA_ID, runner_version: RUNNER_VERSION, strict: !!opts.strict, receipts, summary, exit_code: exit };
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// ── the human output ─────────────────────────────────────────────────────────
|
|
161
|
+
const LABEL = { model: 'model', harness: 'harness', skill: 'skill', suite: 'suite', judge_model: 'judge model', judge_template: 'judge template', rubric: 'rubric' };
|
|
162
|
+
const short = (v) => (typeof v === 'string' && /^[0-9a-f]{64}$/.test(v) ? v.slice(0, 8) : v == null ? 'unrecorded' : String(v));
|
|
163
|
+
function effectText(e) {
|
|
164
|
+
if (e.effect === 'unknown') return `unknown: ${e.reason}`;
|
|
165
|
+
if (e.effect === 'advisory') return 'a patch release only (advisory)';
|
|
166
|
+
if (e.effect === 'regrade') return e.cases ? `regrade case(s) ${e.cases.join(', ')}` : 'regrade both arms';
|
|
167
|
+
if (e.effect === 'rerun') {
|
|
168
|
+
if (e.arms.length === 1) return 'rerun the with-skill arm; baseline still valid';
|
|
169
|
+
const extra = [e.cases_added && e.cases_added.length ? `added ${e.cases_added.join(', ')}` : '', e.cases_removed && e.cases_removed.length ? `removed ${e.cases_removed.join(', ')}` : '', e.cases ? `rubric moved on ${e.cases.join(', ')}` : ''].filter(Boolean).join('; ');
|
|
170
|
+
return `rerun both arms${extra ? ` (${extra})` : ''}`;
|
|
171
|
+
}
|
|
172
|
+
return 'unchanged';
|
|
173
|
+
}
|
|
174
|
+
function renderText(doc) {
|
|
175
|
+
const L = [];
|
|
176
|
+
for (const x of doc.receipts) {
|
|
177
|
+
if (x.error) { L.push(`${x.path} ERROR`, ` ${x.error}`, ''); continue; }
|
|
178
|
+
L.push(`${x.skill || '(no skill)'} ${x.model_id || '(no model)'} ${x.date_utc ? String(x.date_utc).slice(0, 10) : '(no date)'} ${x.status.toUpperCase()}`);
|
|
179
|
+
for (const e of x.axes) {
|
|
180
|
+
if (e.effect === 'current' && e.axis !== 'judge_model') continue;
|
|
181
|
+
const now = e.current == null ? 'not known' : short(e.current);
|
|
182
|
+
const move = e.effect === 'current' ? `${now}, unchanged` : `${short(e.recorded)} to ${now}`;
|
|
183
|
+
// Two spaces at least between the move and its effect, however long the move (approval F-1).
|
|
184
|
+
L.push(` ${LABEL[e.axis].padEnd(14)}${(move + ' ').padEnd(38)}${e.effect === 'current' ? '' : effectText(e)}`.trimEnd());
|
|
185
|
+
}
|
|
186
|
+
for (const a of x.advisories) L.push(` ${'newer model'.padEnd(14)}${a.model_id}, released ${a.released} (advisory)`);
|
|
187
|
+
L.push(` ${'next'.padEnd(14)}${x.next || (x.status === 'current' ? 'nothing to run' : 'nothing can be decided until the unknown axes are known')}`, '');
|
|
188
|
+
}
|
|
189
|
+
const s = doc.summary;
|
|
190
|
+
L.push(`${s.receipts} receipt(s): ${s.current} current, ${s.stale} stale, ${s.unknown} unknown, ${s.errors} error(s); exit ${doc.exit_code}${doc.strict ? ' (--strict)' : ''}`);
|
|
191
|
+
return L.join('\n') + '\n';
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
module.exports = { staleReport, renderText, currentProvenance, advisories, SCHEMA_ID };
|