driftproof 0.8.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +87 -16
- package/bin/driftproof +289 -24
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/decision.js +425 -0
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/runner.js +35 -3
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/checks.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const { CHECK_MAX_PATTERN } = require('../config');
|
|
5
|
+
|
|
4
6
|
// Deterministic post-checks.
|
|
5
7
|
//
|
|
6
8
|
// A per-case eval suite may declare optional `checks[]`: structural / regex
|
|
@@ -19,6 +21,22 @@
|
|
|
19
21
|
// contains — the output includes the literal `value` substring
|
|
20
22
|
// not_contains — the output does NOT include the literal `value` substring
|
|
21
23
|
// min_length — the trimmed output is at least `value` characters long
|
|
24
|
+
//
|
|
25
|
+
// BOUNDED (spec 026 AC-13, audit A7). A suite's regex is third-party input
|
|
26
|
+
// compiled and run against model output, and JavaScript cannot time a regex
|
|
27
|
+
// out. A pattern longer than CHECK_MAX_PATTERN, or one carrying a nested
|
|
28
|
+
// quantifier of the (x+)+ family (a quantified group whose body ends in a
|
|
29
|
+
// quantifier: the shape that is exponential on every backtracking engine),
|
|
30
|
+
// is REFUSED: recorded pass: false with a reason that says so, never
|
|
31
|
+
// evaluated. The test is syntactic and narrow by design; it is not a general
|
|
32
|
+
// ReDoS detector, and the spec says so.
|
|
33
|
+
const NESTED_QUANTIFIER = /\((?:[^()\\]|\\.)*[+*}]\)\s*[+*]|\((?:[^()\\]|\\.)*\|(?:[^()\\]|\\.)*\)\s*[+*]/;
|
|
34
|
+
function refusedPattern(pattern) {
|
|
35
|
+
const p = String(pattern == null ? '' : pattern);
|
|
36
|
+
if (p.length > CHECK_MAX_PATTERN) return `refused: pattern of ${p.length} characters exceeds CHECK_MAX_PATTERN (${CHECK_MAX_PATTERN})`;
|
|
37
|
+
if (NESTED_QUANTIFIER.test(p)) return 'refused: pattern carries a nested quantifier of the (x+)+ family, which is exponential to evaluate';
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
22
40
|
|
|
23
41
|
function runOneCheck(check, output) {
|
|
24
42
|
const text = String(output || '');
|
|
@@ -40,11 +58,15 @@ function runOneCheck(check, output) {
|
|
|
40
58
|
// `checks`). Empty array when the case declares no checks.
|
|
41
59
|
function runChecks(output, checks) {
|
|
42
60
|
if (!Array.isArray(checks) || !checks.length) return [];
|
|
43
|
-
return checks.map((c) =>
|
|
44
|
-
|
|
45
|
-
kind: c && c.kind,
|
|
46
|
-
|
|
47
|
-
|
|
61
|
+
return checks.map((c) => {
|
|
62
|
+
const reason = c && c.kind === 'regex' ? refusedPattern(c.pattern) : null;
|
|
63
|
+
if (reason) return { name: String((c && c.name) || (c && c.kind) || 'check'), kind: c && c.kind, pass: false, reason };
|
|
64
|
+
return {
|
|
65
|
+
name: String((c && c.name) || (c && c.kind) || 'check'),
|
|
66
|
+
kind: c && c.kind,
|
|
67
|
+
pass: !!runOneCheck(c, output),
|
|
68
|
+
};
|
|
69
|
+
});
|
|
48
70
|
}
|
|
49
71
|
|
|
50
|
-
module.exports = { runChecks, runOneCheck };
|
|
72
|
+
module.exports = { runChecks, runOneCheck, refusedPattern, NESTED_QUANTIFIER };
|
package/lib/decision.js
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// The decision a RUN reaches, over every receipt it produced (spec 030).
|
|
5
|
+
//
|
|
6
|
+
// `lib/verdict.js` answers a question about ONE receipt: does the skill still
|
|
7
|
+
// help on THIS model? That question is still asked and still answered there,
|
|
8
|
+
// unchanged. This module answers the one the action actually needs and never
|
|
9
|
+
// had a place to ask: a run takes a LIST of models and must reach a single
|
|
10
|
+
// decision over all of them.
|
|
11
|
+
//
|
|
12
|
+
// Until spec 030, `action/run.sh` picked one receipt with `ls … | head -1` and
|
|
13
|
+
// decided the job from it. Receipts are named `<skill>-<model>-<date>.json`, so
|
|
14
|
+
// that pick is ordered by model id: `claude-haiku-4-5` sorts before
|
|
15
|
+
// `claude-sonnet-5`, and a regression on any model that did not sort first was
|
|
16
|
+
// reported as a pass, with a green check and a brightgreen badge. The seam lives
|
|
17
|
+
// here, in lib/, rather than in the shell, because `bin/driftproof badge` needs
|
|
18
|
+
// it too and because a decision state is a property of a receipt.
|
|
19
|
+
|
|
20
|
+
const fs = require('fs');
|
|
21
|
+
const path = require('path');
|
|
22
|
+
const { EFFECT_FLOOR } = require('../config');
|
|
23
|
+
const { shortModel, githubOutputEntry } = require('./verdict');
|
|
24
|
+
|
|
25
|
+
// The six decision states (spec 030 AC-4).
|
|
26
|
+
//
|
|
27
|
+
// `helped` is the sixth, and is a RECORDED DEVIATION from the brief's five
|
|
28
|
+
// (spec.md A-030-1): a model on which the skill measurably helped needs a row
|
|
29
|
+
// that says so. `no detected effect` is false about such a model - the effect
|
|
30
|
+
// was detected and it cleared the floor - and widening it to mean "not a
|
|
31
|
+
// regression" would put a measured lift and a measured nothing in one cell.
|
|
32
|
+
const STATES = ['regression', 'refused', 'inconclusive', 'not measured', 'no detected effect', 'helped'];
|
|
33
|
+
|
|
34
|
+
// WORST FIRST. This is the single definition of the ordering; AC-1's
|
|
35
|
+
// enforcement and AC-3's badge both read it, so they cannot disagree about
|
|
36
|
+
// which of two states governs a run.
|
|
37
|
+
const STATE_ORDER = STATES.slice();
|
|
38
|
+
|
|
39
|
+
// States that must never render as success on any surface the action writes -
|
|
40
|
+
// the badge, the summary row, or the check title (AC-4).
|
|
41
|
+
const NEVER_SUCCESS = ['inconclusive', 'not measured'];
|
|
42
|
+
|
|
43
|
+
// States that fail the job. `refused` fails CLOSED (AC-2): a model that
|
|
44
|
+
// produced no receipt is not a model that passed. `regression` fails subject to
|
|
45
|
+
// fail-on-regression. An unmeasured run does not fail on its own (AC-4's rule)
|
|
46
|
+
// - it is not evidence that the skill hurt - but it never renders as success.
|
|
47
|
+
function failsJob(state, { failOnRegression = true } = {}) {
|
|
48
|
+
// AC-2. `refused` fails CLOSED and is NOT subject to fail-on-regression: that
|
|
49
|
+
// input says what to do about a measured regression, and a model that
|
|
50
|
+
// produced no receipt was not measured at all. A run that silently narrowed
|
|
51
|
+
// to the models that survived would be AC-1's defect with a different cause -
|
|
52
|
+
// the decision taken over a subset nobody chose.
|
|
53
|
+
if (state === 'refused') return true;
|
|
54
|
+
if (state === 'regression') return !!failOnRegression;
|
|
55
|
+
return false;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// The mapping rule of spec 030 AC-4, in order; the first match wins. A null or
|
|
59
|
+
// missing receipt is `refused`.
|
|
60
|
+
//
|
|
61
|
+
// `not measured` is tested BEFORE `inconclusive` and before any delta is read,
|
|
62
|
+
// because a receipt below TESTED has not measured anything its delta could be
|
|
63
|
+
// about - spec 026 AC-1 is what makes a stub receipt unable to claim TESTED,
|
|
64
|
+
// and reading its delta first would grade a run that never ran.
|
|
65
|
+
// The rule is read off AC-4's table and nothing is supplied that the receipt
|
|
66
|
+
// does not carry. Until this it read
|
|
67
|
+
//
|
|
68
|
+
// const level = receipt.verification_level || 'TESTED';
|
|
69
|
+
// ... || (kind && kind !== 'model')
|
|
70
|
+
//
|
|
71
|
+
// - a receipt with no `verification_level` was GRADED ON ITS DELTA as though it
|
|
72
|
+
// had claimed TESTED, and a receipt with no `answered_by` block had the second
|
|
73
|
+
// clause skipped entirely. Both defaults point the same way: they let a receipt
|
|
74
|
+
// that says nothing about how it was answered reach `helped`. AC-4 says each
|
|
75
|
+
// state SHALL be derived "by the rule below and by no other", and the rule says
|
|
76
|
+
// `verification_level != "TESTED"` and `run.answered_by.kind != "model"` - an
|
|
77
|
+
// absent field satisfies both. The receipts the shipped runner writes always
|
|
78
|
+
// carry a level and an answered_by block (lib/run.js answeredByOf always
|
|
79
|
+
// returns a kind), so this changes nothing about a run the action produced; it
|
|
80
|
+
// changes what happens to a receipt from somewhere else, and it changes it in
|
|
81
|
+
// the fail-safe direction.
|
|
82
|
+
function decisionState(receipt) {
|
|
83
|
+
if (!receipt) return 'refused';
|
|
84
|
+
const level = receipt.verification_level;
|
|
85
|
+
const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
|
|
86
|
+
if (level !== 'TESTED' || kind !== 'model') return 'not measured';
|
|
87
|
+
const cmp = receipt.comparison || {};
|
|
88
|
+
if ((receipt.run && receipt.run.status === 'incomplete') || typeof cmp.delta !== 'number') return 'inconclusive';
|
|
89
|
+
if (cmp.delta <= -EFFECT_FLOOR) return 'regression';
|
|
90
|
+
if (cmp.delta >= EFFECT_FLOOR) return 'helped';
|
|
91
|
+
return 'no detected effect';
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// The worst state in a set, by STATE_ORDER. An empty set has no decision, and
|
|
95
|
+
// says so with null rather than defaulting to something benign.
|
|
96
|
+
function worstState(states) {
|
|
97
|
+
let worst = null;
|
|
98
|
+
for (const s of states) {
|
|
99
|
+
const i = STATE_ORDER.indexOf(s);
|
|
100
|
+
if (i < 0) continue;
|
|
101
|
+
if (worst === null || i < STATE_ORDER.indexOf(worst)) worst = s;
|
|
102
|
+
}
|
|
103
|
+
return worst;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// Every receipt in a directory, excluding the summaries and the badge writer's
|
|
107
|
+
// own earlier output. Mirrors what action/run.sh's glob selected from, so the
|
|
108
|
+
// set this reads is the set that was there to be read.
|
|
109
|
+
function receiptFiles(dir) {
|
|
110
|
+
return fs.readdirSync(dir)
|
|
111
|
+
.filter((f) => f.endsWith('.json') && !f.includes('.summary.') && f !== 'badge.json')
|
|
112
|
+
.sort()
|
|
113
|
+
.map((f) => path.join(dir, f));
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// Parse the `models` input the same way the runner does: comma-separated, with
|
|
117
|
+
// surrounding whitespace ignored and empty entries dropped.
|
|
118
|
+
function parseModels(csv) {
|
|
119
|
+
return String(csv || '').split(',').map((s) => s.trim()).filter(Boolean);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// A receipt matches a requested id exactly, or after the release date stamp is
|
|
123
|
+
// stripped from both - `claude-haiku-4-5-20251001` IS `claude-haiku-4-5`, which
|
|
124
|
+
// is the equivalence lib/verdict.js's shortModel already defines for the badge.
|
|
125
|
+
function matchesModel(receiptModelId, requested) {
|
|
126
|
+
return receiptModelId === requested || shortModel(receiptModelId) === shortModel(requested);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// The same question for a receipt that CANNOT BE READ. `matchesModel` reads
|
|
130
|
+
// `run.model_id`, and for an unreadable receipt that field is inside the bytes
|
|
131
|
+
// that will not parse. The FILENAME is the only thing left that names a model:
|
|
132
|
+
// the runner writes `<skill>-<model>-<date>.json`, so the id appears in the
|
|
133
|
+
// basename as a dash-delimited run.
|
|
134
|
+
//
|
|
135
|
+
// Until this, the unreadable branch matched by POSITION - the first unreadable
|
|
136
|
+
// file in the directory, whichever model it belonged to. With one requested model
|
|
137
|
+
// absent and another's receipt unreadable, the two causes were hung on the wrong
|
|
138
|
+
// models: the absent model was reported as "exists and is unreadable" naming the
|
|
139
|
+
// OTHER model's file, and the model whose file was actually corrupt was reported
|
|
140
|
+
// as having produced nothing. Both still failed closed and both ids still
|
|
141
|
+
// appeared somewhere, which is why AC-2's arms stayed green - they drove one
|
|
142
|
+
// cause at a time and never both at once (F-1 of 2026-09-12).
|
|
143
|
+
function fileNamesModel(file, requested) {
|
|
144
|
+
const base = path.basename(file);
|
|
145
|
+
// Both spellings, for the reason `matchesModel` takes both: a receipt for
|
|
146
|
+
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`.
|
|
147
|
+
for (const id of new Set([requested, shortModel(requested)])) {
|
|
148
|
+
if (base.includes(`-${id}-`)) return true;
|
|
149
|
+
}
|
|
150
|
+
return false;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// The decision over a run: one row per REQUESTED model, in the order requested.
|
|
154
|
+
//
|
|
155
|
+
// Rows come from the requested list rather than from the directory listing, so
|
|
156
|
+
// a model that produced no receipt is a row that says `refused` instead of a
|
|
157
|
+
// row that is simply absent. That is the whole of AC-2: absence is not a pass.
|
|
158
|
+
function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
159
|
+
const requested = parseModels(requestedCsv);
|
|
160
|
+
const files = receiptFiles(dir);
|
|
161
|
+
const loaded = files.map((f) => {
|
|
162
|
+
let receipt = null, error = null;
|
|
163
|
+
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (e) { error = e.message; }
|
|
164
|
+
return { file: f, receipt, error };
|
|
165
|
+
});
|
|
166
|
+
const used = new Set();
|
|
167
|
+
const rows = requested.map((model) => {
|
|
168
|
+
const hit = loaded.find((l) => !used.has(l.file) && l.receipt
|
|
169
|
+
&& l.receipt.run && matchesModel(l.receipt.run.model_id, model));
|
|
170
|
+
if (hit) used.add(hit.file);
|
|
171
|
+
// A receipt that exists and cannot be parsed is UNREADABLE, not absent.
|
|
172
|
+
// Both fail closed, and the row says which - the register's
|
|
173
|
+
// absence-vs-unreadable distinction, kept at the point it is decided.
|
|
174
|
+
const unreadable = !hit && loaded.find((l) => !used.has(l.file) && l.error
|
|
175
|
+
&& fileNamesModel(l.file, model));
|
|
176
|
+
if (unreadable) used.add(unreadable.file);
|
|
177
|
+
const receipt = hit ? hit.receipt : null;
|
|
178
|
+
const state = decisionState(receipt);
|
|
179
|
+
return {
|
|
180
|
+
model,
|
|
181
|
+
state,
|
|
182
|
+
delta: receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
|
|
183
|
+
? receipt.comparison.delta : null,
|
|
184
|
+
file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
|
|
185
|
+
// The distinction is carried on the ROW, not left in a sentence, because
|
|
186
|
+
// the enforcement message has to say which of the two happened (AC-2's
|
|
187
|
+
// mutation class, absence-vs-unreadable).
|
|
188
|
+
unreadable: unreadable ? unreadable.error : null,
|
|
189
|
+
// A SHORT CAUSE, not the parser's text. V8's JSON.parse message echoes the
|
|
190
|
+
// first bytes of the input - "Unexpected token '|', \"|{bad\" is not valid
|
|
191
|
+
// JSON" - and this string is rendered into a markdown table cell, where one
|
|
192
|
+
// unescaped `|` adds a column and a newline ends the row. The detail still
|
|
193
|
+
// reaches the annotation, which is not a table; the cell carries the cause,
|
|
194
|
+
// and the filename is already beside it in the same cell (F-2 of
|
|
195
|
+
// 2026-09-12).
|
|
196
|
+
reason: unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model'),
|
|
197
|
+
};
|
|
198
|
+
});
|
|
199
|
+
// Receipts the run wrote for models nobody requested are reported rather than
|
|
200
|
+
// dropped: they are evidence that the run did something other than what it
|
|
201
|
+
// was asked for.
|
|
202
|
+
const unexpected = loaded.filter((l) => !used.has(l.file))
|
|
203
|
+
// An unreadable file whose name matches no requested model is still reported,
|
|
204
|
+
// and reported AS unreadable. Calling it a receipt for a model nobody asked
|
|
205
|
+
// for would state as fact the one thing its bytes could not tell us.
|
|
206
|
+
.map((l) => (l.error ? `${path.basename(l.file)} (unreadable)` : path.basename(l.file)));
|
|
207
|
+
const worst = worstState(rows.map((r) => r.state));
|
|
208
|
+
return {
|
|
209
|
+
rows,
|
|
210
|
+
worst,
|
|
211
|
+
regressed: rows.filter((r) => r.state === 'regression').map((r) => r.model),
|
|
212
|
+
missing: rows.filter((r) => r.state === 'refused').map((r) => r.model),
|
|
213
|
+
// `missing` is every model that reached no readable receipt, which is what
|
|
214
|
+
// the missing_models output has always published. `absent` and `unreadable`
|
|
215
|
+
// split it by CAUSE: nothing was written for this model, or something was
|
|
216
|
+
// written and cannot be read. Both fail closed; they are not the same fact,
|
|
217
|
+
// and the failure has to say which.
|
|
218
|
+
absent: rows.filter((r) => r.state === 'refused' && !r.unreadable).map((r) => r.model),
|
|
219
|
+
unreadable: rows.filter((r) => r.state === 'refused' && r.unreadable)
|
|
220
|
+
.map((r) => ({ model: r.model, file: r.file, error: r.unreadable })),
|
|
221
|
+
unexpected,
|
|
222
|
+
receiptCount: files.length,
|
|
223
|
+
requestedCount: requested.length,
|
|
224
|
+
fails: rows.some((r) => failsJob(r.state, { failOnRegression })),
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
module.exports = {
|
|
229
|
+
STATES, STATE_ORDER, NEVER_SUCCESS, EFFECT_FLOOR,
|
|
230
|
+
decisionState, worstState, failsJob, decideSet, receiptFiles, parseModels, matchesModel,
|
|
231
|
+
};
|
|
232
|
+
|
|
233
|
+
// ── the three surfaces the action writes ────────────────────────────────────
|
|
234
|
+
//
|
|
235
|
+
// AC-4 binds all three: the badge, the summary row and the check title. A state
|
|
236
|
+
// in NEVER_SUCCESS must not render as success on any of them, which is why the
|
|
237
|
+
// renderers live together - three surfaces fed by one table cannot drift apart
|
|
238
|
+
// the way three hand-written strings do.
|
|
239
|
+
|
|
240
|
+
// How each state renders. `color` is the shields colour; `marker` is the
|
|
241
|
+
// summary row's marker. Only `helped` gets a success marker: `no detected
|
|
242
|
+
// effect` is not a failure but it is not a success either, and the two
|
|
243
|
+
// NEVER_SUCCESS states carry a warning.
|
|
244
|
+
const RENDER = {
|
|
245
|
+
regression: { word: 'regressed', color: 'red', marker: '\u274c' },
|
|
246
|
+
refused: { word: 'refused', color: 'red', marker: '\u274c' },
|
|
247
|
+
inconclusive: { word: 'inconclusive', color: 'yellow', marker: '\u26a0\ufe0f' },
|
|
248
|
+
'not measured': { word: 'not measured', color: 'lightgrey', marker: '\u26a0\ufe0f' },
|
|
249
|
+
'no detected effect': { word: 'no effect', color: 'lightgrey', marker: '\u2014' },
|
|
250
|
+
helped: { word: 'passing', color: 'brightgreen', marker: '\u2705' },
|
|
251
|
+
};
|
|
252
|
+
|
|
253
|
+
// The one place a state becomes a success claim. AC-4's clause is enforced here
|
|
254
|
+
// rather than restated at each surface.
|
|
255
|
+
function rendersAsSuccess(state) {
|
|
256
|
+
return state === 'helped';
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
// The badge over a SET (AC-3): the worst state, naming the model it came from,
|
|
260
|
+
// and how many models share it when more than one does.
|
|
261
|
+
function badgeMessage(d) {
|
|
262
|
+
const worst = d.worst;
|
|
263
|
+
if (!worst) return 'no receipts';
|
|
264
|
+
const r = RENDER[worst];
|
|
265
|
+
const sharing = d.rows.filter((row) => row.state === worst);
|
|
266
|
+
const named = sharing[0] ? sharing[0].model : 'unknown';
|
|
267
|
+
const extra = sharing.length > 1 ? ` (+${sharing.length - 1} more)` : '';
|
|
268
|
+
return `${r.word} on ${shortModel(named)}${extra}`;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function badgeEndpointForSet(d, { label = 'driftproof' } = {}) {
|
|
272
|
+
const worst = d.worst;
|
|
273
|
+
return {
|
|
274
|
+
schemaVersion: 1,
|
|
275
|
+
label,
|
|
276
|
+
message: badgeMessage(d),
|
|
277
|
+
color: worst ? RENDER[worst].color : 'lightgrey',
|
|
278
|
+
};
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// The step outputs. `verdict` keeps the vocabulary lib/verdict.js publishes, so
|
|
282
|
+
// a workflow reading steps.run.outputs.verdict is not broken by this change;
|
|
283
|
+
// `worst`, `regressed` and `missing` are new and carry what it could not say.
|
|
284
|
+
const VERDICT_WORD = {
|
|
285
|
+
regression: 'REGRESSED', refused: 'REFUSED', inconclusive: 'INCONCLUSIVE',
|
|
286
|
+
'not measured': 'NOT_MEASURED', 'no detected effect': 'NO_EFFECT', helped: 'PASSED',
|
|
287
|
+
};
|
|
288
|
+
|
|
289
|
+
function githubOutputLines(d) {
|
|
290
|
+
const worst = d.worst;
|
|
291
|
+
const badge = badgeEndpointForSet(d);
|
|
292
|
+
const worstRow = d.rows.find((r) => r.state === worst);
|
|
293
|
+
return [
|
|
294
|
+
['verdict', worst ? VERDICT_WORD[worst] : 'NOT_MEASURED'],
|
|
295
|
+
['delta', worstRow && worstRow.delta !== null ? worstRow.delta : 0],
|
|
296
|
+
['message', badge.message],
|
|
297
|
+
['color', badge.color],
|
|
298
|
+
['worst_state', worst || 'none'],
|
|
299
|
+
['regressed_models', d.regressed.join(',')],
|
|
300
|
+
['missing_models', d.missing.join(',')],
|
|
301
|
+
['receipt_count', d.receiptCount],
|
|
302
|
+
].map(([k, v]) => githubOutputEntry(k, v)).join('\n');
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// Anything reaching a table cell from outside this module - a filename, a reason,
|
|
306
|
+
// a model id - is escaped FOR THE CELL it sits in. A markdown row is delimited by
|
|
307
|
+
// `|`, so one unescaped pipe adds a column and the row stops lining up with its
|
|
308
|
+
// header; a newline ends the row outright. The cause AC-4's own finding had was a
|
|
309
|
+
// parser echo, and that echo is gone from the cell - but the filename is still
|
|
310
|
+
// read off a directory, so the row is made to survive the byte rather than made
|
|
311
|
+
// to depend on where the byte came from.
|
|
312
|
+
function cell(s) {
|
|
313
|
+
return s === null || s === undefined ? '' : String(s).replace(/\r?\n/g, ' ').replace(/\|/g, '\\|');
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
// The job summary: one row per REQUESTED model, in the order requested (AC-4).
|
|
317
|
+
function summaryMarkdown(d) {
|
|
318
|
+
const L = [];
|
|
319
|
+
L.push('### Driftproof');
|
|
320
|
+
L.push('');
|
|
321
|
+
L.push('| Model | Decision | Lift | Receipt |');
|
|
322
|
+
L.push('|---|---|---|---|');
|
|
323
|
+
for (const row of d.rows) {
|
|
324
|
+
const r = RENDER[row.state];
|
|
325
|
+
const delta = row.delta === null ? 'n/a' : (row.delta >= 0 ? '+' : '') + row.delta.toFixed(3);
|
|
326
|
+
const note = row.reason ? ` <br><sub>${cell(row.reason)}</sub>` : '';
|
|
327
|
+
L.push(`| \`${cell(row.model)}\` | ${r.marker} ${row.state} | ${delta} | ${cell(row.file) || '\u2014'}${note} |`);
|
|
328
|
+
}
|
|
329
|
+
L.push('');
|
|
330
|
+
L.push(`Decided over ${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
|
|
331
|
+
+ `Worst: **${d.worst || 'none'}**.`);
|
|
332
|
+
if (d.unexpected.length) {
|
|
333
|
+
L.push('');
|
|
334
|
+
L.push(`Receipts in the directory that no requested model claimed: ${d.unexpected.join(', ')}.`);
|
|
335
|
+
}
|
|
336
|
+
return L.join('\n');
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
// The check title and the human line the enforcement step prints.
|
|
340
|
+
//
|
|
341
|
+
// The message names THE MODELS THAT REGRESSED, never the requested list. The
|
|
342
|
+
// old line interpolated the requested list, so one model's regression was
|
|
343
|
+
// reported against every model asked for - the same defect as the selection,
|
|
344
|
+
// seen from the other side.
|
|
345
|
+
// The parser's message still reaches the ANNOTATION, where the detail is worth
|
|
346
|
+
// having - but a workflow command is ONE LINE: a newline inside it ends the
|
|
347
|
+
// command and drops everything after, and an unbounded echo of a file's bytes
|
|
348
|
+
// does not belong in a check annotation either. One line, bounded, cut marked.
|
|
349
|
+
function oneLine(s, max = 160) {
|
|
350
|
+
const flat = String(s).replace(/\s+/g, ' ').trim();
|
|
351
|
+
return flat.length > max ? `${flat.slice(0, max - 1)}\u2026` : flat;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
function enforcementLines(d, { failOnRegression = true } = {}) {
|
|
355
|
+
const L = [];
|
|
356
|
+
const per = d.rows.map((r) => `${r.model}: ${r.state}`).join(', ');
|
|
357
|
+
L.push(`Driftproof decision over ${d.receiptCount} receipt(s) for ${d.requestedCount} model(s) - ${per}`);
|
|
358
|
+
// AC-2: the count is reported before any verdict, and names each model that
|
|
359
|
+
// reached no readable receipt. Absence is not a pass.
|
|
360
|
+
//
|
|
361
|
+
// The two causes are reported apart (absence-vs-unreadable). A model with NO
|
|
362
|
+
// receipt and a model whose receipt exists and cannot be parsed both fail
|
|
363
|
+
// closed, but they send the reader to different places - one to the run that
|
|
364
|
+
// never wrote it, one to the file that is there - so a single message naming
|
|
365
|
+
// both as "no receipt was produced" would be false about one of them.
|
|
366
|
+
if (d.absent.length) {
|
|
367
|
+
L.push(`::error title=Driftproof::No receipt was produced for ${d.absent.join(', ')} - `
|
|
368
|
+
+ `${d.receiptCount} receipt(s) for ${d.requestedCount} requested model(s). `
|
|
369
|
+
+ `Failing closed: a model that was not measured did not pass.`);
|
|
370
|
+
}
|
|
371
|
+
if (d.unreadable.length) {
|
|
372
|
+
L.push(`::error title=Driftproof::The receipt for `
|
|
373
|
+
+ `${d.unreadable.map((u) => `${u.model} exists and is unreadable (${u.file}: ${oneLine(u.error)})`).join('; ')}. `
|
|
374
|
+
+ `Failing closed: a receipt that cannot be read is not a pass, and is not the same as one that was never written.`);
|
|
375
|
+
}
|
|
376
|
+
if (d.regressed.length) {
|
|
377
|
+
const deltas = d.rows.filter((r) => r.state === 'regression')
|
|
378
|
+
.map((r) => `${r.model} (delta ${r.delta})`).join(', ');
|
|
379
|
+
if (failOnRegression) {
|
|
380
|
+
L.push(`::error title=Driftproof::Skill REGRESSED on ${deltas}`);
|
|
381
|
+
} else {
|
|
382
|
+
L.push(`::warning title=Driftproof::Skill REGRESSED on ${deltas} (fail-on-regression is false)`);
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
// AC-4: an unmeasured run renders as unmeasured, never as passing, and does
|
|
386
|
+
// not fail the job on its own.
|
|
387
|
+
const unmeasured = d.rows.filter((r) => NEVER_SUCCESS.includes(r.state));
|
|
388
|
+
if (unmeasured.length) {
|
|
389
|
+
// AC-4 binds the TITLE: a `::warning` "whose title names the state and the
|
|
390
|
+
// models it came from". Until this it named `unmeasured[0].state` and no
|
|
391
|
+
// model at all - both facts were in the MESSAGE, which is not where the
|
|
392
|
+
// criterion puts them, and with two never-success states present the title
|
|
393
|
+
// carried only the first row's. A reader who sees the annotation collapsed
|
|
394
|
+
// to its title in a check list saw neither the second state nor any model
|
|
395
|
+
// (F-4 of 2026-09-12).
|
|
396
|
+
//
|
|
397
|
+
// Grouped by state, worst first, from STATE_ORDER - the same single ordering
|
|
398
|
+
// AC-1's enforcement and AC-3's badge read, so the title cannot disagree with
|
|
399
|
+
// them about which state governs.
|
|
400
|
+
//
|
|
401
|
+
// NO COMMA IN THE TITLE, and that is load-bearing rather than a style
|
|
402
|
+
// choice: GitHub parses a workflow command's properties as a
|
|
403
|
+
// COMMA-SEPARATED list, so `title=not measured,inconclusive` ends the title
|
|
404
|
+
// at the comma and leaves the rest to be read as a property GitHub does not
|
|
405
|
+
// know. States are joined with ' / ' and the models under one state with
|
|
406
|
+
// ' + ', neither of which GitHub reads.
|
|
407
|
+
const states = STATE_ORDER.filter((s) => unmeasured.some((r) => r.state === s));
|
|
408
|
+
const title = states
|
|
409
|
+
.map((s) => `${s} on ${unmeasured.filter((r) => r.state === s).map((r) => r.model).join(' + ')}`)
|
|
410
|
+
.join(' / ');
|
|
411
|
+
L.push(`::warning title=Driftproof: ${title}::`
|
|
412
|
+
+ `${unmeasured.map((r) => `${r.model}: ${r.state}`).join(', ')} - `
|
|
413
|
+
+ `this run did not measure these models, and is not a pass on them.`);
|
|
414
|
+
}
|
|
415
|
+
return L;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
module.exports.RENDER = RENDER;
|
|
419
|
+
module.exports.rendersAsSuccess = rendersAsSuccess;
|
|
420
|
+
module.exports.badgeMessage = badgeMessage;
|
|
421
|
+
module.exports.badgeEndpointForSet = badgeEndpointForSet;
|
|
422
|
+
module.exports.githubOutputLines = githubOutputLines;
|
|
423
|
+
module.exports.summaryMarkdown = summaryMarkdown;
|
|
424
|
+
module.exports.enforcementLines = enforcementLines;
|
|
425
|
+
module.exports.VERDICT_WORD = VERDICT_WORD;
|
package/lib/diff.js
CHANGED
|
@@ -59,7 +59,10 @@ function withSkillBands(receipt) {
|
|
|
59
59
|
// which is exactly the v0.1 weakness v0.2 fixes).
|
|
60
60
|
function aggWithBand(receipt) {
|
|
61
61
|
const a = receipt.results.aggregates.with_skill;
|
|
62
|
-
|
|
62
|
+
// v0.6 (spec 026 AC-7): a null band is carried as null and rendered as what
|
|
63
|
+
// it is; a legacy receipt with no aggregate band keeps its 0 (the archive's
|
|
64
|
+
// rendering is unmoved, AC-16).
|
|
65
|
+
return { mean: a.mean_score, stddev: a.stddev === null ? null : (a.stddev || 0), cases: a.case_count };
|
|
63
66
|
}
|
|
64
67
|
|
|
65
68
|
function fmt(n) { return n == null ? 'n/a' : (n >= 0 ? '+' : '') + n.toFixed(3); }
|
|
@@ -81,11 +84,41 @@ function pct(n) { return n == null ? 'n/a' : n.toFixed(3); }
|
|
|
81
84
|
// not from one case's draws).
|
|
82
85
|
|
|
83
86
|
function bandStr(x) {
|
|
87
|
+
if (x && x.mean == null) return 'n/a (0 cases)';
|
|
88
|
+
if (x && x.stddev == null) return `${x.mean.toFixed(3)} ± n/a (${x.cases === 1 ? '1 case' : 'no band'})`;
|
|
84
89
|
const label = x && x.source;
|
|
85
90
|
return `${x.mean.toFixed(3)} ± ${x.stddev.toFixed(3)}${label ? ` (${label})` : ''}`;
|
|
86
91
|
}
|
|
87
92
|
function short(h) { return h ? String(h).slice(0, 12) : 'n/a'; }
|
|
88
93
|
|
|
94
|
+
// ── THE JUDGE IS PART OF THE INSTRUMENT (spec 026 AC-12, A4) ─────────────────
|
|
95
|
+
// Two receipts graded by different judges are two measurements with different
|
|
96
|
+
// instruments, and a verdict across them says nothing about the skill.
|
|
97
|
+
// lib/reuse.js's triage already says a changed judge means regrade; the
|
|
98
|
+
// differ did not ask. It asks here: run.judge.model_id (on canonical ids, so
|
|
99
|
+
// an alias and its dated form are one judge), run.judge.prompt_template_hash,
|
|
100
|
+
// run.judge.temperature, and every shared case's judge.rubric_hash. A field
|
|
101
|
+
// that differs names itself in the caveats and suppresses every per-case
|
|
102
|
+
// verdict. A pre-v0.6 receipt carries no template hash: the report says the
|
|
103
|
+
// template is unrecorded on that side and still compares the rest.
|
|
104
|
+
const canonicalJudgeId = (id) => String(id == null ? '' : id).replace(/-\d{8}$/, '');
|
|
105
|
+
function judgeDiffers(a, b, ids, aB, bB) {
|
|
106
|
+
const ja = (a.run && a.run.judge) || {}; const jb = (b.run && b.run.judge) || {};
|
|
107
|
+
const problems = [];
|
|
108
|
+
const idA = ja.model_id || (a.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
109
|
+
const idB = jb.model_id || (b.results.cases.find((c) => c.judge && c.judge.model_id) || { judge: {} }).judge.model_id;
|
|
110
|
+
if (idA && idB && canonicalJudgeId(idA) !== canonicalJudgeId(idB)) problems.push(`run.judge.model_id differs (${idA} vs ${idB})`);
|
|
111
|
+
const unrecorded = [];
|
|
112
|
+
if (!ja.prompt_template_hash) unrecorded.push('A'); if (!jb.prompt_template_hash) unrecorded.push('B');
|
|
113
|
+
if (ja.prompt_template_hash && jb.prompt_template_hash && ja.prompt_template_hash !== jb.prompt_template_hash) problems.push(`run.judge.prompt_template_hash differs (${short(ja.prompt_template_hash)} vs ${short(jb.prompt_template_hash)}): the grading template is a different judge`);
|
|
114
|
+
if (ja.temperature !== undefined && jb.temperature !== undefined && ja.temperature !== jb.temperature) problems.push(`run.judge.temperature differs (${ja.temperature} vs ${jb.temperature})`);
|
|
115
|
+
const rubricA = Object.fromEntries(a.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
116
|
+
const rubricB = Object.fromEntries(b.results.cases.filter((c) => c.mode === 'with_skill' && c.judge).map((c) => [c.id, c.judge.rubric_hash]));
|
|
117
|
+
const moved = ids.filter((id) => rubricA[id] && rubricB[id] && rubricA[id] !== rubricB[id]);
|
|
118
|
+
if (moved.length) problems.push(`judge.rubric_hash differs on ${moved.length} shared case(s) (${moved.slice(0, 3).map((x) => `\`${x}\``).join(', ')}${moved.length > 3 ? ', …' : ''})`);
|
|
119
|
+
return { problems, unrecorded };
|
|
120
|
+
}
|
|
121
|
+
|
|
89
122
|
// The headline is a SUMMARY of the per-case band-overlap verdicts — the credibility
|
|
90
123
|
// core lives per case (judge-sample bands), and the headline just aggregates it.
|
|
91
124
|
// It deliberately does NOT run a separate band test on the aggregate mean: the
|
|
@@ -146,9 +179,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
146
179
|
// improvement verdicts are never computed from evidence we did not run.
|
|
147
180
|
const levelOf = (r) => r.verification_level || 'TESTED';
|
|
148
181
|
const belowTested = [[labelA, a], [labelB, b]].filter(([, r]) => levelOf(r) !== 'TESTED');
|
|
182
|
+
// Spec 026 AC-4: an INCOMPLETE receipt on either side computes no per-case
|
|
183
|
+
// verdict. RECEIPT.md has said so since v0.3.1; the differ did not ask.
|
|
184
|
+
const incompleteSides = [[labelA, a], [labelB, b]].filter(([, r]) => r.run && r.run.status === 'incomplete');
|
|
185
|
+
// Spec 026 AC-12: a different judge on either side suppresses the verdict.
|
|
186
|
+
const judge = judgeDiffers(a, b, ids, aB, bB);
|
|
187
|
+
const judgeProblem = judge.problems.length ? judge.problems.join('; ') : null;
|
|
149
188
|
// A REFUSAL IS A RESULT, and it suppresses the verdict exactly as an
|
|
150
189
|
// untested input does: no delta is asserted, and the reason travels with it.
|
|
151
|
-
const measured = belowTested.length === 0 && !refusal;
|
|
190
|
+
const measured = belowTested.length === 0 && !refusal && incompleteSides.length === 0 && !judgeProblem;
|
|
152
191
|
|
|
153
192
|
const perCase = ids.map((id) => {
|
|
154
193
|
const before = aB[id] || null;
|
|
@@ -250,6 +289,15 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
250
289
|
if (belowTested.length) {
|
|
251
290
|
warnings.push(`verdicts NOT computed — ${belowTested.map(([l, r]) => `${l} is ${levelOf(r)}${r.run && r.run.source ? ` (${r.run.source})` : ''}`).join('; ')}. Drift verdicts require TESTED receipts on both sides; declared numbers are shown as context only (see /interop.html).`);
|
|
252
291
|
}
|
|
292
|
+
if (incompleteSides.length) {
|
|
293
|
+
warnings.push(`verdicts NOT computed — ${incompleteSides.map(([l, r]) => `${l} is incomplete (run.status incomplete; it excluded ${r.run.failed_case_count || 0} case(s) whose arm could not be measured)`).join('; ')}. A receipt with an unmeasured arm is not evidence for a per-case verdict; its aggregates are shown as context only.`);
|
|
294
|
+
}
|
|
295
|
+
if (judgeProblem) {
|
|
296
|
+
warnings.push(`verdicts NOT computed — the two receipts were graded by different judges: ${judgeProblem}. A verdict across two instruments says nothing about the skill; regrade one side with the other's judge (lib/reuse.js triage: regrade).`);
|
|
297
|
+
}
|
|
298
|
+
if (judge.unrecorded.length) {
|
|
299
|
+
warnings.push(`judge template unrecorded on ${judge.unrecorded.map((x) => (x === 'A' ? labelA : labelB)).join(' and ')} (a pre-v0.6 receipt carries no run.judge.prompt_template_hash); the judge model id and the rubric hashes are compared, the template is not.`);
|
|
300
|
+
}
|
|
253
301
|
// Cross-provider / cross-surface disclosure (Phase 6). A comparison across
|
|
254
302
|
// providers is a skill-DURABILITY comparison across substrates, not model drift
|
|
255
303
|
// over time; across surfaces, sampling control differs. Both are flagged so a
|
|
@@ -281,7 +329,11 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
281
329
|
// THE HEADLINE NAMES THE CAUSE THAT ACTUALLY APPLIES. `belowTested.length`
|
|
282
330
|
// was printed unconditionally, so a refused pair got "0 receipt(s) below
|
|
283
331
|
// TESTED" — a statement measurably false of the receipts it was given.
|
|
284
|
-
if (
|
|
332
|
+
if (incompleteSides.length) {
|
|
333
|
+
// An incomplete side is a fact about that receipt alone (spec 026 AC-4)
|
|
334
|
+
// and leads whatever else is wrong with the pair.
|
|
335
|
+
L.push(`**NOT MEASURED — ${incompleteSides.map(([l]) => l).join(' and ')} incomplete: a case's arm could not be measured, so no per-case verdict is computed.**`);
|
|
336
|
+
} else if (refusal) {
|
|
285
337
|
// The reason is a complete sentence and already ends by saying no verdict
|
|
286
338
|
// is asserted; prefixing that again produced "REFUSED — no verdict is
|
|
287
339
|
// asserted. the baseline arm did not reproduce…" — a duplicated clause and
|
|
@@ -289,6 +341,8 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
289
341
|
L.push(`**REFUSED — ${refusal.reason}**`);
|
|
290
342
|
} else if (belowTested.length) {
|
|
291
343
|
L.push(`**NOT MEASURED — ${belowTested.length} receipt(s) below TESTED. Drift verdicts require TESTED receipts on both sides; the declared numbers above are context, not band-verified evidence.**`);
|
|
344
|
+
} else if (judgeProblem) {
|
|
345
|
+
L.push(`**NOT MEASURED — different judges: ${judgeProblem}. No verdict crosses two instruments.**`);
|
|
292
346
|
} else {
|
|
293
347
|
L.push('**NOT MEASURED — no verdict is asserted.**');
|
|
294
348
|
}
|
|
@@ -327,4 +381,4 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
327
381
|
};
|
|
328
382
|
}
|
|
329
383
|
|
|
330
|
-
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem };
|
|
384
|
+
module.exports = { buildDriftReport, withSkillBands, revisionPairProblem, judgeDiffers };
|
package/lib/importers.js
CHANGED
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
// compatibility contract) are documented in docs/interop.md.
|
|
18
18
|
|
|
19
19
|
const { RECEIPT_SCHEMA_VERSION, RUNNER_VERSION, SUITE_FORMAT } = require('../config');
|
|
20
|
-
const { sealReceipt } = require('./receipt');
|
|
20
|
+
const { sealReceipt, BAND_RULE } = require('./receipt');
|
|
21
21
|
const { mean, stddev, round, combineUncertainty, aggregateBands } = require('./stats');
|
|
22
22
|
const { outcomeFor } = require('./run');
|
|
23
23
|
const { inferProvider } = require('./provider');
|
|
@@ -64,11 +64,16 @@ function importedReceipt({ tool, skillName, skillVersion, suiteFormat, caseCount
|
|
|
64
64
|
date_utc: dateUtc,
|
|
65
65
|
registry: registryStatus(modelId),
|
|
66
66
|
transcripts: 'none',
|
|
67
|
-
|
|
67
|
+
// v0.6: the source tool's grader is the judge that ran; its template is
|
|
68
|
+
// not ours to hash (null, the one place the schema allows it).
|
|
69
|
+
judge: { ...judgeBlock, model_id: (cases.find((c) => c.judge && c.judge.model_id) || { judge: { model_id: 'unknown' } }).judge.model_id, prompt_template_hash: null },
|
|
70
|
+
// v0.6 (spec 026 AC-1): what answered. An external tool's harness did;
|
|
71
|
+
// nothing here attested a model, and nothing was spawned by us.
|
|
72
|
+
answered_by: { kind: 'external', attested: false, reported_model: null, reported_models: null, isolation: 'none' },
|
|
68
73
|
},
|
|
69
74
|
results: {
|
|
70
75
|
cases,
|
|
71
|
-
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline) },
|
|
76
|
+
aggregates: { with_skill: aggregateMode(withSkill), baseline: aggregateMode(baseline), band_rule: BAND_RULE },
|
|
72
77
|
},
|
|
73
78
|
comparison,
|
|
74
79
|
verification_level: 'DECLARED',
|