driftproof 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -12
- package/bin/driftproof +343 -39
- package/config/models.json +3 -2
- package/config.js +2 -2
- package/lib/counts.js +104 -0
- package/lib/decision.js +72 -31
- package/lib/diff.js +12 -3
- package/lib/export.js +6 -2
- package/lib/importers-anthropic.js +451 -0
- package/lib/importers.js +50 -13
- package/lib/init.js +3 -2
- package/lib/receipt.js +182 -10
- package/lib/regrade.js +266 -0
- package/lib/reuse.js +144 -9
- package/lib/run.js +39 -4
- package/lib/skill.js +70 -15
- package/lib/stale.js +194 -0
- package/lib/verdict.js +50 -16
- package/package.json +1 -1
- package/spec/RECEIPT.md +103 -13
- package/spec/receipt.schema.json +330 -15
- package/spec/receipt.v0.7.schema.json +1563 -0
- package/spec/receipt.v0.8.schema.json +1677 -0
- package/spec/stale.v1.schema.json +353 -0
package/lib/counts.js
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Receipt counts (spec 043, receipt spec v0.8). Two counts describe how a receipt's data
|
|
5
|
+
// came to be, and they are kept apart because they measure different things (Report 006:
|
|
6
|
+
// generation spread ran 3 to 7 times judge spread on the same cases):
|
|
7
|
+
//
|
|
8
|
+
// generations_per_arm independent candidate generations
|
|
9
|
+
// judge_samples_per_generation judge samples taken of each generation
|
|
10
|
+
//
|
|
11
|
+
// A count sits at the narrowest scope that is honest about it: run.counts when it is
|
|
12
|
+
// the same for every case of every arm generated in this run, run.arms.<mode>.counts
|
|
13
|
+
// when it holds for one arm, results.cases[i].counts otherwise. An absent count is
|
|
14
|
+
// unknown, never 1 (agentskills discussion #544, comment 18409751, rule 1).
|
|
15
|
+
|
|
16
|
+
const KINDS = ['generations_per_arm', 'judge_samples_per_generation'];
|
|
17
|
+
|
|
18
|
+
// An arm that carries its own generated_at was generated in an earlier run. run.counts
|
|
19
|
+
// describes this run's generations, so an archived arm never inherits it.
|
|
20
|
+
function archived(receipt, mode) {
|
|
21
|
+
const a = receipt.run && receipt.run.arms && receipt.run.arms[mode];
|
|
22
|
+
return !!(a && a.generated_at);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const has = (o, k) => !!o && Object.prototype.hasOwnProperty.call(o, k) && Number.isInteger(o[k]);
|
|
26
|
+
|
|
27
|
+
// Where a reader looks, narrowest first.
|
|
28
|
+
const ORDER = ['case', 'arm', 'run'];
|
|
29
|
+
const SCOPES = {
|
|
30
|
+
case: (r, c) => c.counts,
|
|
31
|
+
arm: (r, c) => { const a = r.run && r.run.arms && r.run.arms[c.mode]; return a && a.counts; },
|
|
32
|
+
run: (r) => r.run && r.run.counts,
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
// The count of `kind` for results.cases[index], or null when the receipt does not
|
|
36
|
+
// establish it. run.judge.samples is the run-scope judge count (R-5), read only where
|
|
37
|
+
// run.counts is silent and only for an arm generated in this run.
|
|
38
|
+
function resolveCount(receipt, index, kind) {
|
|
39
|
+
if (!KINDS.includes(kind)) throw new Error(`unknown count kind: ${kind}`);
|
|
40
|
+
const c = receipt.results.cases[index];
|
|
41
|
+
for (const scope of ORDER) {
|
|
42
|
+
const at = SCOPES[scope](receipt, c);
|
|
43
|
+
if (has(at, kind)) return at[kind];
|
|
44
|
+
if (scope === 'run' && kind === 'judge_samples_per_generation' && receipt.run && receipt.run.judge && Number.isInteger(receipt.run.judge.samples)) return receipt.run.judge.samples;
|
|
45
|
+
if (scope === 'arm' && archived(receipt, c.mode)) return null;
|
|
46
|
+
}
|
|
47
|
+
return null;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Which model generated an arm, and when: the arm's override, else the run's.
|
|
51
|
+
function resolveArm(receipt, mode) {
|
|
52
|
+
const a = (receipt.run && receipt.run.arms && receipt.run.arms[mode]) || {};
|
|
53
|
+
return {
|
|
54
|
+
model_id: a.model_id || receipt.run.model_id,
|
|
55
|
+
generated_at: a.generated_at || receipt.run.generated_at || null,
|
|
56
|
+
archived: archived(receipt, mode),
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Write counts at the narrowest honest scope (spec 043 AC-3). `perCase` is one entry per
|
|
61
|
+
// results.cases index: { index, counts: { <kind>: integer } } with a kind left out when
|
|
62
|
+
// the writer cannot establish it. Mutates and returns the receipt; call before sealing.
|
|
63
|
+
function placeCounts(receipt, perCase) {
|
|
64
|
+
const cases = receipt.results.cases;
|
|
65
|
+
const fresh = perCase.filter((p) => !archived(receipt, cases[p.index].mode));
|
|
66
|
+
const put = (obj, kind, v) => { obj.counts = { ...(obj.counts || {}), [kind]: v }; };
|
|
67
|
+
for (const kind of KINDS) {
|
|
68
|
+
const entries = fresh.map((p) => ({ p, v: p.counts && Number.isInteger(p.counts[kind]) ? p.counts[kind] : null }));
|
|
69
|
+
if (!entries.length) continue;
|
|
70
|
+
const allKnown = entries.length === cases.filter((c) => !archived(receipt, c.mode)).length && entries.every((e) => e.v !== null);
|
|
71
|
+
const values = new Set(entries.map((e) => e.v));
|
|
72
|
+
if (allKnown && values.size === 1) { receipt.run.counts = { ...(receipt.run.counts || {}), [kind]: entries[0].v }; continue; }
|
|
73
|
+
const byMode = new Map();
|
|
74
|
+
for (const e of entries) { const m = cases[e.p.index].mode; if (!byMode.has(m)) byMode.set(m, []); byMode.get(m).push(e); }
|
|
75
|
+
for (const [mode, es] of byMode) {
|
|
76
|
+
const modeKnown = allKnown && es.length === cases.filter((c) => c.mode === mode).length;
|
|
77
|
+
if (modeKnown && new Set(es.map((e) => e.v)).size === 1) {
|
|
78
|
+
receipt.run.arms = receipt.run.arms || {};
|
|
79
|
+
receipt.run.arms[mode] = receipt.run.arms[mode] || {};
|
|
80
|
+
put(receipt.run.arms[mode], kind, es[0].v);
|
|
81
|
+
} else {
|
|
82
|
+
for (const e of es) if (e.v !== null) put(cases[e.p.index], kind, e.v);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return receipt;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// The judge-sample count, or null when the receipt does not establish one at run scope.
|
|
90
|
+
function judgeSamplesOrNull(receipt) {
|
|
91
|
+
const j = receipt && receipt.run && receipt.run.judge;
|
|
92
|
+
if (j && Number.isInteger(j.samples)) return j.samples;
|
|
93
|
+
const c = receipt && receipt.run && receipt.run.counts;
|
|
94
|
+
return has(c, 'judge_samples_per_generation') ? c.judge_samples_per_generation : null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// v0.8 loader rule (spec 043 AC-2): the two spellings of the run's judge count agree.
|
|
98
|
+
function judgeCountsDisagree(receipt) {
|
|
99
|
+
const j = receipt && receipt.run && receipt.run.judge;
|
|
100
|
+
const c = receipt && receipt.run && receipt.run.counts;
|
|
101
|
+
return !!(j && Number.isInteger(j.samples) && has(c, 'judge_samples_per_generation') && c.judge_samples_per_generation !== j.samples);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
module.exports = { KINDS, ORDER, resolveCount, resolveArm, placeCounts, judgeSamplesOrNull, judgeCountsDisagree };
|
package/lib/decision.js
CHANGED
|
@@ -17,10 +17,10 @@
|
|
|
17
17
|
// here, in lib/, rather than in the shell, because `bin/driftproof badge` needs
|
|
18
18
|
// it too and because a decision state is a property of a receipt.
|
|
19
19
|
|
|
20
|
-
const fs = require('fs');
|
|
21
20
|
const path = require('path');
|
|
22
21
|
const { EFFECT_FLOOR } = require('../config');
|
|
23
|
-
const { shortModel, githubOutputEntry, receiptVerdict, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
22
|
+
const { shortModel, githubOutputEntry, receiptVerdict, readCases, drawsLine, UNDERPOWERED_LINE } = require('./verdict');
|
|
23
|
+
const { duplicateCaseRows, ambiguityLine, suiteNarrowed, listReceipts, fileSlug } = require('./receipt');
|
|
24
24
|
|
|
25
25
|
// The six decision states (spec 030 AC-4).
|
|
26
26
|
//
|
|
@@ -86,40 +86,57 @@ function failsJob(state, { failOnRegression = true } = {}) {
|
|
|
86
86
|
// the fail-safe direction.
|
|
87
87
|
function decisionState(receipt) {
|
|
88
88
|
if (!receipt) return 'refused';
|
|
89
|
+
// Spec 050 AC-3: a receipt with two rows for one (id, mode) is `refused` before
|
|
90
|
+
// anything else is read. It is not `not measured`, which does not fail a job: the
|
|
91
|
+
// receipt may well have measured a regression, and which row a reader keeps decides
|
|
92
|
+
// whether that regression is seen. `refused` fails closed whatever
|
|
93
|
+
// fail-on-regression says, which is the direction an ambiguity has to fail.
|
|
94
|
+
if (duplicateCaseRows(receipt).length) return 'refused';
|
|
89
95
|
const level = receipt.verification_level;
|
|
90
96
|
const kind = receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
|
|
91
97
|
if (level !== 'TESTED' || kind !== 'model') return 'not measured';
|
|
98
|
+
// Spec 062 (register row 1): THE MEASURED CASES BEFORE THE INCOMPLETE RULE. Until this an
|
|
99
|
+
// incomplete run read `inconclusive` before any case was read, so one unmeasured case hid a
|
|
100
|
+
// regression another case had measured: the same receipt read `regression`, exit 1, with every
|
|
101
|
+
// case measured, and `inconclusive`, exit 0, with one with-skill arm unmeasured (the register's
|
|
102
|
+
// pass 2 probe 7). A case that separated down was measured whatever happened to the others.
|
|
103
|
+
if (readCases(receipt).some((c) => c.state === 'separated-down')) return 'regression';
|
|
92
104
|
const cmp = receipt.comparison || {};
|
|
93
|
-
|
|
105
|
+
// Incomplete: a case failed (run.status), or the receipt ran fewer cases than its suite holds
|
|
106
|
+
// (spec 062, register row 4).
|
|
107
|
+
if ((receipt.run && receipt.run.status === 'incomplete') || suiteNarrowed(receipt) || typeof cmp.delta !== 'number') return 'inconclusive';
|
|
94
108
|
// Spec 035: the measured states are the receipt verdict's, read per case by the
|
|
95
109
|
// band rule (lib/verdict.js receiptVerdict), not the aggregate delta against the floor.
|
|
96
|
-
|
|
97
|
-
//
|
|
98
|
-
|
|
110
|
+
const verdict = receiptVerdict(receipt).verdict;
|
|
111
|
+
// Spec 062: every verdict has a state. NOT_MEASURED is reachable here - a receipt whose every
|
|
112
|
+
// case was dropped reads it by spec 036's no_readable_case rung - and until this it had no row
|
|
113
|
+
// in the table, so the state was undefined and the model left the set (pass 6 V3).
|
|
114
|
+
if (!Object.hasOwn(MEASURED_STATE, verdict)) throw new Error(`decisionState: the verdict ${JSON.stringify(verdict)} has no decision state`);
|
|
115
|
+
return MEASURED_STATE[verdict];
|
|
99
116
|
}
|
|
100
117
|
|
|
101
|
-
const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect' };
|
|
118
|
+
const MEASURED_STATE = { REGRESSED: 'regression', PASSED: 'helped', UNDERPOWERED: 'underpowered', NO_EFFECT: 'no detected effect', NOT_MEASURED: 'not measured' };
|
|
102
119
|
|
|
103
120
|
// The worst state in a set, by STATE_ORDER. An empty set has no decision, and
|
|
104
|
-
// says so with null rather than defaulting to something benign.
|
|
121
|
+
// says so with null rather than defaulting to something benign. A state this
|
|
122
|
+
// ordering does not know is refused (spec 062): skipping it dropped its model
|
|
123
|
+
// from the decision, which is a pass by omission.
|
|
105
124
|
function worstState(states) {
|
|
106
125
|
let worst = null;
|
|
107
126
|
for (const s of states) {
|
|
108
127
|
const i = STATE_ORDER.indexOf(s);
|
|
109
|
-
if (i < 0)
|
|
128
|
+
if (i < 0) throw new Error(`worstState: unknown decision state ${JSON.stringify(s)}`);
|
|
110
129
|
if (worst === null || i < STATE_ORDER.indexOf(worst)) worst = s;
|
|
111
130
|
}
|
|
112
131
|
return worst;
|
|
113
132
|
}
|
|
114
133
|
|
|
115
|
-
// Every receipt in a directory,
|
|
116
|
-
//
|
|
117
|
-
//
|
|
134
|
+
// Every receipt file in a directory, by lib/receipt.js listReceipts (spec 062): the files that
|
|
135
|
+
// parse and carry receipt_hash or results and are no sidecar by name and shape (spec 069), and the
|
|
136
|
+
// files that do not parse. Until spec 062 it was
|
|
137
|
+
// every `.json` but the summaries and badge.json, so a regrade sidecar was read as a receipt.
|
|
118
138
|
function receiptFiles(dir) {
|
|
119
|
-
return
|
|
120
|
-
.filter((f) => f.endsWith('.json') && !f.includes('.summary.') && f !== 'badge.json')
|
|
121
|
-
.sort()
|
|
122
|
-
.map((f) => path.join(dir, f));
|
|
139
|
+
return listReceipts(dir).map((e) => e.file);
|
|
123
140
|
}
|
|
124
141
|
|
|
125
142
|
// Parse the `models` input the same way the runner does: comma-separated, with
|
|
@@ -152,9 +169,11 @@ function matchesModel(receiptModelId, requested) {
|
|
|
152
169
|
function fileNamesModel(file, requested) {
|
|
153
170
|
const base = path.basename(file);
|
|
154
171
|
// Both spellings, for the reason `matchesModel` takes both: a receipt for
|
|
155
|
-
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`.
|
|
172
|
+
// `claude-haiku-4-5-20251001` answers a request for `claude-haiku-4-5`. Each is
|
|
173
|
+
// looked for in the form the runner writes it, lib/receipt.js fileSlug (spec 062): an
|
|
174
|
+
// id with a dot or a capital is not in the file name as itself.
|
|
156
175
|
for (const id of new Set([requested, shortModel(requested)])) {
|
|
157
|
-
if (base.includes(`-${id}-`)) return true;
|
|
176
|
+
if (base.includes(`-${fileSlug(id)}-`)) return true;
|
|
158
177
|
}
|
|
159
178
|
return false;
|
|
160
179
|
}
|
|
@@ -166,37 +185,49 @@ function fileNamesModel(file, requested) {
|
|
|
166
185
|
// row that is simply absent. That is the whole of AC-2: absence is not a pass.
|
|
167
186
|
function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
168
187
|
const requested = parseModels(requestedCsv);
|
|
169
|
-
const
|
|
170
|
-
const
|
|
171
|
-
let receipt = null, error = null;
|
|
172
|
-
try { receipt = JSON.parse(fs.readFileSync(f, 'utf8')); } catch (e) { error = e.message; }
|
|
173
|
-
return { file: f, receipt, error };
|
|
174
|
-
});
|
|
188
|
+
const loaded = listReceipts(dir);
|
|
189
|
+
const files = loaded.map((l) => l.file);
|
|
175
190
|
const used = new Set();
|
|
176
191
|
const rows = requested.map((model) => {
|
|
177
|
-
|
|
192
|
+
// Spec 050 AC-6: EVERY readable receipt that matches, not the first. Until this the
|
|
193
|
+
// row took `loaded.find(...)`, the first match in sorted filename order, and reported
|
|
194
|
+
// the rest as unexpected. A job with two Driftproof steps shared one directory, so
|
|
195
|
+
// the second step's model row was filled by whichever skill's receipt sorted first
|
|
196
|
+
// (an outside correctness audit, finding 2; spec 030's F-1). Two receipts for one
|
|
197
|
+
// model are a question nobody answered, so the row is `refused`, names every file,
|
|
198
|
+
// and reads none of them.
|
|
199
|
+
const matches = loaded.filter((l) => !used.has(l.file) && l.receipt
|
|
178
200
|
&& l.receipt.run && matchesModel(l.receipt.run.model_id, model));
|
|
179
|
-
|
|
201
|
+
for (const m of matches) used.add(m.file);
|
|
202
|
+
const many = matches.length > 1;
|
|
203
|
+
const hit = matches.length === 1 ? matches[0] : null;
|
|
180
204
|
// A receipt that exists and cannot be parsed is UNREADABLE, not absent.
|
|
181
205
|
// Both fail closed, and the row says which - the register's
|
|
182
206
|
// absence-vs-unreadable distinction, kept at the point it is decided.
|
|
183
|
-
const unreadable = !hit && loaded.find((l) => !used.has(l.file) && l.error
|
|
207
|
+
const unreadable = !hit && !many && loaded.find((l) => !used.has(l.file) && l.error
|
|
184
208
|
&& fileNamesModel(l.file, model));
|
|
185
209
|
if (unreadable) used.add(unreadable.file);
|
|
186
210
|
const receipt = hit ? hit.receipt : null;
|
|
211
|
+
// Spec 050 AC-3: one receipt, but one with two rows for a case and arm.
|
|
212
|
+
const dups = receipt ? duplicateCaseRows(receipt) : [];
|
|
213
|
+
const ambiguous = many
|
|
214
|
+
? `${matches.length} receipts for this model (${matches.map((m) => path.basename(m.file)).join(', ')}); none was read`
|
|
215
|
+
: (dups.length ? `receipt ambiguous: ${ambiguityLine(dups)}` : null);
|
|
187
216
|
const state = decisionState(receipt);
|
|
188
217
|
return {
|
|
189
218
|
model,
|
|
190
219
|
state,
|
|
191
|
-
delta:
|
|
220
|
+
// An ambiguous row carries no delta: a delta is a reading, and none was taken.
|
|
221
|
+
delta: !ambiguous && receipt && receipt.comparison && typeof receipt.comparison.delta === 'number'
|
|
192
222
|
? receipt.comparison.delta : null,
|
|
193
|
-
file: hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
|
|
223
|
+
file: many ? null : hit ? path.basename(hit.file) : (unreadable ? path.basename(unreadable.file) : null),
|
|
194
224
|
// Spec 035 AC-3: what an underpowered row would have needed, from the same reading.
|
|
195
225
|
drawsNeeded: state === 'underpowered' ? receiptVerdict(receipt).drawsNeeded : null,
|
|
196
226
|
// The distinction is carried on the ROW, not left in a sentence, because
|
|
197
227
|
// the enforcement message has to say which of the two happened (AC-2's
|
|
198
228
|
// mutation class, absence-vs-unreadable).
|
|
199
229
|
unreadable: unreadable ? unreadable.error : null,
|
|
230
|
+
ambiguous,
|
|
200
231
|
// A SHORT CAUSE, not the parser's text. V8's JSON.parse message echoes the
|
|
201
232
|
// first bytes of the input - "Unexpected token '|', \"|{bad\" is not valid
|
|
202
233
|
// JSON" - and this string is rendered into a markdown table cell, where one
|
|
@@ -204,7 +235,7 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
|
204
235
|
// reaches the annotation, which is not a table; the cell carries the cause,
|
|
205
236
|
// and the filename is already beside it in the same cell (F-2 of
|
|
206
237
|
// 2026-09-12).
|
|
207
|
-
reason: unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model'),
|
|
238
|
+
reason: ambiguous || (unreadable ? 'receipt unreadable (not valid JSON)' : (hit ? null : 'no receipt for this model')),
|
|
208
239
|
};
|
|
209
240
|
});
|
|
210
241
|
// Receipts the run wrote for models nobody requested are reported rather than
|
|
@@ -226,9 +257,14 @@ function decideSet(dir, requestedCsv, { failOnRegression = true } = {}) {
|
|
|
226
257
|
// split it by CAUSE: nothing was written for this model, or something was
|
|
227
258
|
// written and cannot be read. Both fail closed; they are not the same fact,
|
|
228
259
|
// and the failure has to say which.
|
|
229
|
-
absent: rows.filter((r) => r.state === 'refused' && !r.unreadable).map((r) => r.model),
|
|
260
|
+
absent: rows.filter((r) => r.state === 'refused' && !r.unreadable && !r.ambiguous).map((r) => r.model),
|
|
230
261
|
unreadable: rows.filter((r) => r.state === 'refused' && r.unreadable)
|
|
231
262
|
.map((r) => ({ model: r.model, file: r.file, error: r.unreadable })),
|
|
263
|
+
// Spec 050: a third cause of `refused`, apart from the other two for the same reason
|
|
264
|
+
// they are apart from each other. Something was written, and it can be read, but it
|
|
265
|
+
// does not say one thing.
|
|
266
|
+
ambiguous: rows.filter((r) => r.state === 'refused' && r.ambiguous)
|
|
267
|
+
.map((r) => ({ model: r.model, reason: r.ambiguous })),
|
|
232
268
|
unexpected,
|
|
233
269
|
receiptCount: files.length,
|
|
234
270
|
requestedCount: requested.length,
|
|
@@ -390,6 +426,11 @@ function enforcementLines(d, { failOnRegression = true } = {}) {
|
|
|
390
426
|
+ `${d.unreadable.map((u) => `${u.model} exists and is unreadable (${u.file}: ${oneLine(u.error)})`).join('; ')}. `
|
|
391
427
|
+ `Failing closed: a receipt that cannot be read is not a pass, and is not the same as one that was never written.`);
|
|
392
428
|
}
|
|
429
|
+
if (d.ambiguous && d.ambiguous.length) {
|
|
430
|
+
L.push(`::error title=Driftproof::The receipt set is ambiguous for `
|
|
431
|
+
+ `${d.ambiguous.map((a) => `${a.model}: ${oneLine(a.reason, 240)}`).join('; ')}. `
|
|
432
|
+
+ `Failing closed: a decision that depends on which receipt or row was read is not a pass.`);
|
|
433
|
+
}
|
|
393
434
|
if (d.regressed.length) {
|
|
394
435
|
const deltas = d.rows.filter((r) => r.state === 'regression')
|
|
395
436
|
.map((r) => `${r.model} (delta ${r.delta})`).join(', ');
|
package/lib/diff.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const { judgeSamplesOrNull } = require('./counts');
|
|
4
5
|
const { bandVerdict, round, WITHIN_NOISE } = require('./stats');
|
|
5
6
|
const { EFFECT_FLOOR } = require('../config');
|
|
6
7
|
const { revisionHeadline } = require('./revision');
|
|
@@ -158,6 +159,12 @@ function revisionPairProblem(a, b) {
|
|
|
158
159
|
return null;
|
|
159
160
|
}
|
|
160
161
|
|
|
162
|
+
// The run's judge-sample count as a reader shows it: the number, or 'unknown'.
|
|
163
|
+
function judgeN(r) {
|
|
164
|
+
const n = judgeSamplesOrNull(r);
|
|
165
|
+
return n === null ? 'unknown' : n;
|
|
166
|
+
}
|
|
167
|
+
|
|
161
168
|
function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' } = {}) {
|
|
162
169
|
const revision = mode === 'revision';
|
|
163
170
|
|
|
@@ -253,7 +260,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
253
260
|
L.push(`| surface (held) | ${a.run.surface} | ${b.run.surface} |`);
|
|
254
261
|
L.push(`| suite_hash (held) | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
|
|
255
262
|
L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
|
|
256
|
-
L.push(`| judge samples/case | ${(a
|
|
263
|
+
L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
|
|
257
264
|
L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
|
|
258
265
|
L.push(`| baseline score | ${pct(a.comparison.baseline_score)} | ${pct(b.comparison.baseline_score)} |`);
|
|
259
266
|
L.push(`| skill lift (Δ) | ${fmt(a.comparison.delta)} | ${fmt(b.comparison.delta)} |`);
|
|
@@ -268,7 +275,7 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
268
275
|
L.push(`| model | \`${a.run.model_id}\` | \`${b.run.model_id}\` |`);
|
|
269
276
|
L.push(`| run date (UTC) | ${a.run.date_utc} | ${b.run.date_utc} |`);
|
|
270
277
|
L.push(`| surface | ${a.run.surface} | ${b.run.surface} |`);
|
|
271
|
-
L.push(`| judge samples/case | ${(a
|
|
278
|
+
L.push(`| judge samples/case | ${judgeN(a)} | ${judgeN(b)} |`);
|
|
272
279
|
L.push(`| skill content_hash | \`${short(a.skill.content_hash)}\` | \`${short(b.skill.content_hash)}\` |`);
|
|
273
280
|
L.push(`| suite_hash | \`${short(a.suite.suite_hash)}\` | \`${short(b.suite.suite_hash)}\` |`);
|
|
274
281
|
L.push(`| with_skill (mean ± band) | ${bandStr(aAgg)} | ${bandStr(bAgg)} |`);
|
|
@@ -289,7 +296,9 @@ function buildDriftReport(a, b, { labelA = 'A', labelB = 'B', mode = 'release' }
|
|
|
289
296
|
if (!revision && a.skill.content_hash !== b.skill.content_hash) warnings.push('skill content_hash differs — the skill itself changed between receipts, so drift mixes skill edits with model drift.');
|
|
290
297
|
if (!revision && a.suite.suite_hash !== b.suite.suite_hash) warnings.push('suite_hash differs — the eval suite changed; per-case comparison may be misleading.');
|
|
291
298
|
if (a.skill.name !== b.skill.name) warnings.push(`different skills (${a.skill.name} vs ${b.skill.name}) — comparison is not meaningful.`);
|
|
292
|
-
|
|
299
|
+
// v0.8 (spec 043 AC-1): an absent judge count is unknown, said as such, never read as 1.
|
|
300
|
+
if (judgeN(a) === 'unknown' || judgeN(b) === 'unknown') warnings.push('the judge-sample count is unknown on one or both receipts: the receipt does not establish it, so whether the bands are sampled cannot be read from it.');
|
|
301
|
+
if (judgeN(a) === 1 || judgeN(b) === 1) warnings.push('one or both receipts are single-sample (no bands): non-overlap can only be trusted when both sides are sampled.');
|
|
293
302
|
// WHAT SUPPRESSED THE VERDICT IS SAID, AND IT IS SAID CORRECTLY (spec 016 AC-3
|
|
294
303
|
// and AC-4, closing F-015-B).
|
|
295
304
|
//
|
package/lib/export.js
CHANGED
|
@@ -8,13 +8,16 @@
|
|
|
8
8
|
// Documented in docs/interop.md; snapshot-tested in the gate.
|
|
9
9
|
|
|
10
10
|
const { verdictFromReceipt } = require('./verdict');
|
|
11
|
+
const { SUMMARY_SIDECAR } = require('./receipt');
|
|
11
12
|
|
|
12
|
-
const SUMMARY_FORMAT =
|
|
13
|
+
const SUMMARY_FORMAT = SUMMARY_SIDECAR.format;
|
|
13
14
|
const SUMMARY_FORMAT_VERSION = '1';
|
|
14
15
|
|
|
15
16
|
// Build the summary object for one receipt. Deterministic for a given receipt:
|
|
16
17
|
// fixed key order, no export-time timestamps. `reportUrl` is caller-supplied
|
|
17
18
|
// (receipts do not know where their report lives), else null.
|
|
19
|
+
const { judgeSamplesOrNull } = require('./counts');
|
|
20
|
+
|
|
18
21
|
function toSummaryJson(receipt, { reportUrl = null } = {}) {
|
|
19
22
|
const agg = receipt.results.aggregates;
|
|
20
23
|
const cmp = receipt.comparison || {};
|
|
@@ -41,7 +44,8 @@ function toSummaryJson(receipt, { reportUrl = null } = {}) {
|
|
|
41
44
|
source: receipt.run.source || 'driftproof',
|
|
42
45
|
judge: {
|
|
43
46
|
model_id: (receipt.results.cases.find((c) => c.judge) || { judge: { model_id: 'unknown' } }).judge.model_id,
|
|
44
|
-
|
|
47
|
+
// v0.8 (spec 043 AC-1): null when the receipt does not establish the count, never 1.
|
|
48
|
+
samples: judgeSamplesOrNull(receipt),
|
|
45
49
|
},
|
|
46
50
|
receipt_hash: receipt.receipt_hash,
|
|
47
51
|
report_url: reportUrl,
|