driftproof 0.10.2 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -8
- package/bin/driftproof +22 -4
- package/config.js +10 -3
- package/lib/badge-svg.js +130 -0
- package/lib/decision.js +25 -8
- package/lib/diff.js +31 -2
- package/lib/hygiene.js +67 -2
- package/lib/receipt.js +5 -1
- package/lib/verdict.js +197 -26
- package/package.json +1 -1
- package/spec/RECEIPT.md +49 -10
- package/spec/receipt.schema.json +2 -2
- package/spec/receipt.v0.6.schema.json +1563 -0
package/lib/verdict.js
CHANGED
|
@@ -2,21 +2,40 @@
|
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
4
|
const crypto = require('crypto');
|
|
5
|
-
const { EFFECT_FLOOR } = require('../config');
|
|
5
|
+
const { EFFECT_FLOOR, POWER_Z } = require('../config');
|
|
6
|
+
const { bandOf } = require('./reuse');
|
|
6
7
|
|
|
7
8
|
// Single-receipt verdict + shields.io badge.
|
|
8
9
|
//
|
|
9
|
-
// `diff` compares TWO receipts across model releases; a single receipt
|
|
10
|
-
//
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
// EFFECT_FLOOR). Band separation is not applied here — a single run has no second
|
|
14
|
-
// band to separate from — but the effect floor is, so a trivial lift below the
|
|
15
|
-
// judge's quantization grid is reported as "no effect", never as "passing".
|
|
10
|
+
// `diff` compares TWO receipts across model releases; a single receipt carries its
|
|
11
|
+
// own verdict: does the skill still help on THIS model, and could this receipt have
|
|
12
|
+
// told? Spec 035 (R-1 to R-5, spec.md) reads it per case, from the receipt's own
|
|
13
|
+
// fields, with the same band rule the differ applies:
|
|
16
14
|
//
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
15
|
+
// R-1 a case SEPARATES when the with_skill and baseline bands (mean plus or minus
|
|
16
|
+
// one sample sd across draws) do not overlap AND the means differ by at least
|
|
17
|
+
// EFFECT_FLOOR;
|
|
18
|
+
// R-2 its RESOLUTION is s_w + s_b + POWER_Z * sqrt(s_w^2/n_w + s_b^2/n_b);
|
|
19
|
+
// R-3 a case that does not separate is UNDERPOWERED when an arm has fewer than two
|
|
20
|
+
// measured draws or its resolution reaches the floor;
|
|
21
|
+
// R-4 the draws that would have been needed, per arm, with the spreads held;
|
|
22
|
+
// R-5 REGRESSED when any case separated downward, else PASSED when any separated
|
|
23
|
+
// upward, else UNDERPOWERED when any case is underpowered, else NO_EFFECT.
|
|
24
|
+
//
|
|
25
|
+
// Until spec 035 this read only `comparison.delta` against the floor, so a receipt
|
|
26
|
+
// whose aggregate cleared the floor with no case separated rendered passing, and the
|
|
27
|
+
// 0.10.1 reliability audit measured 34.56% of binary-null receipts labelled passing.
|
|
28
|
+
//
|
|
29
|
+
// R-5's four refusals are the ones `lib/decision.js` decisionState has applied since
|
|
30
|
+
// spec 030: not TESTED, `run.answered_by.kind` not `model`, an incomplete run, no
|
|
31
|
+
// numeric delta. This reader used to apply three of them, and to default an absent
|
|
32
|
+
// `verification_level` to TESTED, so a receipt that said nothing about what answered
|
|
33
|
+
// it was badged on its numbers while the Action path read it as `not measured`. Both
|
|
34
|
+
// readers now refuse on an ABSENT field as they refuse on a stated wrong one, which
|
|
35
|
+
// is the fail-safe direction and the only reading under which one receipt cannot get
|
|
36
|
+
// two answers. Every receipt under `receipts/` predates `answered_by` (schema v0.1 to
|
|
37
|
+
// v0.5, the field arrived in v0.6) and is refused by the badge from here on, exactly
|
|
38
|
+
// as `driftproof decide` has always refused it.
|
|
20
39
|
|
|
21
40
|
const VERDICTS = {
|
|
22
41
|
PASSED: { color: 'brightgreen', word: 'passing' },
|
|
@@ -26,42 +45,191 @@ const VERDICTS = {
|
|
|
26
45
|
// delta (a source tool with no baseline mode) never gets a pass/fail verdict
|
|
27
46
|
// — declared numbers are recorded, not verified, so the badge says so.
|
|
28
47
|
NOT_MEASURED: { color: 'lightgrey', word: 'not measured' },
|
|
48
|
+
// Spec 035: the receipt's own spreads and draws cannot resolve a shift of the
|
|
49
|
+
// effect floor. Not a finding about the skill; a finding about this receipt.
|
|
50
|
+
UNDERPOWERED: { color: 'blue', word: 'not enough draws' },
|
|
29
51
|
};
|
|
30
52
|
|
|
53
|
+
// THE PLAIN LINE for the UNDERPOWERED state (spec 035 AC-9). Every surface that
|
|
54
|
+
// renders the state in words carries exactly this sentence.
|
|
55
|
+
const UNDERPOWERED_LINE = 'Not enough draws to conclude at this effect floor';
|
|
56
|
+
|
|
31
57
|
// Strip a trailing -YYYYMMDD date stamp so the badge reads cleanly
|
|
32
58
|
// (claude-haiku-4-5-20251001 → claude-haiku-4-5).
|
|
33
59
|
function shortModel(modelId) {
|
|
34
60
|
return String(modelId || 'unknown').replace(/-\d{8}$/, '');
|
|
35
61
|
}
|
|
36
62
|
|
|
37
|
-
//
|
|
38
|
-
//
|
|
39
|
-
|
|
63
|
+
// One arm of one case, from the receipt's own fields, through the one band
|
|
64
|
+
// definition every comparison path shares (lib/reuse.js bandOf, spec 016 AC-1). A
|
|
65
|
+
// generation-sampled receipt's band counts its measured draws; an older receipt's
|
|
66
|
+
// band is its judge samples over one generation, which is one draw.
|
|
67
|
+
function armOf(c) {
|
|
68
|
+
const band = bandOf(c);
|
|
69
|
+
if (!band) return null;
|
|
70
|
+
return { mean: band.mean, sd: band.sd, n: band.source === 'generation' ? (typeof band.n === 'number' ? band.n : 0) : 1 };
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// R-1 to R-4 for one pair of arms. `a` is the arm the delta is measured from and
|
|
74
|
+
// `b` the arm it is measured to: for a receipt, baseline and with_skill; for the
|
|
75
|
+
// differ, the older receipt's with_skill and the newer one's.
|
|
76
|
+
function caseRule(a, b, { floor = EFFECT_FLOOR, z = POWER_Z } = {}) {
|
|
77
|
+
const delta = b.mean - a.mean;
|
|
78
|
+
// Spec 036 A-036-2: the draws this case actually took, which is the smaller of the two
|
|
79
|
+
// arms' measured draws. The binding arm is the one that would have to be drawn again,
|
|
80
|
+
// so it is the one a reader is owed beside the count that would have been needed.
|
|
81
|
+
const drawsTaken = Math.min(a.n, b.n);
|
|
82
|
+
const apart = b.mean - b.sd > a.mean + a.sd || a.mean - a.sd > b.mean + b.sd;
|
|
83
|
+
if (apart && Math.abs(delta) >= floor) return { state: delta > 0 ? 'separated-up' : 'separated-down', delta, drawsTaken };
|
|
84
|
+
if (a.n < 2 || b.n < 2) return { state: 'underpowered', delta, resolution: null, drawsNeeded: null, reason: 'single_draw', drawsTaken };
|
|
85
|
+
const resolution = a.sd + b.sd + z * Math.sqrt((a.sd * a.sd) / a.n + (b.sd * b.sd) / b.n);
|
|
86
|
+
if (resolution < floor) return { state: 'not-separated', delta, resolution, drawsTaken };
|
|
87
|
+
const spread = a.sd + b.sd;
|
|
88
|
+
if (spread >= floor) return { state: 'underpowered', delta, resolution, drawsNeeded: null, reason: 'spread', spread, drawsTaken };
|
|
89
|
+
const drawsNeeded = Math.max(2, Math.floor((z * z * (a.sd * a.sd + b.sd * b.sd)) / ((floor - spread) ** 2)) + 1);
|
|
90
|
+
return { state: 'underpowered', delta, resolution, drawsNeeded, reason: 'draws', spread, drawsTaken };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// The receipt's draws needed, over its underpowered cases: no draw count when any case's
|
|
94
|
+
// spread alone reaches the floor, else the largest count, else two per arm when a case
|
|
95
|
+
// had a single draw.
|
|
96
|
+
function receiptDrawsNeeded(cases) {
|
|
97
|
+
const u = cases.filter((c) => c.state === 'underpowered');
|
|
98
|
+
if (!u.length) return null;
|
|
99
|
+
const spread = u.find((c) => c.reason === 'spread');
|
|
100
|
+
if (spread) return { value: null, reason: 'spread', case: spread.id, spread: spread.spread, taken: spread.drawsTaken };
|
|
101
|
+
const draws = u.filter((c) => c.reason === 'draws').sort((x, y) => y.drawsNeeded - x.drawsNeeded)[0];
|
|
102
|
+
if (draws) return { value: draws.drawsNeeded, reason: 'draws', case: draws.id, taken: draws.drawsTaken };
|
|
103
|
+
return { value: 2, reason: 'single_draw', case: u[0].id, taken: u[0].drawsTaken };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// Spec 036 A-036-2: the draws a receipt actually took, read WITHOUT the four refusals
|
|
107
|
+
// receiptVerdict applies. A NOT_MEASURED receipt still holds draws, and the badge is
|
|
108
|
+
// owed them: the refusals say the receipt carries no verdict, not that it carries no
|
|
109
|
+
// draws. The figure is the smallest binding arm over the receipt's readable cases, so
|
|
110
|
+
// it is the count that would have to grow. Null when no case has two readable arms.
|
|
111
|
+
function receiptDrawsTaken(receipt) {
|
|
112
|
+
const byId = new Map();
|
|
113
|
+
for (const c of (receipt && receipt.results && receipt.results.cases) || []) {
|
|
114
|
+
if (c.case_status && c.case_status !== 'ok') continue;
|
|
115
|
+
const e = byId.get(c.id) || {};
|
|
116
|
+
e[c.mode] = c;
|
|
117
|
+
byId.set(c.id, e);
|
|
118
|
+
}
|
|
119
|
+
let taken = null;
|
|
120
|
+
for (const [, e] of byId) {
|
|
121
|
+
const w = armOf(e.with_skill);
|
|
122
|
+
const b = armOf(e.baseline);
|
|
123
|
+
if (!w || !b) continue;
|
|
124
|
+
const n = Math.min(w.n, b.n);
|
|
125
|
+
taken = taken === null ? n : Math.min(taken, n);
|
|
126
|
+
}
|
|
127
|
+
return taken;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// The one sentence that says it, for every surface that prints the draws needed.
|
|
131
|
+
function drawsLine(d) {
|
|
132
|
+
if (!d) return null;
|
|
133
|
+
if (d.reason === 'spread') return `Draws needed: no draw count at these spreads (case ${d.case}: the two arms' spreads sum to ${Number(d.spread).toFixed(3)}, at or above the ${EFFECT_FLOOR} floor).`;
|
|
134
|
+
if (d.reason === 'single_draw') return `Draws needed: at least 2 draws per arm to measure a spread (case ${d.case} has one).`;
|
|
135
|
+
return `Draws needed: ${d.value} draws per arm (case ${d.case}), with the spreads held.`;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
// Spec 036 A-036-4: the five routes into NOT_MEASURED, as tokens. Four are R-5's refusals
|
|
139
|
+
// and the fifth is A-036-3's rung, which is reachable only once all four have cleared. The
|
|
140
|
+
// order is the order receiptVerdict tests them, and it is the order a surface states them in.
|
|
141
|
+
const NOT_MEASURED_ROUTES = ['below_tested', 'no_answered_by', 'no_numeric_lift', 'incomplete', 'no_readable_case'];
|
|
142
|
+
|
|
143
|
+
// Every refusal that fired, not the first. A receipt can be silent about what answered it AND
|
|
144
|
+
// mark its run incomplete, and a sentence that names one of those while the other is equally
|
|
145
|
+
// true is thinner than the receipt, which is the direction this reader refuses to go.
|
|
146
|
+
function notMeasuredRoutes(level, kind, delta, incomplete) {
|
|
147
|
+
const r = [];
|
|
148
|
+
if (level !== 'TESTED') r.push('below_tested');
|
|
149
|
+
if (kind !== 'model') r.push('no_answered_by');
|
|
150
|
+
if (typeof delta !== 'number') r.push('no_numeric_lift');
|
|
151
|
+
if (incomplete) r.push('incomplete');
|
|
152
|
+
return r;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
// R-5, all four refusals, read off spec 035 spec.md and identical to the rule
|
|
156
|
+
// lib/decision.js states. An absent field satisfies "not TESTED" and "not model":
|
|
157
|
+
// a receipt that does not say is not taken to have said the reassuring thing.
|
|
158
|
+
function receiptVerdict(receipt) {
|
|
40
159
|
const cmp = (receipt && receipt.comparison) || {};
|
|
41
|
-
const delta = typeof cmp.delta === 'number' ? cmp.delta : 0;
|
|
42
|
-
const model = shortModel(receipt && receipt.run && receipt.run.model_id);
|
|
43
160
|
// Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
|
|
44
161
|
// verdicted — we did not run the suite, so we do not certify the outcome.
|
|
45
|
-
const level =
|
|
162
|
+
const level = receipt && receipt.verification_level;
|
|
163
|
+
// Spec 026 AC-1 made a stub receipt unable to claim TESTED, and v0.6 made a
|
|
164
|
+
// TESTED receipt say what answered it. A receipt answered by a tool, a stub or
|
|
165
|
+
// anything that is not a model has not measured a model, so its numbers are
|
|
166
|
+
// recorded rather than verified and the badge says so.
|
|
167
|
+
const kind = receipt && receipt.run && receipt.run.answered_by ? receipt.run.answered_by.kind : undefined;
|
|
46
168
|
// Spec 026 AC-4: an INCOMPLETE receipt (its run block's status field reads
|
|
47
169
|
// incomplete: a case had an arm that could not be measured) is not verdicted
|
|
48
170
|
// either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
|
|
49
171
|
// compute a verdict from one; the badge is the same reader with a shorter path.
|
|
50
172
|
const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
173
|
+
// Spec 036 A-036-4: WHICH REFUSAL FIRED. The guard below is one line and says only that
|
|
174
|
+
// some refusal did; a surface that renders NOT_MEASURED in words owes its reader the one
|
|
175
|
+
// that actually fired, and cannot get it from a boolean. `notMeasuredRoutes` names the same
|
|
176
|
+
// four conditions as tokens, in the order they are tested, and carries EVERY route that
|
|
177
|
+
// fired rather than the first: two can fire at once, and 2 of the 120 archived receipts
|
|
178
|
+
// are exactly that case (absent answered_by and an incomplete run).
|
|
179
|
+
//
|
|
180
|
+
// The guard is written out in full beside it rather than as `routes.length` so that the
|
|
181
|
+
// four conditions stay one readable line. The two therefore state the same rule twice, and
|
|
182
|
+
// the drift that invites is held by AC-8's row, which reads the shipped routes against an
|
|
183
|
+
// oracle that transcribes them independently, over the archive, the five state fixtures and
|
|
184
|
+
// the rung edges. A guard and a route list that disagree read RED there.
|
|
185
|
+
const routes = notMeasuredRoutes(level, kind, cmp.delta, incomplete);
|
|
186
|
+
if (level !== 'TESTED' || kind !== 'model' || typeof cmp.delta !== 'number' || incomplete) return { verdict: 'NOT_MEASURED', cases: [], drawsNeeded: null, notMeasured: routes };
|
|
187
|
+
const byId = new Map();
|
|
188
|
+
for (const c of (receipt.results && receipt.results.cases) || []) {
|
|
189
|
+
if (c.case_status && c.case_status !== 'ok') continue;
|
|
190
|
+
const e = byId.get(c.id) || {};
|
|
191
|
+
e[c.mode] = c;
|
|
192
|
+
byId.set(c.id, e);
|
|
193
|
+
}
|
|
194
|
+
const cases = [];
|
|
195
|
+
for (const [id, e] of byId) {
|
|
196
|
+
const w = armOf(e.with_skill);
|
|
197
|
+
const b = armOf(e.baseline);
|
|
198
|
+
if (!w || !b) continue;
|
|
199
|
+
cases.push({ id, ...caseRule(b, w) });
|
|
200
|
+
}
|
|
201
|
+
// Spec 036 A-036-3, THE RUNG R-5 NEVER HAD. R-5 reads its ladder off the cases, and
|
|
202
|
+
// every rung below REGRESSED is a statement about what the cases showed. NO_EFFECT is
|
|
203
|
+
// the bottom rung and it is a MEASUREMENT: "no separation detected at this sample size".
|
|
204
|
+
// A receipt whose every case was dropped -- case_status not ok, or an arm with no
|
|
205
|
+
// readable band -- reaches that rung with an EMPTY cases array, so each has() is false
|
|
206
|
+
// and the ladder fell through to NO_EFFECT. Nothing was measured, so the reassuring
|
|
207
|
+
// reading was the one a silent receipt got, which is the exact failure the four
|
|
208
|
+
// refusals above exist to prevent: a receipt that does not say is not taken to have
|
|
209
|
+
// said the reassuring thing. Zero readable cases is NOT_MEASURED, and it is checked
|
|
210
|
+
// BEFORE the ladder so no rung can be reached on no evidence.
|
|
211
|
+
if (!cases.length) return { verdict: 'NOT_MEASURED', cases: [], drawsNeeded: null, notMeasured: ['no_readable_case'] };
|
|
212
|
+
const has = (s) => cases.some((c) => c.state === s);
|
|
213
|
+
const verdict = has('separated-down') ? 'REGRESSED' : has('separated-up') ? 'PASSED' : has('underpowered') ? 'UNDERPOWERED' : 'NO_EFFECT';
|
|
214
|
+
return { verdict, cases, drawsNeeded: verdict === 'UNDERPOWERED' ? receiptDrawsNeeded(cases) : null, notMeasured: null };
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Derive the verdict object from a receipt. Returns
|
|
218
|
+
// { verdict, delta, floor, model, message, color, drawsNeeded }
|
|
219
|
+
function verdictFromReceipt(receipt) {
|
|
220
|
+
const cmp = (receipt && receipt.comparison) || {};
|
|
221
|
+
const delta = typeof cmp.delta === 'number' ? cmp.delta : 0;
|
|
222
|
+
const model = shortModel(receipt && receipt.run && receipt.run.model_id);
|
|
223
|
+
const v = receiptVerdict(receipt);
|
|
224
|
+
const meta = VERDICTS[v.verdict];
|
|
58
225
|
return {
|
|
59
|
-
verdict,
|
|
226
|
+
verdict: v.verdict,
|
|
60
227
|
delta,
|
|
61
228
|
floor: EFFECT_FLOOR,
|
|
62
229
|
model,
|
|
63
230
|
message: `${meta.word} on ${model}`,
|
|
64
231
|
color: meta.color,
|
|
232
|
+
drawsNeeded: v.drawsNeeded,
|
|
65
233
|
};
|
|
66
234
|
}
|
|
67
235
|
|
|
@@ -113,4 +281,7 @@ function githubOutputLines(receipt) {
|
|
|
113
281
|
].map(([k, val]) => githubOutputEntry(k, val)).join('\n');
|
|
114
282
|
}
|
|
115
283
|
|
|
116
|
-
module.exports = {
|
|
284
|
+
module.exports = {
|
|
285
|
+
verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS,
|
|
286
|
+
receiptVerdict, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
|
|
287
|
+
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.1",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
package/spec/RECEIPT.md
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
<!-- SPDX-License-Identifier: Apache-2.0 -->
|
|
2
|
-
# Driftproof receipt
|
|
2
|
+
# Driftproof receipt, spec v0.7
|
|
3
3
|
|
|
4
4
|
A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
|
|
5
5
|
**with** and **without** the skill on one model version, with the judge **sampled**
|
|
@@ -12,21 +12,60 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
|
|
|
12
12
|
(JSON Schema, draft 2020-12). This document is the human companion. Where they
|
|
13
13
|
disagree, the schema wins.
|
|
14
14
|
|
|
15
|
-
**Versioning.** The current schema is v0.
|
|
15
|
+
**Versioning.** The current schema is v0.7
|
|
16
16
|
([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept frozen as
|
|
17
|
+
[`receipt.v0.6.schema.json`](./receipt.v0.6.schema.json),
|
|
17
18
|
[`receipt.v0.5.schema.json`](./receipt.v0.5.schema.json),
|
|
18
19
|
[`receipt.v0.4.schema.json`](./receipt.v0.4.schema.json),
|
|
19
20
|
[`receipt.v0.3.1.schema.json`](./receipt.v0.3.1.schema.json),
|
|
20
21
|
[`receipt.v0.3.schema.json`](./receipt.v0.3.schema.json),
|
|
21
22
|
[`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
|
|
22
23
|
[`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
|
|
23
|
-
schema by the receipt's own `schema_version`, so
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
reader
|
|
27
|
-
v0.
|
|
28
|
-
|
|
29
|
-
|
|
24
|
+
schema by the receipt's own `schema_version`, so every earlier receipt still loads and
|
|
25
|
+
validates against its own frozen schema, including every receipt behind the published
|
|
26
|
+
reports, which the gate asserts on each run. v0.7 adds no field. It adds a verdict value
|
|
27
|
+
a reader derives from fields every generation-sampled receipt already carries, and a
|
|
28
|
+
v0.6 receipt restamped `"0.7"` is refused by the version constant, which is why v0.6 is
|
|
29
|
+
frozen rather than widened (see *What v0.7 adds*). v0.6 was not additive for the
|
|
30
|
+
validator either: a v0.5 receipt restamped `"0.6"` is refused, because v0.6 requires
|
|
31
|
+
the receipt to say what answered it (see *What v0.6 adds*).
|
|
32
|
+
|
|
33
|
+
## What v0.7 adds: a receipt that could not resolve the effect floor says so
|
|
34
|
+
|
|
35
|
+
Spec 035. Until v0.7 a receipt whose with_skill and baseline bands did not separate
|
|
36
|
+
was read as NO_EFFECT, and the single-receipt badge read only the aggregate delta
|
|
37
|
+
against the effect floor. The 0.10.1 reliability audit measured that badge labelling
|
|
38
|
+
34.56% of binary-null receipts passing. v0.7 reads every receipt per case, from its
|
|
39
|
+
own fields, by six rules. For one case, arm `w` is with_skill and arm `b` is baseline;
|
|
40
|
+
`m`, `s` and `n` are an arm's `generation.mean`, `generation.sd` and
|
|
41
|
+
`generation.n_measured` (a receipt without a `generation` block has one draw per arm,
|
|
42
|
+
`n = 1`, and its judge-sample band); `F` is `EFFECT_FLOOR` (0.05) and `Z` is `POWER_Z`
|
|
43
|
+
(2), both in `config.js`.
|
|
44
|
+
|
|
45
|
+
- **R-1, separation.** The bands do not overlap (`m_w - s_w > m_b + s_b` or the
|
|
46
|
+
reverse; touching is overlap) and `|m_w - m_b| >= F`.
|
|
47
|
+
- **R-2, resolution.** With both `n >= 2`: `r = s_w + s_b + Z * sqrt(s_w^2/n_w + s_b^2/n_b)`.
|
|
48
|
+
- **R-3, underpowered.** A case that does not separate is underpowered when an arm has
|
|
49
|
+
`n < 2`, or when `r >= F`.
|
|
50
|
+
- **R-4, draws needed.** If `s_w + s_b >= F`, no draw count reaches the floor at these
|
|
51
|
+
spreads, because more draws narrow the standard error and not the spread; otherwise
|
|
52
|
+
`floor(Z^2 * (s_w^2 + s_b^2) / (F - s_w - s_b)^2) + 1` per arm, at least 2, with the
|
|
53
|
+
spreads held; with an arm at `n < 2`, at least 2 per arm to measure a spread.
|
|
54
|
+
- **R-5, the receipt verdict.** A receipt below `TESTED`, incomplete, or with no numeric
|
|
55
|
+
delta is `NOT_MEASURED`, as before. Otherwise `REGRESSED` when any case separated
|
|
56
|
+
downward, else `PASSED` when any separated upward, else `UNDERPOWERED` when any case is
|
|
57
|
+
underpowered, else `NO_EFFECT`.
|
|
58
|
+
- **R-6, the differ.** Per case, over two receipts' with_skill arms, R-1 to R-4 apply;
|
|
59
|
+
the differ's per-case verdict value is unchanged, and an underpowered case carries
|
|
60
|
+
`power: { state: "underpowered", drawsNeeded, reason }` beside it.
|
|
61
|
+
|
|
62
|
+
**The token.** `UNDERPOWERED` is a machine-facing verdict value on `driftproof badge
|
|
63
|
+
--github-output`, the Action's `verdict` output, and `lib/verdict.js` `VERDICTS`; its
|
|
64
|
+
decision state is `underpowered`, ordered after `not measured` and before `no detected
|
|
65
|
+
effect`, and it never renders as success. Every surface that renders it in words
|
|
66
|
+
carries the plain line: **Not enough draws to conclude at this effect floor.** The
|
|
67
|
+
Action also publishes `draws_needed`. UNDERPOWERED is a statement about the instrument
|
|
68
|
+
at this receipt's spreads and draws; it is not evidence that the skill has an effect.
|
|
30
69
|
|
|
31
70
|
## What v0.6 adds: the receipt says what answered it
|
|
32
71
|
|
|
@@ -316,7 +355,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
|
|
|
316
355
|
|
|
317
356
|
## Fields
|
|
318
357
|
|
|
319
|
-
### `schema_version` (string, required)
|
|
358
|
+
### `schema_version` (string, required): `"0.7"`.
|
|
320
359
|
|
|
321
360
|
### `skill` (object, required)
|
|
322
361
|
| field | type | notes |
|
package/spec/receipt.schema.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
3
|
"$id": "https://driftproofhq.com/spec/receipt.schema.json",
|
|
4
4
|
"title": "driftproof receipt",
|
|
5
|
-
"description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.6
|
|
5
|
+
"description": "A hash-verified, dated record of running one agent skill eval suite with and without the skill on one model version, with the GENERATION sampled n times per arm and the judge sampled k times inside each draw. Receipt spec v0.7: over v0.6, the fields are unchanged and the verdict a reader derives gains UNDERPOWERED: a case the band rule does not separate whose spreads and measured draws cannot resolve a shift of the effect floor (spec 035, spec/RECEIPT.md § What v0.7 adds). Over v0.5, as v0.6 said, the receipt says what answered it: `run.answered_by` (kind model|stub|external, whether the surface echoed a model id and which, and which spawn path produced it); `run.surface` and `run.judge.surface` may read `stub` (nothing answered; the text was canned) and a TESTED receipt requires answered_by.kind model; `run.judge.model_id` and `run.judge.prompt_template_hash` say which judge ran; every draw records its `stop_reason` and whether it was `truncated` (a truncated draw is unmeasured, never judged) and the case counts `n_truncated`; `case_status` gains `failed_unmeasured` (every draw of the arm was unmeasured for a reason that is not a timeout), recorded without fabricated samples or hashes exactly as a timeout is; `results.aggregates.band_rule` states the formula the aggregate band is derived by; an aggregate `stddev` is null only when its arm has fewer than two cases and `comparison.delta_uncertainty` is null only beside `delta_uncertainty_unavailable`. Every prior receipt still validates against its own frozen schema (v0.1, v0.2, v0.3, v0.3.1, v0.4, v0.5, v0.6).",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"additionalProperties": false,
|
|
8
8
|
"required": [
|
|
@@ -274,7 +274,7 @@
|
|
|
274
274
|
"properties": {
|
|
275
275
|
"schema_version": {
|
|
276
276
|
"type": "string",
|
|
277
|
-
"const": "0.
|
|
277
|
+
"const": "0.7"
|
|
278
278
|
},
|
|
279
279
|
"skill": {
|
|
280
280
|
"type": "object",
|