driftproof 0.5.0 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -29
- package/bin/driftproof +46 -12
- package/config.js +49 -5
- package/lib/canary.js +27 -0
- package/lib/cost.js +20 -2
- package/lib/diff.js +172 -15
- package/lib/hygiene.js +113 -0
- package/lib/judge.js +6 -1
- package/lib/receipt.js +76 -4
- package/lib/reuse.js +134 -0
- package/lib/revision.js +169 -0
- package/lib/run.js +165 -45
- package/lib/sampling.js +110 -0
- package/lib/stats.js +16 -1
- package/lib/value.js +27 -2
- package/package.json +2 -2
- package/spec/RECEIPT.md +102 -3
- package/spec/receipt.schema.json +360 -3
- package/spec/receipt.v0.4.schema.json +961 -0
package/lib/sampling.js
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
// SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Generation sampling policy — receipt spec v0.5.
|
|
5
|
+
//
|
|
6
|
+
// WHY THIS EXISTS. Until v0.5 a verdict rested on ONE generation draw per arm,
|
|
7
|
+
// and the band a receipt carried was the judge re-scoring that single response.
|
|
8
|
+
// Report #006 measured the other axis directly and found it larger: across-draw
|
|
9
|
+
// sd 0.186 and 0.183, against judge-level noise several times smaller. Every
|
|
10
|
+
// figure this instrument published was error-barred on the smaller of the two
|
|
11
|
+
// noise sources. This module samples the larger one.
|
|
12
|
+
//
|
|
13
|
+
// PURE. No I/O, no provider, no clock. Two functions over a draw list, so the
|
|
14
|
+
// policy is testable from fixtures with no model call and the runner cannot
|
|
15
|
+
// smuggle a different rule past the gate.
|
|
16
|
+
|
|
17
|
+
const {
|
|
18
|
+
GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX,
|
|
19
|
+
GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
|
|
20
|
+
} = require('../config');
|
|
21
|
+
const { mean, stddev, round, varianceRatio } = require('./stats');
|
|
22
|
+
|
|
23
|
+
const SAMPLING = {
|
|
24
|
+
min: GENERATION_SAMPLES_MIN,
|
|
25
|
+
max: GENERATION_SAMPLES_MAX,
|
|
26
|
+
sdThreshold: GENERATION_SD_THRESHOLD,
|
|
27
|
+
stabilityEps: GENERATION_STABILITY_EPS,
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
// A draw is `measured` or `unmeasured`. An UNMEASURED draw carries no score and
|
|
31
|
+
// is excluded from every statistic — never scored zero (F-009-L). A zero is a
|
|
32
|
+
// measurement claim; a timeout is the absence of one, and filling it with zero
|
|
33
|
+
// invents a regression nobody observed.
|
|
34
|
+
const isMeasured = (d) => d && d.status === 'measured' && typeof d.mean === 'number';
|
|
35
|
+
|
|
36
|
+
// WHICH null the ratio is (F-014-C). The ratio is generation-sd over mean
|
|
37
|
+
// judge-sd, and at k=1 the per-draw judge spread is 0 by construction — so the
|
|
38
|
+
// number v0.5 exists to produce could not exist at the sampling every pre-v0.5
|
|
39
|
+
// example used, and the receipt recorded a null that read identically to "the
|
|
40
|
+
// judge agreed with itself perfectly". Those are different measurements.
|
|
41
|
+
//
|
|
42
|
+
// EVERY REASON NAMES WHAT WAS OBSERVED AND NONE NAMES A CAUSE (spec 014 AC-7).
|
|
43
|
+
// `judge_samples_unknown` exists for exactly that reason: with no sample list on
|
|
44
|
+
// the draws, the count cannot be established, and picking between the other two
|
|
45
|
+
// would be asserting something this function cannot see.
|
|
46
|
+
function unavailableReason(measured, ratio) {
|
|
47
|
+
if (ratio !== null) return null;
|
|
48
|
+
if (!measured.length) return 'no_measured_draws';
|
|
49
|
+
const counts = measured.map((d) => (Array.isArray(d.samples) ? d.samples.length : null));
|
|
50
|
+
if (counts.some((c) => c === null)) return 'judge_samples_unknown';
|
|
51
|
+
if (counts.every((c) => c <= 1)) return 'single_judge_sample';
|
|
52
|
+
return 'judge_sd_zero';
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function acrossDraws(draws) {
|
|
56
|
+
const all = Array.isArray(draws) ? draws : [];
|
|
57
|
+
const measured = all.filter(isMeasured);
|
|
58
|
+
const means = measured.map((d) => d.mean);
|
|
59
|
+
const judgeSds = measured.map((d) => (typeof d.stddev === 'number' ? d.stddev : 0));
|
|
60
|
+
const sd = means.length ? stddev(means) : 0;
|
|
61
|
+
const judgeSdMean = judgeSds.length ? mean(judgeSds) : 0;
|
|
62
|
+
const ratio = varianceRatio(sd, judgeSdMean);
|
|
63
|
+
return {
|
|
64
|
+
n_drawn: all.length,
|
|
65
|
+
n_measured: measured.length,
|
|
66
|
+
n_unmeasured: all.length - measured.length,
|
|
67
|
+
mean: means.length ? round(mean(means)) : null,
|
|
68
|
+
sd: round(sd),
|
|
69
|
+
judge_sd_mean: round(judgeSdMean),
|
|
70
|
+
variance_ratio: ratio,
|
|
71
|
+
variance_ratio_unavailable: unavailableReason(measured, ratio),
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Should the runner draw again? Returns the STOPPING REASON as well as the
|
|
76
|
+
// decision: an adaptive rule that does not record why it stopped is unauditable,
|
|
77
|
+
// and #006's own probe stopped at 20 draws because a person decided to.
|
|
78
|
+
function nextAction(draws) {
|
|
79
|
+
const a = acrossDraws(draws);
|
|
80
|
+
const drawn = a.n_drawn;
|
|
81
|
+
|
|
82
|
+
// Every draw failed. More draws are not obviously wrong, but continuing to
|
|
83
|
+
// burn calls on a surface that is not answering is, and the receipt should say
|
|
84
|
+
// that is what happened rather than report an empty mean.
|
|
85
|
+
if (drawn >= SAMPLING.min && a.n_measured === 0) return { stop: true, reason: 'unmeasured_exhausted' };
|
|
86
|
+
|
|
87
|
+
// THE CAP IS CHECKED BEFORE THE MINIMUM. Reversed, a run whose draws mostly
|
|
88
|
+
// failed reported `below_min` while what actually stopped it was the ceiling —
|
|
89
|
+
// the receipt named a reason that was not the reason. Approval finding, AC-4.
|
|
90
|
+
if (drawn >= SAMPLING.max) return { stop: true, reason: 'max_reached' };
|
|
91
|
+
|
|
92
|
+
if (a.n_measured < SAMPLING.min) return { stop: false, reason: 'below_min' };
|
|
93
|
+
|
|
94
|
+
// Tight enough at the minimum: the extra draws would buy nothing.
|
|
95
|
+
if (a.sd <= SAMPLING.sdThreshold) return { stop: true, reason: 'min_reached' };
|
|
96
|
+
|
|
97
|
+
// (the bounded-escalation ceiling is enforced above, before the minimum)
|
|
98
|
+
|
|
99
|
+
// …unless the estimate has stopped moving: two successive prefixes agreeing to
|
|
100
|
+
// within eps means further draws are not changing what we would report.
|
|
101
|
+
if (a.n_measured > SAMPLING.min) {
|
|
102
|
+
const prev = acrossDraws(draws.slice(0, -1));
|
|
103
|
+
if (prev.n_measured >= SAMPLING.min && Math.abs(a.sd - prev.sd) < SAMPLING.stabilityEps) {
|
|
104
|
+
return { stop: true, reason: 'stabilised' };
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
return { stop: false, reason: 'escalating' };
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
module.exports = { SAMPLING, acrossDraws, nextAction, isMeasured, unavailableReason };
|
package/lib/stats.js
CHANGED
|
@@ -63,4 +63,19 @@ function bandVerdict(meanA, hwA, meanB, hwB) {
|
|
|
63
63
|
return 'within noise';
|
|
64
64
|
}
|
|
65
65
|
|
|
66
|
-
|
|
66
|
+
// The variance ratio: how much larger the GENERATION-level spread is than the
|
|
67
|
+
// JUDGE-level spread this instrument has always sampled. The number Report #006
|
|
68
|
+
// wanted and could not produce.
|
|
69
|
+
//
|
|
70
|
+
// A ZERO DENOMINATOR YIELDS null, NEVER A DIVISION RESULT. Infinity (or NaN) is
|
|
71
|
+
// a fabricated finding: it reads as "infinitely noisier" when what actually
|
|
72
|
+
// happened is that the judge agreed with itself perfectly and the ratio is
|
|
73
|
+
// undefined. A null says the ratio could not be formed, which is true.
|
|
74
|
+
function varianceRatio(generationSd, judgeSdMean) {
|
|
75
|
+
const g = Number(generationSd), j = Number(judgeSdMean);
|
|
76
|
+
if (!Number.isFinite(g) || !Number.isFinite(j)) return null;
|
|
77
|
+
if (j === 0) return null;
|
|
78
|
+
return round(g / j);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio };
|
package/lib/value.js
CHANGED
|
@@ -151,8 +151,33 @@ function meanOf(nums) {
|
|
|
151
151
|
// One arm = all completed case rows of one mode (with_skill | baseline). Only
|
|
152
152
|
// the GENERATION usage of each row feeds this; judge usage is measurement
|
|
153
153
|
// overhead and is never passed in (see computeEconomics).
|
|
154
|
+
// THE GENERATION USAGE OF A CASE ROW, wherever that version put it (spec 017
|
|
155
|
+
// AC-3, AC-4).
|
|
156
|
+
//
|
|
157
|
+
// v0.4 and earlier record one generation per case and put its usage at case
|
|
158
|
+
// level. v0.5 draws the generation n times and records each draw's usage under
|
|
159
|
+
// `generation.draws[].usage`; the case-level field is then absent. This function
|
|
160
|
+
// read only the case-level one, so `call_count` was 0 and every arm figure null
|
|
161
|
+
// on all three #007 receipts — while the draws carried complete usage. Judge
|
|
162
|
+
// economics never broke, because `judge_usage` stayed at case level, and that
|
|
163
|
+
// asymmetry is what located the defect.
|
|
164
|
+
//
|
|
165
|
+
// A DRAW IS A CALL. An arm that drew three generations made three generation
|
|
166
|
+
// calls, so each measured draw contributes one usage row; the v0.4 path yields
|
|
167
|
+
// exactly one row per case, which is what it always yielded. Unmeasured draws
|
|
168
|
+
// carry no usage and are excluded — a timed-out call has no tokens to count, and
|
|
169
|
+
// counting it as zero would deflate the mean the way scoring it zero would
|
|
170
|
+
// deflate the band (F-009-L, applied to cost).
|
|
171
|
+
function generationUsages(row) {
|
|
172
|
+
const g = row && row.generation;
|
|
173
|
+
if (g && Array.isArray(g.draws)) {
|
|
174
|
+
return g.draws.filter((d) => d && d.status === 'measured' && d.usage).map((d) => d.usage);
|
|
175
|
+
}
|
|
176
|
+
return row && row.usage ? [row.usage] : [];
|
|
177
|
+
}
|
|
178
|
+
|
|
154
179
|
function armEconomics(rows, price) {
|
|
155
|
-
const usages = rows.
|
|
180
|
+
const usages = rows.flatMap(generationUsages).filter(Boolean);
|
|
156
181
|
const costs = usages.map((u) => costForUsage(u, price)).filter((c) => c != null);
|
|
157
182
|
const wall = usages.map((u) => (u.wall_ms == null ? null : u.wall_ms)).filter((x) => x != null);
|
|
158
183
|
const q = quartiles(wall);
|
|
@@ -491,7 +516,7 @@ function round4(n) { return Math.round(Number(n) * 1e4) / 1e4; }
|
|
|
491
516
|
function round6(n) { return Math.round(Number(n) * 1e6) / 1e6; }
|
|
492
517
|
|
|
493
518
|
module.exports = {
|
|
494
|
-
buildPricingSnapshot, costForUsage, computeEconomics, armEconomics,
|
|
519
|
+
buildPricingSnapshot, costForUsage, computeEconomics, armEconomics, generationUsages,
|
|
495
520
|
liftIsReportable, costPerLiftPoint, isZeroLooking, dollarsTraceable, runTotalFromReceipts,
|
|
496
521
|
runWallClockFromReceipts,
|
|
497
522
|
receiptCostBreakdown,
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "0.7.1",
|
|
4
|
+
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"agent-skills",
|
package/spec/RECEIPT.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
<!-- SPDX-License-Identifier: Apache-2.0 -->
|
|
2
|
-
# Driftproof receipt — spec v0.
|
|
2
|
+
# Driftproof receipt — spec v0.5
|
|
3
3
|
|
|
4
|
-
A **receipt** is a
|
|
4
|
+
A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
|
|
5
5
|
**with** and **without** the skill on one model version, with the judge **sampled**
|
|
6
6
|
so every score carries a confidence band. Receipts are the unit of evidence
|
|
7
7
|
Driftproof produces; diffing two receipts across model releases yields a **drift
|
|
@@ -22,6 +22,105 @@ receipts still load and validate — including every receipt behind the four
|
|
|
22
22
|
published reports, which the gate asserts on each run. v0.4 is an **additive**
|
|
23
23
|
bump: everything it adds is optional, and no earlier receipt is invalidated.
|
|
24
24
|
|
|
25
|
+
## What v0.5 adds: the generation is sampled too
|
|
26
|
+
|
|
27
|
+
Until v0.5 a verdict rested on **one generation draw per arm**. The band a
|
|
28
|
+
receipt carried was the *judge* re-scoring that single response — not the spread
|
|
29
|
+
of the model writing a different one. Report #006 measured the second directly
|
|
30
|
+
and found it larger: across-draw **sd 0.186** and **sd 0.183**, against
|
|
31
|
+
judge-level noise several times smaller. Every figure this instrument published
|
|
32
|
+
was error-barred on the smaller of the two noise sources.
|
|
33
|
+
|
|
34
|
+
v0.5 samples the other axis. Each case and arm carries a `generation` block:
|
|
35
|
+
|
|
36
|
+
- `draws[]` — **every draw, listed**, each with its own `generation_hash`, its
|
|
37
|
+
own nested judge `samples`, and its own `mean` and `stddev`. An aggregate a
|
|
38
|
+
reader cannot re-derive is not evidence; the list is what makes it evidence.
|
|
39
|
+
- `mean`, `sd` — across draws.
|
|
40
|
+
- `judge_sd_mean` — the mean of the per-draw judge stddevs.
|
|
41
|
+
- `variance_ratio` — `sd / judge_sd_mean`: how much larger the generation-level
|
|
42
|
+
spread is than the judge-level spread. **`null` when the judge sd is zero**,
|
|
43
|
+
never a division result: an infinity reads as "infinitely noisier" when what
|
|
44
|
+
happened is that the ratio is undefined.
|
|
45
|
+
- `variance_ratio_unavailable` — **which null it is**, and it is present whenever
|
|
46
|
+
the ratio is null. One of `single_judge_sample`, `judge_sd_zero`,
|
|
47
|
+
`no_measured_draws`, `judge_samples_unknown`. Each names what was *observed*
|
|
48
|
+
and none names a cause.
|
|
49
|
+
- `n_planned`, `n_drawn`, `n_measured`, `n_unmeasured`, `stopping_reason` — the
|
|
50
|
+
sampling policy **as applied**, so an adaptive run says why it stopped.
|
|
51
|
+
|
|
52
|
+
**What `--samples 1` costs.** The ratio is generation-sd over mean judge-sd, and
|
|
53
|
+
with **one judge sample per draw the per-draw judge spread is zero by
|
|
54
|
+
construction** — so the number v0.5 exists to produce cannot be formed at `k=1`.
|
|
55
|
+
`k=1` remains legal and the run still emits a receipt; the ratio is `null` and
|
|
56
|
+
`variance_ratio_unavailable` reads **`single_judge_sample`**, which is a
|
|
57
|
+
different statement from `judge_sd_zero` ("the judge agreed with itself perfectly
|
|
58
|
+
across two or more samples"). Before this field the two nulls were the same null.
|
|
59
|
+
Documented sampling is `--samples 5`, the default, and a run whose ratio you
|
|
60
|
+
intend to publish must use at least 2.
|
|
61
|
+
|
|
62
|
+
## The `generation_sampled` capability flag
|
|
63
|
+
|
|
64
|
+
A receipt that carries across-draw statistics **must declare
|
|
65
|
+
`generation_sampled: true` at the top level**, and a receipt that declares it
|
|
66
|
+
**must carry at least one case with a `generation` block AND must carry
|
|
67
|
+
`suite.canary`**. All three directions are enforced by the schema, so neither a
|
|
68
|
+
producer nor a third-party emitter can claim the capability without the evidence,
|
|
69
|
+
or carry the evidence without saying so.
|
|
70
|
+
|
|
71
|
+
**Both blocks, not one.** The exposure this closes named two things v0.5 adds —
|
|
72
|
+
the draw set and the per-suite canary — and binding only the first left a receipt
|
|
73
|
+
able to claim conformance while carrying half of it. A receipt that declares the
|
|
74
|
+
capability has, by construction, run a suite of ours, so it has a canary to
|
|
75
|
+
record.
|
|
76
|
+
|
|
77
|
+
**Absent means legacy.** A v0.5 receipt that ran no generation sampling — an
|
|
78
|
+
imported `DECLARED` receipt, for instance — omits the field and stays valid, and
|
|
79
|
+
every v0.4-and-earlier receipt is unaffected and unmodified. That is the whole
|
|
80
|
+
reason the requirement is bound to the DECLARATION rather than to
|
|
81
|
+
`verification_level`: binding it to the level would have invalidated hand-built
|
|
82
|
+
fixtures across the archive that never ran a generation, which is a different
|
|
83
|
+
change from the one this flag makes.
|
|
84
|
+
|
|
85
|
+
**The judge nests inside the draw.** Pooling k×n scores into one list per case is
|
|
86
|
+
exactly what makes generation noise read as judge noise, and it is the defect
|
|
87
|
+
Report #006 exists to name.
|
|
88
|
+
|
|
89
|
+
**A timed-out draw is `unmeasured`, never zero.** It carries no score and is
|
|
90
|
+
excluded from every mean, sd and ratio. A zero asserts a measurement; a timeout
|
|
91
|
+
is the absence of one, and filling it with zero invents a regression nobody
|
|
92
|
+
observed. The schema enforces this: a draw with `status: "unmeasured"` may not
|
|
93
|
+
carry a numeric `mean`.
|
|
94
|
+
|
|
95
|
+
## Which schema validates which receipt (the coexistence rule)
|
|
96
|
+
|
|
97
|
+
- **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
|
|
98
|
+
decides it, and at this revision that is **v0.5**.
|
|
99
|
+
- **A reader validates against the receipt's own `schema_version`**, never
|
|
100
|
+
against the newest schema it happens to have. `validateReceipt()` selects the
|
|
101
|
+
schema by that field, which is why a v0.1 receipt from the first report still
|
|
102
|
+
validates today.
|
|
103
|
+
- **v0.4 receipts remain producible and readable.** v0.4 moved from the
|
|
104
|
+
unversioned filename to `spec/receipt.v0.4.schema.json` when v0.5 took the
|
|
105
|
+
current pointer; the number still resolves to the schema it meant. Nothing in
|
|
106
|
+
the archive is retroactively invalidated, and the frozen v0.1, v0.2, v0.3 and
|
|
107
|
+
v0.3.1 schemas are byte-for-byte unchanged.
|
|
108
|
+
- **A v0.4 reader meeting a v0.5 receipt** finds every field it knows, and the
|
|
109
|
+
ones it does not know are additive. `mean`, `score` and `stddev` on a case now
|
|
110
|
+
describe the **draw set** rather than one arbitrary draw, so an older reader
|
|
111
|
+
reads the aggregate rather than whichever draw happened to be last;
|
|
112
|
+
`generation_hash`, `samples` and `judge_sample_hashes` continue to describe the
|
|
113
|
+
last measured draw, which is what they have always described. Older readers
|
|
114
|
+
that ignore unknown fields therefore stay correct rather than silently wrong.
|
|
115
|
+
|
|
116
|
+
## Suite canary
|
|
117
|
+
|
|
118
|
+
`suite.canary` is a GUID derived from the suite's identity and its case ids —
|
|
119
|
+
stable for a suite, distinct across suites, and requiring no registry. It exists
|
|
120
|
+
so a suite that has leaked into training data is **detectable** in a corpus. It
|
|
121
|
+
is a detection aid and not a control: it cannot prevent a leak, and its absence
|
|
122
|
+
proves nothing.
|
|
123
|
+
|
|
25
124
|
## What changed in v0.4 (economics — what a skill costs to run)
|
|
26
125
|
|
|
27
126
|
Reports #001–#004 could say whether a skill still helps. They could not say what
|
|
@@ -164,7 +263,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
|
|
|
164
263
|
|
|
165
264
|
## Fields
|
|
166
265
|
|
|
167
|
-
### `schema_version` (string, required) — `"0.
|
|
266
|
+
### `schema_version` (string, required) — `"0.4"`.
|
|
168
267
|
|
|
169
268
|
### `skill` (object, required)
|
|
170
269
|
| field | type | notes |
|