driftproof 0.5.0 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,110 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ 'use strict';
3
+
4
+ // Generation sampling policy — receipt spec v0.5.
5
+ //
6
+ // WHY THIS EXISTS. Until v0.5 a verdict rested on ONE generation draw per arm,
7
+ // and the band a receipt carried was the judge re-scoring that single response.
8
+ // Report #006 measured the other axis directly and found it larger: across-draw
9
+ // sd 0.186 and 0.183, against judge-level noise several times smaller. Every
10
+ // figure this instrument published was error-barred on the smaller of the two
11
+ // noise sources. This module samples the larger one.
12
+ //
13
+ // PURE. No I/O, no provider, no clock. Two functions over a draw list, so the
14
+ // policy is testable from fixtures with no model call and the runner cannot
15
+ // smuggle a different rule past the gate.
16
+
17
+ const {
18
+ GENERATION_SAMPLES_MIN, GENERATION_SAMPLES_MAX,
19
+ GENERATION_SD_THRESHOLD, GENERATION_STABILITY_EPS,
20
+ } = require('../config');
21
+ const { mean, stddev, round, varianceRatio } = require('./stats');
22
+
23
+ const SAMPLING = {
24
+ min: GENERATION_SAMPLES_MIN,
25
+ max: GENERATION_SAMPLES_MAX,
26
+ sdThreshold: GENERATION_SD_THRESHOLD,
27
+ stabilityEps: GENERATION_STABILITY_EPS,
28
+ };
29
+
30
+ // A draw is `measured` or `unmeasured`. An UNMEASURED draw carries no score and
31
+ // is excluded from every statistic — never scored zero (F-009-L). A zero is a
32
+ // measurement claim; a timeout is the absence of one, and filling it with zero
33
+ // invents a regression nobody observed.
34
+ const isMeasured = (d) => d && d.status === 'measured' && typeof d.mean === 'number';
35
+
36
+ // WHICH null the ratio is (F-014-C). The ratio is generation-sd over mean
37
+ // judge-sd, and at k=1 the per-draw judge spread is 0 by construction — so the
38
+ // number v0.5 exists to produce could not exist at the sampling every pre-v0.5
39
+ // example used, and the receipt recorded a null that read identically to "the
40
+ // judge agreed with itself perfectly". Those are different measurements.
41
+ //
42
+ // EVERY REASON NAMES WHAT WAS OBSERVED AND NONE NAMES A CAUSE (spec 014 AC-7).
43
+ // `judge_samples_unknown` exists for exactly that reason: with no sample list on
44
+ // the draws, the count cannot be established, and picking between the other two
45
+ // would be asserting something this function cannot see.
46
+ function unavailableReason(measured, ratio) {
47
+ if (ratio !== null) return null;
48
+ if (!measured.length) return 'no_measured_draws';
49
+ const counts = measured.map((d) => (Array.isArray(d.samples) ? d.samples.length : null));
50
+ if (counts.some((c) => c === null)) return 'judge_samples_unknown';
51
+ if (counts.every((c) => c <= 1)) return 'single_judge_sample';
52
+ return 'judge_sd_zero';
53
+ }
54
+
55
+ function acrossDraws(draws) {
56
+ const all = Array.isArray(draws) ? draws : [];
57
+ const measured = all.filter(isMeasured);
58
+ const means = measured.map((d) => d.mean);
59
+ const judgeSds = measured.map((d) => (typeof d.stddev === 'number' ? d.stddev : 0));
60
+ const sd = means.length ? stddev(means) : 0;
61
+ const judgeSdMean = judgeSds.length ? mean(judgeSds) : 0;
62
+ const ratio = varianceRatio(sd, judgeSdMean);
63
+ return {
64
+ n_drawn: all.length,
65
+ n_measured: measured.length,
66
+ n_unmeasured: all.length - measured.length,
67
+ mean: means.length ? round(mean(means)) : null,
68
+ sd: round(sd),
69
+ judge_sd_mean: round(judgeSdMean),
70
+ variance_ratio: ratio,
71
+ variance_ratio_unavailable: unavailableReason(measured, ratio),
72
+ };
73
+ }
74
+
75
+ // Should the runner draw again? Returns the STOPPING REASON as well as the
76
+ // decision: an adaptive rule that does not record why it stopped is unauditable,
77
+ // and #006's own probe stopped at 20 draws because a person decided to.
78
+ function nextAction(draws) {
79
+ const a = acrossDraws(draws);
80
+ const drawn = a.n_drawn;
81
+
82
+ // Every draw failed. More draws are not obviously wrong, but continuing to
83
+ // burn calls on a surface that is not answering is, and the receipt should say
84
+ // that is what happened rather than report an empty mean.
85
+ if (drawn >= SAMPLING.min && a.n_measured === 0) return { stop: true, reason: 'unmeasured_exhausted' };
86
+
87
+ // THE CAP IS CHECKED BEFORE THE MINIMUM. Reversed, a run whose draws mostly
88
+ // failed reported `below_min` while what actually stopped it was the ceiling —
89
+ // the receipt named a reason that was not the reason. Approval finding, AC-4.
90
+ if (drawn >= SAMPLING.max) return { stop: true, reason: 'max_reached' };
91
+
92
+ if (a.n_measured < SAMPLING.min) return { stop: false, reason: 'below_min' };
93
+
94
+ // Tight enough at the minimum: the extra draws would buy nothing.
95
+ if (a.sd <= SAMPLING.sdThreshold) return { stop: true, reason: 'min_reached' };
96
+
97
+ // (the bounded-escalation ceiling is enforced above, before the minimum)
98
+
99
+ // …unless the estimate has stopped moving: two successive prefixes agreeing to
100
+ // within eps means further draws are not changing what we would report.
101
+ if (a.n_measured > SAMPLING.min) {
102
+ const prev = acrossDraws(draws.slice(0, -1));
103
+ if (prev.n_measured >= SAMPLING.min && Math.abs(a.sd - prev.sd) < SAMPLING.stabilityEps) {
104
+ return { stop: true, reason: 'stabilised' };
105
+ }
106
+ }
107
+ return { stop: false, reason: 'escalating' };
108
+ }
109
+
110
+ module.exports = { SAMPLING, acrossDraws, nextAction, isMeasured, unavailableReason };
package/lib/stats.js CHANGED
@@ -63,4 +63,19 @@ function bandVerdict(meanA, hwA, meanB, hwB) {
63
63
  return 'within noise';
64
64
  }
65
65
 
66
- module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round };
66
+ // The variance ratio: how much larger the GENERATION-level spread is than the
67
+ // JUDGE-level spread this instrument has always sampled. The number Report #006
68
+ // wanted and could not produce.
69
+ //
70
+ // A ZERO DENOMINATOR YIELDS null, NEVER A DIVISION RESULT. Infinity (or NaN) is
71
+ // a fabricated finding: it reads as "infinitely noisier" when what actually
72
+ // happened is that the judge agreed with itself perfectly and the ratio is
73
+ // undefined. A null says the ratio could not be formed, which is true.
74
+ function varianceRatio(generationSd, judgeSdMean) {
75
+ const g = Number(generationSd), j = Number(judgeSdMean);
76
+ if (!Number.isFinite(g) || !Number.isFinite(j)) return null;
77
+ if (j === 0) return null;
78
+ return round(g / j);
79
+ }
80
+
81
+ module.exports = { mean, stddev, stderr, combineUncertainty, aggregateBands, bandVerdict, round, varianceRatio };
package/lib/value.js CHANGED
@@ -151,8 +151,33 @@ function meanOf(nums) {
151
151
  // One arm = all completed case rows of one mode (with_skill | baseline). Only
152
152
  // the GENERATION usage of each row feeds this; judge usage is measurement
153
153
  // overhead and is never passed in (see computeEconomics).
154
+ // THE GENERATION USAGE OF A CASE ROW, wherever that version put it (spec 017
155
+ // AC-3, AC-4).
156
+ //
157
+ // v0.4 and earlier record one generation per case and put its usage at case
158
+ // level. v0.5 draws the generation n times and records each draw's usage under
159
+ // `generation.draws[].usage`; the case-level field is then absent. This function
160
+ // read only the case-level one, so `call_count` was 0 and every arm figure null
161
+ // on all three #007 receipts — while the draws carried complete usage. Judge
162
+ // economics never broke, because `judge_usage` stayed at case level, and that
163
+ // asymmetry is what located the defect.
164
+ //
165
+ // A DRAW IS A CALL. An arm that drew three generations made three generation
166
+ // calls, so each measured draw contributes one usage row; the v0.4 path yields
167
+ // exactly one row per case, which is what it always yielded. Unmeasured draws
168
+ // carry no usage and are excluded — a timed-out call has no tokens to count, and
169
+ // counting it as zero would deflate the mean the way scoring it zero would
170
+ // deflate the band (F-009-L, applied to cost).
171
+ function generationUsages(row) {
172
+ const g = row && row.generation;
173
+ if (g && Array.isArray(g.draws)) {
174
+ return g.draws.filter((d) => d && d.status === 'measured' && d.usage).map((d) => d.usage);
175
+ }
176
+ return row && row.usage ? [row.usage] : [];
177
+ }
178
+
154
179
  function armEconomics(rows, price) {
155
- const usages = rows.map((r) => r.usage).filter(Boolean);
180
+ const usages = rows.flatMap(generationUsages).filter(Boolean);
156
181
  const costs = usages.map((u) => costForUsage(u, price)).filter((c) => c != null);
157
182
  const wall = usages.map((u) => (u.wall_ms == null ? null : u.wall_ms)).filter((x) => x != null);
158
183
  const q = quartiles(wall);
@@ -491,7 +516,7 @@ function round4(n) { return Math.round(Number(n) * 1e4) / 1e4; }
491
516
  function round6(n) { return Math.round(Number(n) * 1e6) / 1e6; }
492
517
 
493
518
  module.exports = {
494
- buildPricingSnapshot, costForUsage, computeEconomics, armEconomics,
519
+ buildPricingSnapshot, costForUsage, computeEconomics, armEconomics, generationUsages,
495
520
  liftIsReportable, costPerLiftPoint, isZeroLooking, dollarsTraceable, runTotalFromReceipts,
496
521
  runWallClockFromReceipts,
497
522
  receiptCostBreakdown,
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "driftproof",
3
- "version": "0.5.0",
4
- "description": "Continuous verification of agent skills: run a skill's eval suite with and without the skill across model versions, emit signed dated receipts, and diff receipts into drift reports.",
3
+ "version": "0.7.1",
4
+ "description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
7
7
  "agent-skills",
package/spec/RECEIPT.md CHANGED
@@ -1,7 +1,7 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
- # Driftproof receipt — spec v0.4
2
+ # Driftproof receipt — spec v0.5
3
3
 
4
- A **receipt** is a signed, dated record of running one agent skill's eval suite
4
+ A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
5
5
  **with** and **without** the skill on one model version, with the judge **sampled**
6
6
  so every score carries a confidence band. Receipts are the unit of evidence
7
7
  Driftproof produces; diffing two receipts across model releases yields a **drift
@@ -22,6 +22,105 @@ receipts still load and validate — including every receipt behind the four
22
22
  published reports, which the gate asserts on each run. v0.4 is an **additive**
23
23
  bump: everything it adds is optional, and no earlier receipt is invalidated.
24
24
 
25
+ ## What v0.5 adds: the generation is sampled too
26
+
27
+ Until v0.5 a verdict rested on **one generation draw per arm**. The band a
28
+ receipt carried was the *judge* re-scoring that single response — not the spread
29
+ of the model writing a different one. Report #006 measured the second directly
30
+ and found it larger: across-draw **sd 0.186** and **sd 0.183**, against
31
+ judge-level noise several times smaller. Every figure this instrument published
32
+ was error-barred on the smaller of the two noise sources.
33
+
34
+ v0.5 samples the other axis. Each case and arm carries a `generation` block:
35
+
36
+ - `draws[]` — **every draw, listed**, each with its own `generation_hash`, its
37
+ own nested judge `samples`, and its own `mean` and `stddev`. An aggregate a
38
+ reader cannot re-derive is not evidence; the list is what makes it evidence.
39
+ - `mean`, `sd` — across draws.
40
+ - `judge_sd_mean` — the mean of the per-draw judge stddevs.
41
+ - `variance_ratio` — `sd / judge_sd_mean`: how much larger the generation-level
42
+ spread is than the judge-level spread. **`null` when the judge sd is zero**,
43
+ never a division result: an infinity reads as "infinitely noisier" when what
44
+ happened is that the ratio is undefined.
45
+ - `variance_ratio_unavailable` — **which null it is**, and it is present whenever
46
+ the ratio is null. One of `single_judge_sample`, `judge_sd_zero`,
47
+ `no_measured_draws`, `judge_samples_unknown`. Each names what was *observed*
48
+ and none names a cause.
49
+ - `n_planned`, `n_drawn`, `n_measured`, `n_unmeasured`, `stopping_reason` — the
50
+ sampling policy **as applied**, so an adaptive run says why it stopped.
51
+
52
+ **What `--samples 1` costs.** The ratio is generation-sd over mean judge-sd, and
53
+ with **one judge sample per draw the per-draw judge spread is zero by
54
+ construction** — so the number v0.5 exists to produce cannot be formed at `k=1`.
55
+ `k=1` remains legal and the run still emits a receipt; the ratio is `null` and
56
+ `variance_ratio_unavailable` reads **`single_judge_sample`**, which is a
57
+ different statement from `judge_sd_zero` ("the judge agreed with itself perfectly
58
+ across two or more samples"). Before this field the two nulls were the same null.
59
+ Documented sampling is `--samples 5`, the default, and a run whose ratio you
60
+ intend to publish must use at least 2.
61
+
62
+ ## The `generation_sampled` capability flag
63
+
64
+ A receipt that carries across-draw statistics **must declare
65
+ `generation_sampled: true` at the top level**, and a receipt that declares it
66
+ **must carry at least one case with a `generation` block AND must carry
67
+ `suite.canary`**. All three directions are enforced by the schema, so neither a
68
+ producer nor a third-party emitter can claim the capability without the evidence,
69
+ or carry the evidence without saying so.
70
+
71
+ **Both blocks, not one.** The exposure this closes named two things v0.5 adds —
72
+ the draw set and the per-suite canary — and binding only the first left a receipt
73
+ able to claim conformance while carrying half of it. A receipt that declares the
74
+ capability has, by construction, run a suite of ours, so it has a canary to
75
+ record.
76
+
77
+ **Absent means legacy.** A v0.5 receipt that ran no generation sampling — an
78
+ imported `DECLARED` receipt, for instance — omits the field and stays valid, and
79
+ every v0.4-and-earlier receipt is unaffected and unmodified. That is the whole
80
+ reason the requirement is bound to the DECLARATION rather than to
81
+ `verification_level`: binding it to the level would have invalidated hand-built
82
+ fixtures across the archive that never ran a generation, which is a different
83
+ change from the one this flag makes.
84
+
85
+ **The judge nests inside the draw.** Pooling k×n scores into one list per case is
86
+ exactly what makes generation noise read as judge noise, and it is the defect
87
+ Report #006 exists to name.
88
+
89
+ **A timed-out draw is `unmeasured`, never zero.** It carries no score and is
90
+ excluded from every mean, sd and ratio. A zero asserts a measurement; a timeout
91
+ is the absence of one, and filling it with zero invents a regression nobody
92
+ observed. The schema enforces this: a draw with `status: "unmeasured"` may not
93
+ carry a numeric `mean`.
94
+
95
+ ## Which schema validates which receipt (the coexistence rule)
96
+
97
+ - **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
98
+ decides it, and at this revision that is **v0.5**.
99
+ - **A reader validates against the receipt's own `schema_version`**, never
100
+ against the newest schema it happens to have. `validateReceipt()` selects the
101
+ schema by that field, which is why a v0.1 receipt from the first report still
102
+ validates today.
103
+ - **v0.4 receipts remain producible and readable.** v0.4 moved from the
104
+ unversioned filename to `spec/receipt.v0.4.schema.json` when v0.5 took the
105
+ current pointer; the number still resolves to the schema it meant. Nothing in
106
+ the archive is retroactively invalidated, and the frozen v0.1, v0.2, v0.3 and
107
+ v0.3.1 schemas are byte-for-byte unchanged.
108
+ - **A v0.4 reader meeting a v0.5 receipt** finds every field it knows, and the
109
+ ones it does not know are additive. `mean`, `score` and `stddev` on a case now
110
+ describe the **draw set** rather than one arbitrary draw, so an older reader
111
+ reads the aggregate rather than whichever draw happened to be last;
112
+ `generation_hash`, `samples` and `judge_sample_hashes` continue to describe the
113
+ last measured draw, which is what they have always described. Older readers
114
+ that ignore unknown fields therefore stay correct rather than silently wrong.
115
+
116
+ ## Suite canary
117
+
118
+ `suite.canary` is a GUID derived from the suite's identity and its case ids —
119
+ stable for a suite, distinct across suites, and requiring no registry. It exists
120
+ so a suite that has leaked into training data is **detectable** in a corpus. It
121
+ is a detection aid and not a control: it cannot prevent a leak, and its absence
122
+ proves nothing.
123
+
25
124
  ## What changed in v0.4 (economics — what a skill costs to run)
26
125
 
27
126
  Reports #001–#004 could say whether a skill still helps. They could not say what
@@ -164,7 +263,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
164
263
 
165
264
  ## Fields
166
265
 
167
- ### `schema_version` (string, required) — `"0.3.1"`.
266
+ ### `schema_version` (string, required) — `"0.4"`.
168
267
 
169
268
  ### `skill` (object, required)
170
269
  | field | type | notes |