driftproof 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/verdict.js CHANGED
@@ -4,6 +4,18 @@
4
4
  const crypto = require('crypto');
5
5
  const { EFFECT_FLOOR, POWER_Z } = require('../config');
6
6
  const { bandOf } = require('./reuse');
7
+ const { duplicateCaseRows, ambiguityLine, suiteNarrowed } = require('./receipt');
8
+
9
+ // Spec 050 AC-3. A receipt with two rows for one (id, mode) has no verdict: which row
10
+ // the byId map below kept would decide it. The error carries a code so a caller that
11
+ // renders refusals can tell this one from a defect.
12
+ class AmbiguousReceiptError extends Error {
13
+ constructor(dups) {
14
+ super(`ambiguous receipt: ${ambiguityLine(dups)}; a verdict over it would depend on which row was read`);
15
+ this.code = 'AMBIGUOUS_RECEIPT';
16
+ this.duplicates = dups;
17
+ }
18
+ }
7
19
 
8
20
  // Single-receipt verdict + shields.io badge.
9
21
  //
@@ -156,6 +168,11 @@ function notMeasuredRoutes(level, kind, delta, incomplete) {
156
168
  // lib/decision.js states. An absent field satisfies "not TESTED" and "not model":
157
169
  // a receipt that does not say is not taken to have said the reassuring thing.
158
170
  function receiptVerdict(receipt) {
171
+ // Spec 050 AC-3: refused before anything is read, whatever the level. An existing
172
+ // receipt written before the loader refused duplicate ids reaches here unchanged,
173
+ // and a surface that caught nothing must not render a verdict for it.
174
+ const dups = duplicateCaseRows(receipt);
175
+ if (dups.length) throw new AmbiguousReceiptError(dups);
159
176
  const cmp = (receipt && receipt.comparison) || {};
160
177
  // Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
161
178
  // verdicted — we did not run the suite, so we do not certify the outcome.
@@ -169,7 +186,11 @@ function receiptVerdict(receipt) {
169
186
  // incomplete: a case had an arm that could not be measured) is not verdicted
170
187
  // either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
171
188
  // compute a verdict from one; the badge is the same reader with a shorter path.
172
- const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
189
+ //
190
+ // Spec 062 (register row 4): a receipt that ran fewer cases than its suite holds is incomplete
191
+ // too, by the same route. Its suite_hash names the whole suite, so a verdict read off it would
192
+ // be a verdict on cases that were never run.
193
+ const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete') || suiteNarrowed(receipt);
173
194
  // Spec 036 A-036-4: WHICH REFUSAL FIRED. The guard below is one line and says only that
174
195
  // some refusal did; a surface that renders NOT_MEASURED in words owes its reader the one
175
196
  // that actually fired, and cannot get it from a boolean. `notMeasuredRoutes` names the same
@@ -184,20 +205,7 @@ function receiptVerdict(receipt) {
184
205
  // the rung edges. A guard and a route list that disagree read RED there.
185
206
  const routes = notMeasuredRoutes(level, kind, cmp.delta, incomplete);
186
207
  if (level !== 'TESTED' || kind !== 'model' || typeof cmp.delta !== 'number' || incomplete) return { verdict: 'NOT_MEASURED', cases: [], drawsNeeded: null, notMeasured: routes };
187
- const byId = new Map();
188
- for (const c of (receipt.results && receipt.results.cases) || []) {
189
- if (c.case_status && c.case_status !== 'ok') continue;
190
- const e = byId.get(c.id) || {};
191
- e[c.mode] = c;
192
- byId.set(c.id, e);
193
- }
194
- const cases = [];
195
- for (const [id, e] of byId) {
196
- const w = armOf(e.with_skill);
197
- const b = armOf(e.baseline);
198
- if (!w || !b) continue;
199
- cases.push({ id, ...caseRule(b, w) });
200
- }
208
+ const cases = readCases(receipt);
201
209
  // Spec 036 A-036-3, THE RUNG R-5 NEVER HAD. R-5 reads its ladder off the cases, and
202
210
  // every rung below REGRESSED is a statement about what the cases showed. NO_EFFECT is
203
211
  // the bottom rung and it is a MEASUREMENT: "no separation detected at this sample size".
@@ -214,6 +222,31 @@ function receiptVerdict(receipt) {
214
222
  return { verdict, cases, drawsNeeded: verdict === 'UNDERPOWERED' ? receiptDrawsNeeded(cases) : null, notMeasured: null };
215
223
  }
216
224
 
225
+ // R-1 to R-4 over every readable case of a receipt, with none of R-5's refusals applied: a case
226
+ // with a failed arm, or an arm with no readable band, is left out. receiptVerdict reads its ladder
227
+ // off this list; lib/decision.js reads it BEFORE the incomplete rule (spec 062, register row 1),
228
+ // because a case that was measured and separated down is a measured regression however many other
229
+ // cases went unmeasured.
230
+ function readCases(receipt) {
231
+ const dups = duplicateCaseRows(receipt);
232
+ if (dups.length) throw new AmbiguousReceiptError(dups);
233
+ const byId = new Map();
234
+ for (const c of (receipt && receipt.results && receipt.results.cases) || []) {
235
+ if (c.case_status && c.case_status !== 'ok') continue;
236
+ const e = byId.get(c.id) || {};
237
+ e[c.mode] = c;
238
+ byId.set(c.id, e);
239
+ }
240
+ const cases = [];
241
+ for (const [id, e] of byId) {
242
+ const w = armOf(e.with_skill);
243
+ const b = armOf(e.baseline);
244
+ if (!w || !b) continue;
245
+ cases.push({ id, ...caseRule(b, w) });
246
+ }
247
+ return cases;
248
+ }
249
+
217
250
  // Derive the verdict object from a receipt. Returns
218
251
  // { verdict, delta, floor, model, message, color, drawsNeeded }
219
252
  function verdictFromReceipt(receipt) {
@@ -283,5 +316,6 @@ function githubOutputLines(receipt) {
283
316
 
284
317
  module.exports = {
285
318
  verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS,
286
- receiptVerdict, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
319
+ receiptVerdict, readCases, caseRule, armOf, drawsLine, receiptDrawsTaken, UNDERPOWERED_LINE, NOT_MEASURED_ROUTES,
320
+ AmbiguousReceiptError,
287
321
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "driftproof",
3
- "version": "0.11.2",
3
+ "version": "0.12.0",
4
4
  "description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
package/spec/RECEIPT.md CHANGED
@@ -1,5 +1,5 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
- # Driftproof receipt, spec v0.7
2
+ # Driftproof receipt, spec v0.9
3
3
 
4
4
  A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
5
5
  **with** and **without** the skill on one model version, with the judge **sampled**
@@ -12,8 +12,10 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
12
12
  (JSON Schema, draft 2020-12). This document is the human companion. Where they
13
13
  disagree, the schema wins.
14
14
 
15
- **Versioning.** The current schema is v0.7
15
+ **Versioning.** The current schema is v0.9
16
16
  ([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept frozen as
17
+ [`receipt.v0.8.schema.json`](./receipt.v0.8.schema.json),
18
+ [`receipt.v0.7.schema.json`](./receipt.v0.7.schema.json),
17
19
  [`receipt.v0.6.schema.json`](./receipt.v0.6.schema.json),
18
20
  [`receipt.v0.5.schema.json`](./receipt.v0.5.schema.json),
19
21
  [`receipt.v0.4.schema.json`](./receipt.v0.4.schema.json),
@@ -23,12 +25,94 @@ disagree, the schema wins.
23
25
  [`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
24
26
  schema by the receipt's own `schema_version`, so every earlier receipt still loads and
25
27
  validates against its own frozen schema, including every receipt behind the published
26
- reports, which the gate asserts on each run. v0.7 adds no field. It adds a verdict value
27
- a reader derives from fields every generation-sampled receipt already carries, and a
28
- v0.6 receipt restamped `"0.7"` is refused by the version constant, which is why v0.6 is
29
- frozen rather than widened (see *What v0.7 adds*). v0.6 was not additive for the
30
- validator either: a v0.5 receipt restamped `"0.6"` is refused, because v0.6 requires
31
- the receipt to say what answered it (see *What v0.6 adds*).
28
+ reports, which the gate asserts on each run. v0.9 is additive for a reader: it adds
29
+ optional fields and widens three values, so a v0.8 receipt restamped `"0.9"` validates;
30
+ v0.8 is frozen because its version constant refuses anything else (see *What v0.9 adds*).
31
+ v0.8 was additive over v0.7 in the same way (see *What v0.8 adds*).
32
+
33
+ ## What v0.9 adds: an imported receipt says what produced it
34
+
35
+ Spec 049, the importers for the two Anthropic formats: `claude plugin eval`'s
36
+ `aggregate-result.json` (`--from claude-plugin-eval`) and skill-creator's `benchmark.json`,
37
+ which skill-up also writes (`--from skill-creator`). The field mappings are in
38
+ [`docs/interop.md`](../docs/interop.md). Every field below is optional, and none is written
39
+ by a Driftproof run.
40
+
41
+ - **`run.harness`** `{name, version}`: the agent harness that produced the answers, as the
42
+ source names it (`claude-code` and `claudeVersion` for a plugin eval; skill-up's engine and
43
+ its observed CLI version). `version` is null when the source does not record it.
44
+ - **`run.import`** `{tool, format, format_version, source_sha256, imported_at}`: the document
45
+ the receipt came from, its own version field as written (null when it has none), the sha256
46
+ of its bytes, and when the import ran. `imported_at` is the only place the import time
47
+ appears; the run date comes from the source. Two optional lists: `sidecars` (other files
48
+ read beside the document, by name and sha256, such as skill-up's `result.json`) and
49
+ `notices` (one sentence per thing the source did not establish, such as an unknown model or a
50
+ date read from a directory name, and per choice a reader must know, such as errored runs
51
+ counted).
52
+ - **`skill.unit`**: `"skill"`, or `"plugin"` when a plugin was measured whole, every skill it
53
+ carries loaded at once (a plugin eval). Absent means a skill.
54
+ - **`results.cases[].activation`**: a list of `{indicator, fired, runs}`, one per unscored
55
+ skill-invocation indicator the source ran (a `tool_used: Skill` grader with `scored: false`):
56
+ in how many of the case's measured runs the skill fired. It is never part of a score, band
57
+ or comparison. On 15 Sep 2026 a native eval scored a skill +0.67 while this indicator read 0
58
+ of 3: the skill never fired.
59
+ - **`results.cases[].excluded_draws`**: runs the source recorded for the case and arm that are
60
+ kept out of its samples, each `{draw_index, reason}` (a run with an error, a run whose paid
61
+ graders were skipped).
62
+ - **`results.cases[].label`**: the source's display name for a case (`eval_name`). Cases are
63
+ identified by `id`, never by label.
64
+ - **`economics.<arm>.mean_total_tokens`**, and two `economics.basis` values:
65
+ `source-list-price-estimate` (the source tool's own cost estimate at list price, not a
66
+ Driftproof figure) and `source-reported` (tokens and time as the source reports them, no
67
+ cost).
68
+ - **Three widenings**, each for an imported receipt: `run.provider` may be `unknown` (the
69
+ source records no model); `run.date_utc` may be null, only on an `external` receipt whose
70
+ source records no date; and an absent model is `run.model_id` `"unknown"`, never a harness's
71
+ default.
72
+
73
+ Imported receipts stay DECLARED with surface `external`, every hash null, and `diff` keeps
74
+ refusing them.
75
+
76
+ ## What v0.8 adds: counts that say what they count, and a generation clock
77
+
78
+ Spec 043, from agentskills discussion #544 (comments 18348575, 18409751, 18409986,
79
+ 18530980 and 18531185). Until v0.8 `run.judge.samples` was required, so a receipt had no
80
+ way to say a count was unknown, and the two importers wrote one anyway: agent-skills-eval
81
+ wrote `1` whatever the source said, and skillgrade wrote its largest trial count, which is
82
+ a count of generations, into the judge-sample field.
83
+
84
+ - **`run.judge.samples` is optional** (#544 comments 18409751 rule 1, 18530980). Absent means unknown, never 1. Every reader in the
85
+ tree that shows it (the differ, `export --format summary-json`, the receipt summary)
86
+ says *unknown* or writes `null`.
87
+ - **Two counts, kept apart** (18348488, 18348575, 18530980). `generations_per_arm` (independent candidate generations;
88
+ 0 only where a source establishes zero) and `judge_samples_per_generation` (at least 1),
89
+ each optional, in a `counts` object. When both `run.judge.samples` and
90
+ `run.counts.judge_samples_per_generation` are present the loader refuses a receipt
91
+ where they differ.
92
+ - **The narrowest honest scope** (18530980). `run.counts` when a count is the same for every case of
93
+ every arm generated in this run; `run.arms.<mode>.counts` when it holds for one arm;
94
+ `results.cases[].counts` otherwise, and only for the cases where it is known. A reader
95
+ takes the narrowest scope present (`lib/counts.js` `resolveCount`). Nothing writes a
96
+ maximum, minimum or mean of differing counts as a count.
97
+ - **Archived arms** (18348488, 18348575, 18409751 rule 2). `run.arms.<mode>` may carry an arm's own `model_id`, `generated_at`
98
+ and `counts`. An arm with its own `generated_at` was generated in an earlier run: it
99
+ never inherits `run.counts`, and with no `counts` of its own its counts are unknown.
100
+ - **Clocks** (18409751 rule 2, 18409986, 18530980). `run.generated_at` is stamped by the runner before its first generation
101
+ call. `run.judged_at` and `run.grader_revision` (the judge template hash and the rubric
102
+ hashes) appear together, and only on a re-judge. `date_utc` keeps its value; its
103
+ description now says what the code writes (the row below).
104
+ - **`no_observations`** (18530980, and the operator's answer A-3 in spec 043). A case the source supplied no observation for is listed with
105
+ `case_status: "no_observations"`, contributes no figure, is excluded from every
106
+ aggregate and is named in `excluded_cases`. An imported task whose trials list is empty
107
+ carries `samples: []` and `generations_per_arm: 0`; a task with no trials field carries
108
+ neither, because its count is unknown.
109
+ - **The importers** (18530857, 18530980; the positive control of 18531185 is read on a native receipt and on uniform skillgrade trials). agent-skills-eval writes no count of either kind: its benchmark
110
+ format defines neither. skillgrade writes no judge-sample count (its format defines
111
+ none) and writes its trial counts as generation counts at the narrowest honest scope.
112
+ Both stay `DECLARED` and `external`, and skillgrade's rewards are kept exactly as
113
+ supplied.
114
+ - **The runner.** A native receipt carries the judge-sample count it was asked for and
115
+ each case's generation count as the draws it recorded, placed by the same rule.
32
116
 
33
117
  ## What v0.7 adds: a receipt that could not resolve the effect floor says so
34
118
 
@@ -183,7 +267,7 @@ carry a numeric `mean`.
183
267
  ## Which schema validates which receipt (the coexistence rule)
184
268
 
185
269
  - **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
186
- decides it, and at this revision that is **v0.7**.
270
+ decides it, and at this revision that is **v0.9**.
187
271
  - **A reader validates against the receipt's own `schema_version`**, never
188
272
  against the newest schema it happens to have. `validateReceipt()` selects the
189
273
  schema by that field, which is why a v0.1 receipt from the first report still
@@ -355,7 +439,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
355
439
 
356
440
  ## Fields
357
441
 
358
- ### `schema_version` (string, required): `"0.7"`.
442
+ ### `schema_version` (string, required): `"0.9"`.
359
443
 
360
444
  ### `skill` (object, required)
361
445
  | field | type | notes |
@@ -363,6 +447,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
363
447
  | `name` | string | From SKILL.md front-matter or its H1. |
364
448
  | `version` | string | Skill's declared version. |
365
449
  | `content_hash` | sha256 hex | Over `SKILL.md` + every bundled file (the `evals/` dir is **excluded** — hashed separately as `suite_hash`), path + content, path-sorted. |
450
+ | `unit` | `"skill"` \| `"plugin"` | **v0.9**, optional. `"plugin"` when a plugin was measured whole (a `claude plugin eval` import). |
366
451
  | `tokens` | integer | **v0.3.1, optional.** Estimated SKILL.md token size (coarse `chars/4` proxy) for the value-per-token axis. |
367
452
 
368
453
  ### `suite` (object, required)
@@ -377,14 +462,18 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
377
462
  |---|---|---|
378
463
  | `model_id` | string | Canonical model id run on. |
379
464
  | `model_release_date` | ISO date or `null` | Release date **if known**, else `null`. |
380
- | `provider` | `"anthropic"` \| `"openai"` | **v0.3.1.** The two-axis provider (registry `provider`, else inferred from the id). |
465
+ | `provider` | `"anthropic"` \| `"openai"` \| `"unknown"` | **v0.3.1.** The two-axis provider (registry `provider`, else inferred from the id). **v0.9:** `"unknown"` on an imported receipt whose source records no model. |
381
466
  | `surface` | `"api"` \| `"claude-cli"` \| `"openai-api"` \| `"openai-cli"` | `api` = Anthropic Messages API; `claude-cli` = spawned `claude -p` (subscription); `openai-api` = Chat-Completions-compatible API (base_url configurable); `openai-cli` = Codex subscription (`codex exec`). |
382
467
  | `surface_overhead_note` | string | **v0.3.1, optional.** On `openai-cli`: the fixed Codex base-instruction preamble (~12–15k input tokens/call) the harness prepends and does not control. |
383
468
  | `runner_version` | string | Runner version, for reproducibility. |
384
- | `date_utc` | ISO 8601 UTC | When the run finished. |
469
+ | `date_utc` | ISO 8601 UTC | As `lib/run.js` writes it: when the run's calls finished if the CLI stamps it, or the stamp a caller passes, which for a batch is one stamp shared by every receipt in the batch. For when generation began, read `generated_at` (v0.8). Before v0.8 this row said *when the run finished*, which was true of the CLI and not of a batch. **v0.9:** null only on an `external` receipt whose source records no date. |
385
470
  | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). **v0.6:** `"unregistered"` is import-only; the runner refuses an unregistered id before any call (spec 026 AC-10). |
386
471
  | `transcripts` | `"retained-local"` \| `"hashes-only"` | **v0.3.** Whether the raw generations + judge outputs were retained on disk (see § Transcripts). |
387
- | `judge` | object | `{ samples, temperature, sampling, surface }` — see below. |
472
+ | `judge` | object | `{ samples, temperature, sampling, surface }`, see below. `samples` is optional from v0.8: absent means unknown. |
473
+ | `generated_at` | ISO 8601 UTC | **v0.8**, optional. When generation began, stamped before the first generation call. |
474
+ | `judged_at`, `grader_revision` | ISO 8601 UTC, object | **v0.8**, optional, together or not at all: a re-judge of frozen outputs, and the grading definition it applied. |
475
+ | `counts`, `arms` | object | **v0.8**, optional. See *What v0.8 adds*. |
476
+ | `harness`, `import` | object | **v0.9**, optional, imported receipts. See *What v0.9 adds*. |
388
477
 
389
478
  **`run.judge`** records how grading was done: `samples` (judge calls per case),
390
479
  `temperature` (a number when the surface lets us set it — the `api` surface pins
@@ -411,6 +500,7 @@ receipt says so.
411
500
  | `reason` | string | One-line judge rationale (optional). |
412
501
  | `checks` | array | **v0.3.1, optional.** Deterministic post-check results, each `{ name, kind, pass }` (`kind ∈ regex/contains/not_contains/min_length`). Run alongside the judge; a **separate column**, **not** folded into `outcome`/the band verdict. |
413
502
  | `judge` | object | `{ model_id, rubric_hash }` — who graded and a hash binding the grade to the exact rubric + judge system prompt. |
503
+ | `activation`, `excluded_draws`, `label` | array, array, string | **v0.9**, optional, imported receipts. Whether the skill fired (never a score), the runs kept out of `samples` with their reasons, and the source's display name. See *What v0.9 adds*. |
414
504
  - **`aggregates`** — `{ with_skill, baseline }`, each
415
505
  `{ case_count, pass_count, borderline_count?, mean_score, stddev }`. Here
416
506
  `stddev` is the **suite dispersion**: the stddev of the per-case means across