driftproof 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/value.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { caseFailed } = require('./receipt');
5
+
4
6
  const { EFFECT_FLOOR } = require('../config');
5
7
 
6
8
  // Economics of a run — what a skill COSTS to run, alongside whether it helps.
@@ -204,7 +206,7 @@ function armEconomics(rows, price) {
204
206
  // surface the run surface, to label the cost basis honestly
205
207
  function computeEconomics({ cases, modelId, judgeModelId, pricingSnapshot, surface, meteredSurface }) {
206
208
  const price = (pricingSnapshot && pricingSnapshot.models && pricingSnapshot.models[modelId]) || null;
207
- const ok = (cases || []).filter((c) => c.case_status !== 'failed_timeout');
209
+ const ok = (cases || []).filter((c) => !caseFailed(c));
208
210
  const withRows = ok.filter((c) => c.mode === 'with_skill');
209
211
  const baseRows = ok.filter((c) => c.mode === 'baseline');
210
212
 
@@ -322,7 +324,7 @@ function receiptCostBreakdown(receipt) {
322
324
  let judge = 0;
323
325
  let missingJudgeRate = null;
324
326
  for (const c of ((receipt.results || {}).cases || [])) {
325
- if (c.case_status === 'failed_timeout') continue;
327
+ if (caseFailed(c)) continue;
326
328
  const g = costForUsage(c.usage, genRate);
327
329
  if (g != null) generation += g;
328
330
  if (c.judge_usage) {
package/lib/verdict.js CHANGED
@@ -1,6 +1,7 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const crypto = require('crypto');
4
5
  const { EFFECT_FLOOR } = require('../config');
5
6
 
6
7
  // Single-receipt verdict + shields.io badge.
@@ -42,7 +43,12 @@ function verdictFromReceipt(receipt) {
42
43
  // Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
43
44
  // verdicted — we did not run the suite, so we do not certify the outcome.
44
45
  const level = (receipt && receipt.verification_level) || 'TESTED';
45
- const measured = level === 'TESTED' && typeof cmp.delta === 'number';
46
+ // Spec 026 AC-4: an INCOMPLETE receipt (its run block's status field reads
47
+ // incomplete: a case had an arm that could not be measured) is not verdicted
48
+ // either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
49
+ // compute a verdict from one; the badge is the same reader with a shorter path.
50
+ const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
51
+ const measured = level === 'TESTED' && typeof cmp.delta === 'number' && !incomplete;
46
52
  let verdict;
47
53
  if (!measured) verdict = 'NOT_MEASURED';
48
54
  else if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
@@ -72,14 +78,39 @@ function badgeEndpoint(receipt, { label = 'driftproof' } = {}) {
72
78
  }
73
79
 
74
80
  // Lines suitable for appending to a GitHub Actions $GITHUB_OUTPUT file.
81
+ //
82
+ // Spec 023 (audit A6). Each entry is written in GitHub's multiline form,
83
+ //
84
+ // name<<ghadelim_<32 hex>
85
+ // value
86
+ // ghadelim_<32 hex>
87
+ //
88
+ // with a delimiter drawn fresh from crypto.randomBytes for EVERY entry, so no
89
+ // value can close a block early; and any value carrying \r or \n is refused
90
+ // with a throw before anything is written. The bare `name=value` form this used
91
+ // to emit let a model_id with a newline in it (any string a receipt carries; a
92
+ // .driftproofrc in a skill directory sets it) append a second, attacker-chosen
93
+ // entry, and that entry was then interpolated into the enforce step's shell.
94
+ // Refusing is the honest writer: a line break in a model id is not a value
95
+ // anybody meant to publish. Entries are joined by '\n' with no trailing newline
96
+ // so the caller's console.log stays the one that terminates the text.
97
+ function githubOutputEntry(name, value) {
98
+ const val = String(value);
99
+ if (/[\r\n]/.test(val)) {
100
+ throw new Error(`refusing to write $GITHUB_OUTPUT: the value of "${name}" contains a line break (${JSON.stringify(val)})`);
101
+ }
102
+ const delim = `ghadelim_${crypto.randomBytes(16).toString('hex')}`;
103
+ return `${name}<<${delim}\n${val}\n${delim}`;
104
+ }
105
+
75
106
  function githubOutputLines(receipt) {
76
107
  const v = verdictFromReceipt(receipt);
77
108
  return [
78
- `verdict=${v.verdict}`,
79
- `delta=${v.delta}`,
80
- `message=${v.message}`,
81
- `color=${v.color}`,
82
- ].join('\n');
109
+ ['verdict', v.verdict],
110
+ ['delta', v.delta],
111
+ ['message', v.message],
112
+ ['color', v.color],
113
+ ].map(([k, val]) => githubOutputEntry(k, val)).join('\n');
83
114
  }
84
115
 
85
- module.exports = { verdictFromReceipt, badgeEndpoint, githubOutputLines, shortModel, VERDICTS };
116
+ module.exports = { verdictFromReceipt, badgeEndpoint, githubOutputLines, githubOutputEntry, shortModel, VERDICTS };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "driftproof",
3
- "version": "0.8.0",
3
+ "version": "0.9.0",
4
4
  "description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
@@ -45,6 +45,7 @@
45
45
  "node": ">=22"
46
46
  },
47
47
  "devDependencies": {
48
- "html-validate": "^11.11.0"
48
+ "html-validate": "^11.11.0",
49
+ "playwright": "^1.62.1"
49
50
  }
50
51
  }
package/spec/RECEIPT.md CHANGED
@@ -1,5 +1,5 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
- # Driftproof receipt — spec v0.5
2
+ # Driftproof receipt — spec v0.6
3
3
 
4
4
  A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
5
5
  **with** and **without** the skill on one model version, with the judge **sampled**
@@ -11,16 +11,64 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
11
11
  (JSON Schema, draft 2020-12). This document is the human companion. Where they
12
12
  disagree, the schema wins.
13
13
 
14
- **Versioning.** The current schema is v0.4
14
+ **Versioning.** The current schema is v0.6
15
15
  ([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept frozen as
16
+ [`receipt.v0.5.schema.json`](./receipt.v0.5.schema.json),
17
+ [`receipt.v0.4.schema.json`](./receipt.v0.4.schema.json),
16
18
  [`receipt.v0.3.1.schema.json`](./receipt.v0.3.1.schema.json),
17
19
  [`receipt.v0.3.schema.json`](./receipt.v0.3.schema.json),
18
20
  [`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
19
21
  [`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
20
- schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3 and v0.3.1
21
- receipts still load and validate — including every receipt behind the four
22
- published reports, which the gate asserts on each run. v0.4 is an **additive**
23
- bump: everything it adds is optional, and no earlier receipt is invalidated.
22
+ schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3, v0.3.1, v0.4
23
+ and v0.5 receipts still load and validate — including every receipt behind the
24
+ published reports, which the gate asserts on each run. v0.6 is additive for a
25
+ reader that ignores unknown fields and **not additive for the validator**: a
26
+ v0.5 receipt restamped `"0.6"` is refused, because v0.6 requires the receipt to
27
+ say what answered it (see *What v0.6 adds*). No earlier receipt is invalidated;
28
+ each validates against its own frozen schema.
29
+
30
+ ## What v0.6 adds: the receipt says what answered it
31
+
32
+ Spec 026 (receipt integrity). Six ways the v0.5 runner wrote a receipt that read
33
+ as more than it was, each closed by a field the receipt now carries and the
34
+ schema now requires:
35
+
36
+ - **`run.answered_by`** `{ kind, attested, reported_model, reported_models,
37
+ isolation }`. `kind` is `model` (a model surface answered), `stub` (the
38
+ canned stub answered and nothing was measured) or `external` (another tool's
39
+ harness answered; the receipt was imported). `attested` is true only when the
40
+ surface echoed a model id for every draw and it matched the requested
41
+ canonical id; `reported_model` and `reported_models` are what it echoed
42
+ (null when nothing). `isolation` names the spawn path: `eval-user`,
43
+ `same-user` (`--trusted-skill`) or `none`. A stub run also records
44
+ `run.surface: "stub"` and `run.judge.surface: "stub"`, and the schema refuses
45
+ `TESTED` unless `answered_by.kind` is `model`: a stub receipt is
46
+ `UNVERIFIED` by schema, not only by the runner.
47
+ - **`run.judge.model_id`** and **`run.judge.prompt_template_hash`**: which
48
+ judge ran, and a sha256 over the grading template with its three slots empty.
49
+ A different template is a different judge; `diff` names the field and
50
+ computes no verdict across one. The template hash is null only on an
51
+ imported receipt.
52
+ - **`stop_reason`** and **`truncated`** on every draw, and
53
+ **`generation.n_truncated`** per case and arm. A draw cut at the output cap
54
+ is unmeasured and never judged; a partial answer scored as a whole one was
55
+ the v0.5 shape.
56
+ - **`case_status: "failed_unmeasured"`**: every draw of the arm was unmeasured
57
+ for a reason that is not a timeout (an empty generation, a judge output with
58
+ no score, a truncated draw). Recorded without fabricated samples or hashes
59
+ exactly as `failed_timeout` is, and excluded pairwise. An absent output is
60
+ not a measurement; a zero is.
61
+ - **`results.aggregates.band_rule`**: the formula the aggregate band is derived
62
+ by, stated on the receipt so a reader recomputes it from `results.cases`.
63
+ - **Bands the formula cannot form are null, never `0`.** An arm's `stddev` is
64
+ null only when its `case_count` is below 2 (the schema refuses a null on a
65
+ two-case arm); `comparison.delta_uncertainty` is null only beside
66
+ `delta_uncertainty_unavailable: "single_case"` or `"no_cases"`, and
67
+ `mean_score` and the comparison are null only under `no_cases`.
68
+
69
+ A v0.5 receipt carries none of these, so the validator refuses one restamped
70
+ `"0.6"`; v0.5 stays frozen at `receipt.v0.5.schema.json` and every archived
71
+ receipt validates against the version it names.
24
72
 
25
73
  ## What v0.5 adds: the generation is sampled too
26
74
 
@@ -95,7 +143,7 @@ carry a numeric `mean`.
95
143
  ## Which schema validates which receipt (the coexistence rule)
96
144
 
97
145
  - **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
98
- decides it, and at this revision that is **v0.5**.
146
+ decides it, and at this revision that is **v0.6**.
99
147
  - **A reader validates against the receipt's own `schema_version`**, never
100
148
  against the newest schema it happens to have. `validateReceipt()` selects the
101
149
  schema by that field, which is why a v0.1 receipt from the first report still
@@ -247,7 +295,11 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
247
295
  outputs were also written to `transcripts/<receipt-id>/` (under
248
296
  `--keep-transcripts`), else `"hashes-only"` (the default — only the hashes).
249
297
  - **`run.registry`** — `"registered"` when `model_id` resolved in the model
250
- registry (`config/models.json`), else `"unregistered"` (the run still executed;
298
+ registry (`config/models.json`); `"unregistered"` is import-only since v0.6, the runner refuses it (spec 026 AC-10):
299
+ a run of ours refuses an id the registry does not carry before any call,
300
+ naming the registry path, so no receipt we write says it; an imported
301
+ receipt may, its model field being another tool's word (before v0.6 the
302
+ run was not refused;
251
303
  cost was estimated with a conservative default price).
252
304
  - These four fields are **required** in a v0.3 receipt. Everything else is
253
305
  unchanged from v0.2.
@@ -263,7 +315,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
263
315
 
264
316
  ## Fields
265
317
 
266
- ### `schema_version` (string, required) — `"0.4"`.
318
+ ### `schema_version` (string, required) — `"0.6"`.
267
319
 
268
320
  ### `skill` (object, required)
269
321
  | field | type | notes |
@@ -290,7 +342,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
290
342
  | `surface_overhead_note` | string | **v0.3.1, optional.** On `openai-cli`: the fixed Codex base-instruction preamble (~12–15k input tokens/call) the harness prepends and does not control. |
291
343
  | `runner_version` | string | Runner version, for reproducibility. |
292
344
  | `date_utc` | ISO 8601 UTC | When the run finished. |
293
- | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). |
345
+ | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). **v0.6:** `"unregistered"` is import-only; the runner refuses an unregistered id before any call (spec 026 AC-10). |
294
346
  | `transcripts` | `"retained-local"` \| `"hashes-only"` | **v0.3.** Whether the raw generations + judge outputs were retained on disk (see § Transcripts). |
295
347
  | `judge` | object | `{ samples, temperature, sampling, surface }` — see below. |
296
348
 
@@ -358,7 +410,12 @@ Every generation and every judge sample is hashed into the receipt
358
410
  generations and judge outputs are written to `transcripts/<receipt_hash>/` (one
359
411
  JSON per case+mode, plus an `index.json`), and the receipt records
360
412
  `transcripts: "retained-local"`. Without the flag the receipt records
361
- `transcripts: "hashes-only"` and only the hashes are kept. The transcript
413
+ `transcripts: "hashes-only"` and only the hashes are kept. **What is retained
414
+ is the last measured draw of each case and mode only**: the runner keeps one
415
+ transcript per case and mode and overwrites it on every draw, so the per-case
416
+ top-level hashes (which are that draw's) can be re-derived from disk, and the
417
+ draw-level hashes of earlier draws are checkable against nothing the runner
418
+ wrote (spec 026, stated as the bound rather than moved). The transcript
362
419
  directory is **gitignored by default** — raw model text is never committed. The
363
420
  default for trigger-initiated runs is `retained-local` (disk is cheap, audits are
364
421
  not). To verify a retained receipt: re-hash each transcript file and compare