driftproof 0.8.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/usage.js CHANGED
@@ -42,6 +42,27 @@
42
42
 
43
43
  function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
44
44
 
45
+ // ── what answered (spec 026, AC-2 and AC-8) ──────────────────────────────────
46
+ // Every lane also reports, when it can, WHICH model served the call and WHY the
47
+ // reply stopped. Both are read here, never guessed: a surface that says nothing
48
+ // yields null for both, and the receipt records that it said nothing.
49
+ //
50
+ // reportedModels the ids the surface named, VERBATIM (the real claude CLI
51
+ // keys the call's usage under the undated form and puts a
52
+ // dated side entry beside it; an api response names one
53
+ // model). Null when the surface names no model. The runner
54
+ // compares on canonical ids (canonicalModelId: a trailing
55
+ // -YYYYMMDD is not part of the identity), never the reader.
56
+ // stopReason the surface's own word for why the reply ended (end_turn,
57
+ // max_tokens, stop, length ...). Null when it gives none.
58
+ function canonicalModelId(id) { return String(id || '').replace(/-\d{8}$/, ''); }
59
+ function reportedModelsOf(map) {
60
+ if (!map || typeof map !== 'object' || Array.isArray(map)) return null;
61
+ const ids = [...new Set(Object.keys(map).filter((k) => k))].sort();
62
+ return ids.length ? ids : null;
63
+ }
64
+ function stopReasonOf(v) { return typeof v === 'string' && v ? v : null; }
65
+
45
66
  // The empty/unknown usage record. Deliberately null (not zeros) so "the surface
46
67
  // did not tell us" never reads as "the call cost nothing".
47
68
  function emptyUsage() {
@@ -75,6 +96,9 @@ function parseClaudeCliJson(stdout) {
75
96
  text: typeof j.result === 'string' ? j.result : '',
76
97
  usage,
77
98
  isError: j.is_error === true,
99
+ // v0.6: the CLI's own stop reason and the models it says served the call.
100
+ stopReason: stopReasonOf(j.stop_reason),
101
+ reportedModels: reportedModelsOf(j.modelUsage),
78
102
  // The CLI reports its own dollar figure. Recorded here for completeness but
79
103
  // NOT used: costs are computed uniformly from the frozen pricing snapshot so
80
104
  // three substrates are on one basis (see lib/value.js).
@@ -85,7 +109,9 @@ function parseClaudeCliJson(stdout) {
85
109
  // ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
86
110
  // One JSON object per line. Usage rides the terminal `turn.completed` event;
87
111
  // input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
88
- // skipped (the stream also carries progress events we do not model).
112
+ // skipped (the stream also carries progress events we do not model). The
113
+ // stream names neither the model that served the turn nor a stop reason, so
114
+ // both read null here and the receipt says the surface did not report them.
89
115
  function parseCodexJsonl(stdout) {
90
116
  const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
91
117
  let usage = null;
@@ -103,16 +129,30 @@ function parseCodexJsonl(stdout) {
103
129
  wall_ms: null,
104
130
  };
105
131
  }
106
- // The final message is normally read from the -o file; this is a fallback.
132
+ // The final message IS the stream: the last `item.completed` /
133
+ // `agent_message` event carries it. Nothing is read from a file after the
134
+ // call, on either spawn mode (spec 022 AC-9; the earlier comment here
135
+ // described an output file the lane no longer reads, F-022-3).
107
136
  if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
108
137
  text = ev.item.text;
109
138
  }
110
139
  }
111
- return { usage, text };
140
+ return { usage, text, stopReason: null, reportedModels: null };
112
141
  }
113
142
 
114
143
  // ── anthropic/api ─────────────────────────────────────────────────────────────
115
144
  // The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
145
+ // The response object also names the model that served it (`model`) and why
146
+ // it stopped (`stop_reason`); read here from the object, null when absent.
147
+ // Exercised against tests/fixtures/response-anthropic-api.json, which is
148
+ // SYNTHESISED from the API reference, not captured from a call.
149
+ function readAnthropicApiResponse(resp) {
150
+ const r = resp && typeof resp === 'object' ? resp : {};
151
+ return {
152
+ stopReason: stopReasonOf(r.stop_reason),
153
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
154
+ };
155
+ }
116
156
  function parseAnthropicApiUsage(u) {
117
157
  if (!u) return emptyUsage();
118
158
  const cacheRead = n(u.cache_read_input_tokens);
@@ -127,7 +167,18 @@ function parseAnthropicApiUsage(u) {
127
167
 
128
168
  // ── openai/api (Chat Completions-compatible) ──────────────────────────────────
129
169
  // prompt_tokens is the total; the cached portion, when present, is nested under
130
- // prompt_tokens_details.cached_tokens.
170
+ // prompt_tokens_details.cached_tokens. The response also names the serving
171
+ // model (`model`) and the first choice's `finish_reason`; read here, null
172
+ // when absent. Exercised against tests/fixtures/response-openai-api.json,
173
+ // SYNTHESISED from the API reference, not captured from a call.
174
+ function readOpenaiApiResponse(resp) {
175
+ const r = resp && typeof resp === 'object' ? resp : {};
176
+ const choice = Array.isArray(r.choices) && r.choices[0] && typeof r.choices[0] === 'object' ? r.choices[0] : {};
177
+ return {
178
+ stopReason: stopReasonOf(choice.finish_reason),
179
+ reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
180
+ };
181
+ }
131
182
  function parseOpenaiApiUsage(u) {
132
183
  if (!u) return emptyUsage();
133
184
  const details = u.prompt_tokens_details || {};
@@ -165,4 +216,5 @@ function normalizeUsage(u) {
165
216
  module.exports = {
166
217
  emptyUsage, hasUsage, sumUsage, normalizeUsage,
167
218
  parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
219
+ readAnthropicApiResponse, readOpenaiApiResponse, canonicalModelId, reportedModelsOf,
168
220
  };
package/lib/value.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // SPDX-License-Identifier: Apache-2.0
2
2
  'use strict';
3
3
 
4
+ const { caseFailed } = require('./receipt');
5
+
4
6
  const { EFFECT_FLOOR } = require('../config');
5
7
 
6
8
  // Economics of a run — what a skill COSTS to run, alongside whether it helps.
@@ -204,7 +206,7 @@ function armEconomics(rows, price) {
204
206
  // surface the run surface, to label the cost basis honestly
205
207
  function computeEconomics({ cases, modelId, judgeModelId, pricingSnapshot, surface, meteredSurface }) {
206
208
  const price = (pricingSnapshot && pricingSnapshot.models && pricingSnapshot.models[modelId]) || null;
207
- const ok = (cases || []).filter((c) => c.case_status !== 'failed_timeout');
209
+ const ok = (cases || []).filter((c) => !caseFailed(c));
208
210
  const withRows = ok.filter((c) => c.mode === 'with_skill');
209
211
  const baseRows = ok.filter((c) => c.mode === 'baseline');
210
212
 
@@ -322,7 +324,7 @@ function receiptCostBreakdown(receipt) {
322
324
  let judge = 0;
323
325
  let missingJudgeRate = null;
324
326
  for (const c of ((receipt.results || {}).cases || [])) {
325
- if (c.case_status === 'failed_timeout') continue;
327
+ if (caseFailed(c)) continue;
326
328
  const g = costForUsage(c.usage, genRate);
327
329
  if (g != null) generation += g;
328
330
  if (c.judge_usage) {
package/lib/verdict.js CHANGED
@@ -43,7 +43,12 @@ function verdictFromReceipt(receipt) {
43
43
  // Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
44
44
  // verdicted — we did not run the suite, so we do not certify the outcome.
45
45
  const level = (receipt && receipt.verification_level) || 'TESTED';
46
- const measured = level === 'TESTED' && typeof cmp.delta === 'number';
46
+ // Spec 026 AC-4: an INCOMPLETE receipt (its run block's status field reads
47
+ // incomplete: a case had an arm that could not be measured) is not verdicted
48
+ // either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
49
+ // compute a verdict from one; the badge is the same reader with a shorter path.
50
+ const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
51
+ const measured = level === 'TESTED' && typeof cmp.delta === 'number' && !incomplete;
47
52
  let verdict;
48
53
  if (!measured) verdict = 'NOT_MEASURED';
49
54
  else if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "driftproof",
3
- "version": "0.8.1",
3
+ "version": "0.10.0",
4
4
  "description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
5
5
  "license": "Apache-2.0",
6
6
  "keywords": [
@@ -45,6 +45,7 @@
45
45
  "node": ">=22"
46
46
  },
47
47
  "devDependencies": {
48
- "html-validate": "^11.11.0"
48
+ "html-validate": "^11.11.0",
49
+ "playwright": "^1.62.1"
49
50
  }
50
51
  }
package/spec/RECEIPT.md CHANGED
@@ -1,5 +1,5 @@
1
1
  <!-- SPDX-License-Identifier: Apache-2.0 -->
2
- # Driftproof receipt — spec v0.5
2
+ # Driftproof receipt — spec v0.6
3
3
 
4
4
  A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
5
5
  **with** and **without** the skill on one model version, with the judge **sampled**
@@ -11,16 +11,64 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
11
11
  (JSON Schema, draft 2020-12). This document is the human companion. Where they
12
12
  disagree, the schema wins.
13
13
 
14
- **Versioning.** The current schema is v0.4
14
+ **Versioning.** The current schema is v0.6
15
15
  ([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept frozen as
16
+ [`receipt.v0.5.schema.json`](./receipt.v0.5.schema.json),
17
+ [`receipt.v0.4.schema.json`](./receipt.v0.4.schema.json),
16
18
  [`receipt.v0.3.1.schema.json`](./receipt.v0.3.1.schema.json),
17
19
  [`receipt.v0.3.schema.json`](./receipt.v0.3.schema.json),
18
20
  [`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
19
21
  [`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
20
- schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3 and v0.3.1
21
- receipts still load and validate — including every receipt behind the four
22
- published reports, which the gate asserts on each run. v0.4 is an **additive**
23
- bump: everything it adds is optional, and no earlier receipt is invalidated.
22
+ schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3, v0.3.1, v0.4
23
+ and v0.5 receipts still load and validate — including every receipt behind the
24
+ published reports, which the gate asserts on each run. v0.6 is additive for a
25
+ reader that ignores unknown fields and **not additive for the validator**: a
26
+ v0.5 receipt restamped `"0.6"` is refused, because v0.6 requires the receipt to
27
+ say what answered it (see *What v0.6 adds*). No earlier receipt is invalidated;
28
+ each validates against its own frozen schema.
29
+
30
+ ## What v0.6 adds: the receipt says what answered it
31
+
32
+ Spec 026 (receipt integrity). Six ways the v0.5 runner wrote a receipt that read
33
+ as more than it was, each closed by a field the receipt now carries and the
34
+ schema now requires:
35
+
36
+ - **`run.answered_by`** `{ kind, attested, reported_model, reported_models,
37
+ isolation }`. `kind` is `model` (a model surface answered), `stub` (the
38
+ canned stub answered and nothing was measured) or `external` (another tool's
39
+ harness answered; the receipt was imported). `attested` is true only when the
40
+ surface echoed a model id for every draw and it matched the requested
41
+ canonical id; `reported_model` and `reported_models` are what it echoed
42
+ (null when nothing). `isolation` names the spawn path: `eval-user`,
43
+ `same-user` (`--trusted-skill`) or `none`. A stub run also records
44
+ `run.surface: "stub"` and `run.judge.surface: "stub"`, and the schema refuses
45
+ `TESTED` unless `answered_by.kind` is `model`: a stub receipt is
46
+ `UNVERIFIED` by schema, not only by the runner.
47
+ - **`run.judge.model_id`** and **`run.judge.prompt_template_hash`**: which
48
+ judge ran, and a sha256 over the grading template with its three slots empty.
49
+ A different template is a different judge; `diff` names the field and
50
+ computes no verdict across one. The template hash is null only on an
51
+ imported receipt.
52
+ - **`stop_reason`** and **`truncated`** on every draw, and
53
+ **`generation.n_truncated`** per case and arm. A draw cut at the output cap
54
+ is unmeasured and never judged; a partial answer scored as a whole one was
55
+ the v0.5 shape.
56
+ - **`case_status: "failed_unmeasured"`**: every draw of the arm was unmeasured
57
+ for a reason that is not a timeout (an empty generation, a judge output with
58
+ no score, a truncated draw). Recorded without fabricated samples or hashes
59
+ exactly as `failed_timeout` is, and excluded pairwise. An absent output is
60
+ not a measurement; a zero is.
61
+ - **`results.aggregates.band_rule`**: the formula the aggregate band is derived
62
+ by, stated on the receipt so a reader recomputes it from `results.cases`.
63
+ - **Bands the formula cannot form are null, never `0`.** An arm's `stddev` is
64
+ null only when its `case_count` is below 2 (the schema refuses a null on a
65
+ two-case arm); `comparison.delta_uncertainty` is null only beside
66
+ `delta_uncertainty_unavailable: "single_case"` or `"no_cases"`, and
67
+ `mean_score` and the comparison are null only under `no_cases`.
68
+
69
+ A v0.5 receipt carries none of these, so the validator refuses one restamped
70
+ `"0.6"`; v0.5 stays frozen at `receipt.v0.5.schema.json` and every archived
71
+ receipt validates against the version it names.
24
72
 
25
73
  ## What v0.5 adds: the generation is sampled too
26
74
 
@@ -95,7 +143,7 @@ carry a numeric `mean`.
95
143
  ## Which schema validates which receipt (the coexistence rule)
96
144
 
97
145
  - **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
98
- decides it, and at this revision that is **v0.5**.
146
+ decides it, and at this revision that is **v0.6**.
99
147
  - **A reader validates against the receipt's own `schema_version`**, never
100
148
  against the newest schema it happens to have. `validateReceipt()` selects the
101
149
  schema by that field, which is why a v0.1 receipt from the first report still
@@ -247,7 +295,11 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
247
295
  outputs were also written to `transcripts/<receipt-id>/` (under
248
296
  `--keep-transcripts`), else `"hashes-only"` (the default — only the hashes).
249
297
  - **`run.registry`** — `"registered"` when `model_id` resolved in the model
250
- registry (`config/models.json`), else `"unregistered"` (the run still executed;
298
+ registry (`config/models.json`); `"unregistered"` is import-only since v0.6, the runner refuses it (spec 026 AC-10):
299
+ a run of ours refuses an id the registry does not carry before any call,
300
+ naming the registry path, so no receipt we write says it; an imported
301
+ receipt may, its model field being another tool's word (before v0.6 the
302
+ run was not refused;
251
303
  cost was estimated with a conservative default price).
252
304
  - These four fields are **required** in a v0.3 receipt. Everything else is
253
305
  unchanged from v0.2.
@@ -263,7 +315,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
263
315
 
264
316
  ## Fields
265
317
 
266
- ### `schema_version` (string, required) — `"0.4"`.
318
+ ### `schema_version` (string, required) — `"0.6"`.
267
319
 
268
320
  ### `skill` (object, required)
269
321
  | field | type | notes |
@@ -290,7 +342,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
290
342
  | `surface_overhead_note` | string | **v0.3.1, optional.** On `openai-cli`: the fixed Codex base-instruction preamble (~12–15k input tokens/call) the harness prepends and does not control. |
291
343
  | `runner_version` | string | Runner version, for reproducibility. |
292
344
  | `date_utc` | ISO 8601 UTC | When the run finished. |
293
- | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). |
345
+ | `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). **v0.6:** `"unregistered"` is import-only; the runner refuses an unregistered id before any call (spec 026 AC-10). |
294
346
  | `transcripts` | `"retained-local"` \| `"hashes-only"` | **v0.3.** Whether the raw generations + judge outputs were retained on disk (see § Transcripts). |
295
347
  | `judge` | object | `{ samples, temperature, sampling, surface }` — see below. |
296
348
 
@@ -358,7 +410,12 @@ Every generation and every judge sample is hashed into the receipt
358
410
  generations and judge outputs are written to `transcripts/<receipt_hash>/` (one
359
411
  JSON per case+mode, plus an `index.json`), and the receipt records
360
412
  `transcripts: "retained-local"`. Without the flag the receipt records
361
- `transcripts: "hashes-only"` and only the hashes are kept. The transcript
413
+ `transcripts: "hashes-only"` and only the hashes are kept. **What is retained
414
+ is the last measured draw of each case and mode only**: the runner keeps one
415
+ transcript per case and mode and overwrites it on every draw, so the per-case
416
+ top-level hashes (which are that draw's) can be re-derived from disk, and the
417
+ draw-level hashes of earlier draws are checkable against nothing the runner
418
+ wrote (spec 026, stated as the bound rather than moved). The transcript
362
419
  directory is **gitignored by default** — raw model text is never committed. The
363
420
  default for trigger-initiated runs is `retained-local` (disk is cheap, audits are
364
421
  not). To verify a retained receipt: re-hash each transcript file and compare