driftproof 0.8.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +87 -16
- package/bin/driftproof +289 -24
- package/config.js +26 -2
- package/lib/checks.js +28 -6
- package/lib/decision.js +425 -0
- package/lib/diff.js +58 -4
- package/lib/importers.js +8 -3
- package/lib/judge.js +69 -12
- package/lib/models.js +52 -11
- package/lib/provider.js +35 -11
- package/lib/receipt.js +51 -11
- package/lib/run.js +221 -20
- package/lib/runner.js +35 -3
- package/lib/skill.js +26 -4
- package/lib/stats.js +12 -3
- package/lib/usage.js +56 -4
- package/lib/value.js +4 -2
- package/lib/verdict.js +6 -1
- package/package.json +3 -2
- package/spec/RECEIPT.md +68 -11
- package/spec/receipt.schema.json +299 -54
- package/spec/receipt.v0.5.schema.json +1318 -0
package/lib/usage.js
CHANGED
|
@@ -42,6 +42,27 @@
|
|
|
42
42
|
|
|
43
43
|
function n(x) { return Number.isFinite(Number(x)) ? Number(x) : 0; }
|
|
44
44
|
|
|
45
|
+
// ── what answered (spec 026, AC-2 and AC-8) ──────────────────────────────────
|
|
46
|
+
// Every lane also reports, when it can, WHICH model served the call and WHY the
|
|
47
|
+
// reply stopped. Both are read here, never guessed: a surface that says nothing
|
|
48
|
+
// yields null for both, and the receipt records that it said nothing.
|
|
49
|
+
//
|
|
50
|
+
// reportedModels the ids the surface named, VERBATIM (the real claude CLI
|
|
51
|
+
// keys the call's usage under the undated form and puts a
|
|
52
|
+
// dated side entry beside it; an api response names one
|
|
53
|
+
// model). Null when the surface names no model. The runner
|
|
54
|
+
// compares on canonical ids (canonicalModelId: a trailing
|
|
55
|
+
// -YYYYMMDD is not part of the identity), never the reader.
|
|
56
|
+
// stopReason the surface's own word for why the reply ended (end_turn,
|
|
57
|
+
// max_tokens, stop, length ...). Null when it gives none.
|
|
58
|
+
function canonicalModelId(id) { return String(id || '').replace(/-\d{8}$/, ''); }
|
|
59
|
+
function reportedModelsOf(map) {
|
|
60
|
+
if (!map || typeof map !== 'object' || Array.isArray(map)) return null;
|
|
61
|
+
const ids = [...new Set(Object.keys(map).filter((k) => k))].sort();
|
|
62
|
+
return ids.length ? ids : null;
|
|
63
|
+
}
|
|
64
|
+
function stopReasonOf(v) { return typeof v === 'string' && v ? v : null; }
|
|
65
|
+
|
|
45
66
|
// The empty/unknown usage record. Deliberately null (not zeros) so "the surface
|
|
46
67
|
// did not tell us" never reads as "the call cost nothing".
|
|
47
68
|
function emptyUsage() {
|
|
@@ -75,6 +96,9 @@ function parseClaudeCliJson(stdout) {
|
|
|
75
96
|
text: typeof j.result === 'string' ? j.result : '',
|
|
76
97
|
usage,
|
|
77
98
|
isError: j.is_error === true,
|
|
99
|
+
// v0.6: the CLI's own stop reason and the models it says served the call.
|
|
100
|
+
stopReason: stopReasonOf(j.stop_reason),
|
|
101
|
+
reportedModels: reportedModelsOf(j.modelUsage),
|
|
78
102
|
// The CLI reports its own dollar figure. Recorded here for completeness but
|
|
79
103
|
// NOT used: costs are computed uniformly from the frozen pricing snapshot so
|
|
80
104
|
// three substrates are on one basis (see lib/value.js).
|
|
@@ -85,7 +109,9 @@ function parseClaudeCliJson(stdout) {
|
|
|
85
109
|
// ── openai/cli — `codex exec --json` JSONL event stream ───────────────────────
|
|
86
110
|
// One JSON object per line. Usage rides the terminal `turn.completed` event;
|
|
87
111
|
// input_tokens already INCLUDES cached_input_tokens. Unparseable lines are
|
|
88
|
-
// skipped (the stream also carries progress events we do not model).
|
|
112
|
+
// skipped (the stream also carries progress events we do not model). The
|
|
113
|
+
// stream names neither the model that served the turn nor a stop reason, so
|
|
114
|
+
// both read null here and the receipt says the surface did not report them.
|
|
89
115
|
function parseCodexJsonl(stdout) {
|
|
90
116
|
const lines = String(stdout || '').split('\n').map((l) => l.trim()).filter(Boolean);
|
|
91
117
|
let usage = null;
|
|
@@ -103,16 +129,30 @@ function parseCodexJsonl(stdout) {
|
|
|
103
129
|
wall_ms: null,
|
|
104
130
|
};
|
|
105
131
|
}
|
|
106
|
-
// The final message
|
|
132
|
+
// The final message IS the stream: the last `item.completed` /
|
|
133
|
+
// `agent_message` event carries it. Nothing is read from a file after the
|
|
134
|
+
// call, on either spawn mode (spec 022 AC-9; the earlier comment here
|
|
135
|
+
// described an output file the lane no longer reads, F-022-3).
|
|
107
136
|
if (ev.type === 'item.completed' && ev.item && ev.item.type === 'agent_message' && typeof ev.item.text === 'string') {
|
|
108
137
|
text = ev.item.text;
|
|
109
138
|
}
|
|
110
139
|
}
|
|
111
|
-
return { usage, text };
|
|
140
|
+
return { usage, text, stopReason: null, reportedModels: null };
|
|
112
141
|
}
|
|
113
142
|
|
|
114
143
|
// ── anthropic/api ─────────────────────────────────────────────────────────────
|
|
115
144
|
// The Messages API reports input_tokens EXCLUDING cache, same as the CLI.
|
|
145
|
+
// The response object also names the model that served it (`model`) and why
|
|
146
|
+
// it stopped (`stop_reason`); read here from the object, null when absent.
|
|
147
|
+
// Exercised against tests/fixtures/response-anthropic-api.json, which is
|
|
148
|
+
// SYNTHESISED from the API reference, not captured from a call.
|
|
149
|
+
function readAnthropicApiResponse(resp) {
|
|
150
|
+
const r = resp && typeof resp === 'object' ? resp : {};
|
|
151
|
+
return {
|
|
152
|
+
stopReason: stopReasonOf(r.stop_reason),
|
|
153
|
+
reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
|
|
154
|
+
};
|
|
155
|
+
}
|
|
116
156
|
function parseAnthropicApiUsage(u) {
|
|
117
157
|
if (!u) return emptyUsage();
|
|
118
158
|
const cacheRead = n(u.cache_read_input_tokens);
|
|
@@ -127,7 +167,18 @@ function parseAnthropicApiUsage(u) {
|
|
|
127
167
|
|
|
128
168
|
// ── openai/api (Chat Completions-compatible) ──────────────────────────────────
|
|
129
169
|
// prompt_tokens is the total; the cached portion, when present, is nested under
|
|
130
|
-
// prompt_tokens_details.cached_tokens.
|
|
170
|
+
// prompt_tokens_details.cached_tokens. The response also names the serving
|
|
171
|
+
// model (`model`) and the first choice's `finish_reason`; read here, null
|
|
172
|
+
// when absent. Exercised against tests/fixtures/response-openai-api.json,
|
|
173
|
+
// SYNTHESISED from the API reference, not captured from a call.
|
|
174
|
+
function readOpenaiApiResponse(resp) {
|
|
175
|
+
const r = resp && typeof resp === 'object' ? resp : {};
|
|
176
|
+
const choice = Array.isArray(r.choices) && r.choices[0] && typeof r.choices[0] === 'object' ? r.choices[0] : {};
|
|
177
|
+
return {
|
|
178
|
+
stopReason: stopReasonOf(choice.finish_reason),
|
|
179
|
+
reportedModels: typeof r.model === 'string' && r.model ? [r.model] : null,
|
|
180
|
+
};
|
|
181
|
+
}
|
|
131
182
|
function parseOpenaiApiUsage(u) {
|
|
132
183
|
if (!u) return emptyUsage();
|
|
133
184
|
const details = u.prompt_tokens_details || {};
|
|
@@ -165,4 +216,5 @@ function normalizeUsage(u) {
|
|
|
165
216
|
module.exports = {
|
|
166
217
|
emptyUsage, hasUsage, sumUsage, normalizeUsage,
|
|
167
218
|
parseClaudeCliJson, parseCodexJsonl, parseAnthropicApiUsage, parseOpenaiApiUsage,
|
|
219
|
+
readAnthropicApiResponse, readOpenaiApiResponse, canonicalModelId, reportedModelsOf,
|
|
168
220
|
};
|
package/lib/value.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
// SPDX-License-Identifier: Apache-2.0
|
|
2
2
|
'use strict';
|
|
3
3
|
|
|
4
|
+
const { caseFailed } = require('./receipt');
|
|
5
|
+
|
|
4
6
|
const { EFFECT_FLOOR } = require('../config');
|
|
5
7
|
|
|
6
8
|
// Economics of a run — what a skill COSTS to run, alongside whether it helps.
|
|
@@ -204,7 +206,7 @@ function armEconomics(rows, price) {
|
|
|
204
206
|
// surface the run surface, to label the cost basis honestly
|
|
205
207
|
function computeEconomics({ cases, modelId, judgeModelId, pricingSnapshot, surface, meteredSurface }) {
|
|
206
208
|
const price = (pricingSnapshot && pricingSnapshot.models && pricingSnapshot.models[modelId]) || null;
|
|
207
|
-
const ok = (cases || []).filter((c) => c
|
|
209
|
+
const ok = (cases || []).filter((c) => !caseFailed(c));
|
|
208
210
|
const withRows = ok.filter((c) => c.mode === 'with_skill');
|
|
209
211
|
const baseRows = ok.filter((c) => c.mode === 'baseline');
|
|
210
212
|
|
|
@@ -322,7 +324,7 @@ function receiptCostBreakdown(receipt) {
|
|
|
322
324
|
let judge = 0;
|
|
323
325
|
let missingJudgeRate = null;
|
|
324
326
|
for (const c of ((receipt.results || {}).cases || [])) {
|
|
325
|
-
if (c
|
|
327
|
+
if (caseFailed(c)) continue;
|
|
326
328
|
const g = costForUsage(c.usage, genRate);
|
|
327
329
|
if (g != null) generation += g;
|
|
328
330
|
if (c.judge_usage) {
|
package/lib/verdict.js
CHANGED
|
@@ -43,7 +43,12 @@ function verdictFromReceipt(receipt) {
|
|
|
43
43
|
// Below-TESTED receipts (imported/DECLARED) and null-delta receipts are never
|
|
44
44
|
// verdicted — we did not run the suite, so we do not certify the outcome.
|
|
45
45
|
const level = (receipt && receipt.verification_level) || 'TESTED';
|
|
46
|
-
|
|
46
|
+
// Spec 026 AC-4: an INCOMPLETE receipt (its run block's status field reads
|
|
47
|
+
// incomplete: a case had an arm that could not be measured) is not verdicted
|
|
48
|
+
// either. spec/RECEIPT.md has said since v0.3.1 that a drift report must not
|
|
49
|
+
// compute a verdict from one; the badge is the same reader with a shorter path.
|
|
50
|
+
const incomplete = !!(receipt && receipt.run && receipt.run.status === 'incomplete');
|
|
51
|
+
const measured = level === 'TESTED' && typeof cmp.delta === 'number' && !incomplete;
|
|
47
52
|
let verdict;
|
|
48
53
|
if (!measured) verdict = 'NOT_MEASURED';
|
|
49
54
|
else if (delta >= EFFECT_FLOOR) verdict = 'PASSED';
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "driftproof",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "A dated proof that this skill, this hash, this model, still helps: run a skill's eval suite with and without the skill across model versions, emit hash-verified dated receipts, and diff receipts into drift reports.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"keywords": [
|
|
@@ -45,6 +45,7 @@
|
|
|
45
45
|
"node": ">=22"
|
|
46
46
|
},
|
|
47
47
|
"devDependencies": {
|
|
48
|
-
"html-validate": "^11.11.0"
|
|
48
|
+
"html-validate": "^11.11.0",
|
|
49
|
+
"playwright": "^1.62.1"
|
|
49
50
|
}
|
|
50
51
|
}
|
package/spec/RECEIPT.md
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
<!-- SPDX-License-Identifier: Apache-2.0 -->
|
|
2
|
-
# Driftproof receipt — spec v0.
|
|
2
|
+
# Driftproof receipt — spec v0.6
|
|
3
3
|
|
|
4
4
|
A **receipt** is a **hash-verified**, dated record of running one agent skill's eval suite
|
|
5
5
|
**with** and **without** the skill on one model version, with the judge **sampled**
|
|
@@ -11,16 +11,64 @@ The machine-readable contract is [`receipt.schema.json`](./receipt.schema.json)
|
|
|
11
11
|
(JSON Schema, draft 2020-12). This document is the human companion. Where they
|
|
12
12
|
disagree, the schema wins.
|
|
13
13
|
|
|
14
|
-
**Versioning.** The current schema is v0.
|
|
14
|
+
**Versioning.** The current schema is v0.6
|
|
15
15
|
([`receipt.schema.json`](./receipt.schema.json)). Prior schemas are kept frozen as
|
|
16
|
+
[`receipt.v0.5.schema.json`](./receipt.v0.5.schema.json),
|
|
17
|
+
[`receipt.v0.4.schema.json`](./receipt.v0.4.schema.json),
|
|
16
18
|
[`receipt.v0.3.1.schema.json`](./receipt.v0.3.1.schema.json),
|
|
17
19
|
[`receipt.v0.3.schema.json`](./receipt.v0.3.schema.json),
|
|
18
20
|
[`receipt.v0.2.schema.json`](./receipt.v0.2.schema.json) and
|
|
19
21
|
[`receipt.v0.1.schema.json`](./receipt.v0.1.schema.json); the validator picks the
|
|
20
|
-
schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3
|
|
21
|
-
receipts still load and validate — including every receipt behind the
|
|
22
|
-
published reports, which the gate asserts on each run. v0.
|
|
23
|
-
|
|
22
|
+
schema by the receipt's own `schema_version`, so v0.1, v0.2, v0.3, v0.3.1, v0.4
|
|
23
|
+
and v0.5 receipts still load and validate — including every receipt behind the
|
|
24
|
+
published reports, which the gate asserts on each run. v0.6 is additive for a
|
|
25
|
+
reader that ignores unknown fields and **not additive for the validator**: a
|
|
26
|
+
v0.5 receipt restamped `"0.6"` is refused, because v0.6 requires the receipt to
|
|
27
|
+
say what answered it (see *What v0.6 adds*). No earlier receipt is invalidated;
|
|
28
|
+
each validates against its own frozen schema.
|
|
29
|
+
|
|
30
|
+
## What v0.6 adds: the receipt says what answered it
|
|
31
|
+
|
|
32
|
+
Spec 026 (receipt integrity). Six ways the v0.5 runner wrote a receipt that read
|
|
33
|
+
as more than it was, each closed by a field the receipt now carries and the
|
|
34
|
+
schema now requires:
|
|
35
|
+
|
|
36
|
+
- **`run.answered_by`** `{ kind, attested, reported_model, reported_models,
|
|
37
|
+
isolation }`. `kind` is `model` (a model surface answered), `stub` (the
|
|
38
|
+
canned stub answered and nothing was measured) or `external` (another tool's
|
|
39
|
+
harness answered; the receipt was imported). `attested` is true only when the
|
|
40
|
+
surface echoed a model id for every draw and it matched the requested
|
|
41
|
+
canonical id; `reported_model` and `reported_models` are what it echoed
|
|
42
|
+
(null when nothing). `isolation` names the spawn path: `eval-user`,
|
|
43
|
+
`same-user` (`--trusted-skill`) or `none`. A stub run also records
|
|
44
|
+
`run.surface: "stub"` and `run.judge.surface: "stub"`, and the schema refuses
|
|
45
|
+
`TESTED` unless `answered_by.kind` is `model`: a stub receipt is
|
|
46
|
+
`UNVERIFIED` by schema, not only by the runner.
|
|
47
|
+
- **`run.judge.model_id`** and **`run.judge.prompt_template_hash`**: which
|
|
48
|
+
judge ran, and a sha256 over the grading template with its three slots empty.
|
|
49
|
+
A different template is a different judge; `diff` names the field and
|
|
50
|
+
computes no verdict across one. The template hash is null only on an
|
|
51
|
+
imported receipt.
|
|
52
|
+
- **`stop_reason`** and **`truncated`** on every draw, and
|
|
53
|
+
**`generation.n_truncated`** per case and arm. A draw cut at the output cap
|
|
54
|
+
is unmeasured and never judged; a partial answer scored as a whole one was
|
|
55
|
+
the v0.5 shape.
|
|
56
|
+
- **`case_status: "failed_unmeasured"`**: every draw of the arm was unmeasured
|
|
57
|
+
for a reason that is not a timeout (an empty generation, a judge output with
|
|
58
|
+
no score, a truncated draw). Recorded without fabricated samples or hashes
|
|
59
|
+
exactly as `failed_timeout` is, and excluded pairwise. An absent output is
|
|
60
|
+
not a measurement; a zero is.
|
|
61
|
+
- **`results.aggregates.band_rule`**: the formula the aggregate band is derived
|
|
62
|
+
by, stated on the receipt so a reader recomputes it from `results.cases`.
|
|
63
|
+
- **Bands the formula cannot form are null, never `0`.** An arm's `stddev` is
|
|
64
|
+
null only when its `case_count` is below 2 (the schema refuses a null on a
|
|
65
|
+
two-case arm); `comparison.delta_uncertainty` is null only beside
|
|
66
|
+
`delta_uncertainty_unavailable: "single_case"` or `"no_cases"`, and
|
|
67
|
+
`mean_score` and the comparison are null only under `no_cases`.
|
|
68
|
+
|
|
69
|
+
A v0.5 receipt carries none of these, so the validator refuses one restamped
|
|
70
|
+
`"0.6"`; v0.5 stays frozen at `receipt.v0.5.schema.json` and every archived
|
|
71
|
+
receipt validates against the version it names.
|
|
24
72
|
|
|
25
73
|
## What v0.5 adds: the generation is sampled too
|
|
26
74
|
|
|
@@ -95,7 +143,7 @@ carry a numeric `mean`.
|
|
|
95
143
|
## Which schema validates which receipt (the coexistence rule)
|
|
96
144
|
|
|
97
145
|
- **A producer emits the current version.** `config.js` `RECEIPT_SCHEMA_VERSION`
|
|
98
|
-
decides it, and at this revision that is **v0.
|
|
146
|
+
decides it, and at this revision that is **v0.6**.
|
|
99
147
|
- **A reader validates against the receipt's own `schema_version`**, never
|
|
100
148
|
against the newest schema it happens to have. `validateReceipt()` selects the
|
|
101
149
|
schema by that field, which is why a v0.1 receipt from the first report still
|
|
@@ -247,7 +295,11 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
|
|
|
247
295
|
outputs were also written to `transcripts/<receipt-id>/` (under
|
|
248
296
|
`--keep-transcripts`), else `"hashes-only"` (the default — only the hashes).
|
|
249
297
|
- **`run.registry`** — `"registered"` when `model_id` resolved in the model
|
|
250
|
-
registry (`config/models.json`)
|
|
298
|
+
registry (`config/models.json`); `"unregistered"` is import-only since v0.6, the runner refuses it (spec 026 AC-10):
|
|
299
|
+
a run of ours refuses an id the registry does not carry before any call,
|
|
300
|
+
naming the registry path, so no receipt we write says it; an imported
|
|
301
|
+
receipt may, its model field being another tool's word (before v0.6 the
|
|
302
|
+
run was not refused;
|
|
251
303
|
cost was estimated with a conservative default price).
|
|
252
304
|
- These four fields are **required** in a v0.3 receipt. Everything else is
|
|
253
305
|
unchanged from v0.2.
|
|
@@ -263,7 +315,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
|
|
|
263
315
|
|
|
264
316
|
## Fields
|
|
265
317
|
|
|
266
|
-
### `schema_version` (string, required) — `"0.
|
|
318
|
+
### `schema_version` (string, required) — `"0.6"`.
|
|
267
319
|
|
|
268
320
|
### `skill` (object, required)
|
|
269
321
|
| field | type | notes |
|
|
@@ -290,7 +342,7 @@ The v0.3.1 schema gained an **additive interop revision** so receipts can be
|
|
|
290
342
|
| `surface_overhead_note` | string | **v0.3.1, optional.** On `openai-cli`: the fixed Codex base-instruction preamble (~12–15k input tokens/call) the harness prepends and does not control. |
|
|
291
343
|
| `runner_version` | string | Runner version, for reproducibility. |
|
|
292
344
|
| `date_utc` | ISO 8601 UTC | When the run finished. |
|
|
293
|
-
| `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). |
|
|
345
|
+
| `registry` | `"registered"` \| `"unregistered"` | **v0.3.** Whether `model_id` resolved in the model registry (`config/models.json`). **v0.6:** `"unregistered"` is import-only; the runner refuses an unregistered id before any call (spec 026 AC-10). |
|
|
294
346
|
| `transcripts` | `"retained-local"` \| `"hashes-only"` | **v0.3.** Whether the raw generations + judge outputs were retained on disk (see § Transcripts). |
|
|
295
347
|
| `judge` | object | `{ samples, temperature, sampling, surface }` — see below. |
|
|
296
348
|
|
|
@@ -358,7 +410,12 @@ Every generation and every judge sample is hashed into the receipt
|
|
|
358
410
|
generations and judge outputs are written to `transcripts/<receipt_hash>/` (one
|
|
359
411
|
JSON per case+mode, plus an `index.json`), and the receipt records
|
|
360
412
|
`transcripts: "retained-local"`. Without the flag the receipt records
|
|
361
|
-
`transcripts: "hashes-only"` and only the hashes are kept.
|
|
413
|
+
`transcripts: "hashes-only"` and only the hashes are kept. **What is retained
|
|
414
|
+
is the last measured draw of each case and mode only**: the runner keeps one
|
|
415
|
+
transcript per case and mode and overwrites it on every draw, so the per-case
|
|
416
|
+
top-level hashes (which are that draw's) can be re-derived from disk, and the
|
|
417
|
+
draw-level hashes of earlier draws are checkable against nothing the runner
|
|
418
|
+
wrote (spec 026, stated as the bound rather than moved). The transcript
|
|
362
419
|
directory is **gitignored by default** — raw model text is never committed. The
|
|
363
420
|
default for trigger-initiated runs is `retained-local` (disk is cheap, audits are
|
|
364
421
|
not). To verify a retained receipt: re-hash each transcript file and compare
|